{
  "snapshotAt": "2026-09-15T22:05:24.400Z",
  "pricing": {
    "currency": "USD",
    "tokenUnit": 1000000,
    "source": "Platform standard model catalog",
    "snapshotAt": "2026-09-15T22:05:24.400Z",
    "comparison": "input tokens × input price / 1,000,000 + output tokens × output price / 1,000,000",
    "exclusions": "Provider reference pricing; excludes AMLR plan charges, taxes, cache/batch discounts and context-specific tiers. Not the historic benchmark bill."
  },
  "reportDate": "2026-09-10",
  "judge": "claude-fable-5-1",
  "firstRun": "2026-09-09T16:34:27.057Z",
  "lastRun": "2026-09-09T17:47:41.064Z",
  "completedRuns": 468,
  "excludedRuns": 11,
  "casesPerStage": {
    "input": 12,
    "understanding": 12,
    "generation": 10,
    "verification": 18
  },
  "models": [
    {
      "id": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "inputUsdPerMillion": 10,
      "outputUsdPerMillion": 50,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 91.7,
          "latencySeconds": 30.2119
        },
        "input": {
          "runs": 12,
          "score": 97.16666666666667,
          "latencySeconds": 13.033083333333332
        },
        "understanding": {
          "runs": 12,
          "score": 100,
          "latencySeconds": 21.18283333333333
        },
        "verification": {
          "runs": 18,
          "score": 100,
          "latencySeconds": 21.232555555555553
        }
      },
      "averageScore": 97.21666666666667,
      "averageLatencySeconds": 21.055692307692308
    },
    {
      "id": "claude-opus-5",
      "name": "Claude Opus 5",
      "inputUsdPerMillion": 5,
      "outputUsdPerMillion": 25,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 89.2,
          "latencySeconds": 36.5946
        },
        "input": {
          "runs": 12,
          "score": 100,
          "latencySeconds": 19.103416666666668
        },
        "understanding": {
          "runs": 12,
          "score": 91.66666666666666,
          "latencySeconds": 33.49625
        },
        "verification": {
          "runs": 18,
          "score": 100,
          "latencySeconds": 20.804944444444445
        }
      },
      "averageScore": 95.21666666666667,
      "averageLatencySeconds": 26.37751923076923
    },
    {
      "id": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "inputUsdPerMillion": 2,
      "outputUsdPerMillion": 10,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 94.19999999999999,
          "latencySeconds": 12.966
        },
        "input": {
          "runs": 12,
          "score": 97.16666666666667,
          "latencySeconds": 6.6265
        },
        "understanding": {
          "runs": 12,
          "score": 97.16666666666667,
          "latencySeconds": 13.914166666666668
        },
        "verification": {
          "runs": 18,
          "score": 99.55555555555556,
          "latencySeconds": 12.806388888888888
        }
      },
      "averageScore": 97.02222222222223,
      "averageLatencySeconds": 11.666596153846154
    },
    {
      "id": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash-Lite",
      "inputUsdPerMillion": 0.25,
      "outputUsdPerMillion": 1.5,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 92.5,
          "latencySeconds": 3.2769
        },
        "input": {
          "runs": 12,
          "score": 84.66666666666667,
          "latencySeconds": 1.4345
        },
        "understanding": {
          "runs": 12,
          "score": 95.83333333333334,
          "latencySeconds": 1.7365
        },
        "verification": {
          "runs": 18,
          "score": 97.5,
          "latencySeconds": 1.8788333333333334
        }
      },
      "averageScore": 92.625,
      "averageLatencySeconds": 2.0123076923076924
    },
    {
      "id": "gemini-3.8-flash",
      "name": "Gemini 3.8 Flash",
      "inputUsdPerMillion": 0.75,
      "outputUsdPerMillion": 3.75,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 95,
          "latencySeconds": 8.3553
        },
        "input": {
          "runs": 12,
          "score": 89.91666666666667,
          "latencySeconds": 4.03925
        },
        "understanding": {
          "runs": 12,
          "score": 91.08333333333334,
          "latencySeconds": 3.2429166666666664
        },
        "verification": {
          "runs": 18,
          "score": 100,
          "latencySeconds": 4.201222222222222
        }
      },
      "averageScore": 94,
      "averageLatencySeconds": 4.741557692307692
    },
    {
      "id": "gpt-5.6-luna",
      "name": "GPT-5.6 Luna",
      "inputUsdPerMillion": 0.2,
      "outputUsdPerMillion": 1.2,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 100,
          "latencySeconds": 10.411299999999999
        },
        "input": {
          "runs": 12,
          "score": 94.41666666666667,
          "latencySeconds": 4.0456666666666665
        },
        "understanding": {
          "runs": 12,
          "score": 92.41666666666667,
          "latencySeconds": 5.8045
        },
        "verification": {
          "runs": 18,
          "score": 97.88888888888889,
          "latencySeconds": 6.193722222222223
        }
      },
      "averageScore": 96.18055555555557,
      "averageLatencySeconds": 6.419269230769231
    },
    {
      "id": "gpt-5.6-sol",
      "name": "GPT-5.6 Sol",
      "inputUsdPerMillion": 4,
      "outputUsdPerMillion": 20,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 95,
          "latencySeconds": 15.0445
        },
        "input": {
          "runs": 12,
          "score": 95.83333333333334,
          "latencySeconds": 6.704333333333333
        },
        "understanding": {
          "runs": 12,
          "score": 89.58333333333334,
          "latencySeconds": 8.910083333333334
        },
        "verification": {
          "runs": 18,
          "score": 100,
          "latencySeconds": 10.918944444444444
        }
      },
      "averageScore": 95.10416666666667,
      "averageLatencySeconds": 10.276134615384613
    },
    {
      "id": "gpt-5.6-terra",
      "name": "GPT-5.6 Terra",
      "inputUsdPerMillion": 2,
      "outputUsdPerMillion": 12,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 100,
          "latencySeconds": 17.707900000000002
        },
        "input": {
          "runs": 12,
          "score": 93,
          "latencySeconds": 5.439083333333333
        },
        "understanding": {
          "runs": 12,
          "score": 92.66666666666666,
          "latencySeconds": 8.6135
        },
        "verification": {
          "runs": 18,
          "score": 97.77777777777777,
          "latencySeconds": 10.20861111111111
        }
      },
      "averageScore": 95.8611111111111,
      "averageLatencySeconds": 10.182019230769232
    },
    {
      "id": "gpt-6-astra",
      "name": "GPT-6 Astra",
      "inputUsdPerMillion": 10,
      "outputUsdPerMillion": 50,
      "stages": {
        "generation": {
          "runs": 10,
          "score": 92.5,
          "latencySeconds": 19.1843
        },
        "input": {
          "runs": 12,
          "score": 96.91666666666666,
          "latencySeconds": 10.569083333333333
        },
        "understanding": {
          "runs": 12,
          "score": 98.58333333333333,
          "latencySeconds": 14.541333333333332
        },
        "verification": {
          "runs": 18,
          "score": 96.66666666666667,
          "latencySeconds": 12.406944444444443
        }
      },
      "averageScore": 96.16666666666667,
      "averageLatencySeconds": 13.778711538461538
    }
  ],
  "methodology": "AICLUDE internal pipeline benchmark, round 1. Mean assertion pass_rate × 100, with partial credit; not end-to-end task success. Score averages weight each stage equally; overall latency averages all completed runs. Completed runs from the corrected round-one batches only; failed attempts and superseded batches excluded. Small samples of 10–18 cases per stage; 1–2 point differences should not be treated as meaningful rankings. Latency measures individual stage calls, not a whole pipeline. No local-model benchmark is included."
}
