{
  "schemaVersion": 2,
  "generatedAtUtc": "2026-09-25T08:31:37.364304+00:00",
  "asOfUtc": "2026-09-25T00:00:00Z",
  "source": "ModelEfficiencyMatrix.Default.Describe(asOfUtc)",
  "priceCatalogPath": "src/TokenEconomy/catalog/model-prices.json",
  "routingPolicyPath": "src/TokenEconomy/catalog/model-routing-policy.json",
  "rows": [
    {
      "displayName": "GPT-5.6 Luna",
      "releaseDate": "2026-06-26",
      "releaseDateSource": "https://openai.com/index/previewing-gpt-5-6-sol/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "No Luna cohort existed in the 2026-07-23 benchmark; this is an unvalidated cost-saving hypothesis.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-gpt-5.6-luna-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 17.27,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/51c6d76e-5baa-48c1-97b3-88656c620eef",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 185705,
              "costPerTaskUsd": 1.050515,
              "latencyMilliseconds": 4088300
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:06.717559+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.85,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 346.67,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "57/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 57/330 attempts resolved (17.27%, 95% CI half-width 2.85 percentage points), using Codex at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 346.67; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "kodus-light-v1-vendor-default-recall-gpt-5.6-luna",
            "benchmarkId": "kodus-light-v1-vendor-default-recall",
            "name": "CodeReviewBench light-v1 / micro recall",
            "version": "light-v1 / kodus dev / vendor defaults / scored 2026-08-13",
            "unit": "known-issue recall ratio",
            "maximumScore": 1,
            "effort": "unspecified",
            "score": 0.29473684210526313,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://github.com/kodustech/codereviewbench/blob/531297bf50e5f065e3888d7e07b55dcacbf8df64/scorecards/gpt-5.6-luna.json",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "Kodus",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-12",
              "sourcePublisher": "Kodus",
              "harness": "Kodus deterministic tool replay",
              "harnessVersion": "dev",
              "taskCount": 30,
              "sampleNotes": "Run 2026-08-04T19:56:24.224Z; scored 2026-08-13T23:28:51.169Z. One replay per model via codex_subscription; effortRequested is null (vendor default), not a known model effort. First observed public snapshot is September 12. 152/1160 replay tool requests were unserved; coverage is limited by the replay. Judge model claude-haiku-4-5.",
              "additionalSourceUrls": [
                "https://www.codereviewbench.com/"
              ]
            },
            "evidenceExcerpt": "Pinned light-v1 scorecard: 28/95 distinct known bugs matched; 30/52 findings classified true by Haiku 4.5, 22 classified false. These are different issue and finding denominators. Original scorecard recallMicro ratio retained."
          },
          {
            "id": "kodus-light-v1-vendor-default-precision-gpt-5.6-luna",
            "benchmarkId": "kodus-light-v1-vendor-default-precision",
            "name": "CodeReviewBench light-v1 / micro precision",
            "version": "light-v1 / kodus dev / vendor defaults / scored 2026-08-13",
            "unit": "finding precision ratio",
            "maximumScore": 1,
            "effort": "unspecified",
            "score": 0.5769230769230769,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://github.com/kodustech/codereviewbench/blob/531297bf50e5f065e3888d7e07b55dcacbf8df64/scorecards/gpt-5.6-luna.json",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "Kodus",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-12",
              "sourcePublisher": "Kodus",
              "harness": "Kodus deterministic tool replay",
              "harnessVersion": "dev",
              "taskCount": 30,
              "sampleNotes": "Run 2026-08-04T19:56:24.224Z; scored 2026-08-13T23:28:51.169Z. One replay per model via codex_subscription; effortRequested is null (vendor default), not a known model effort. First observed public snapshot is September 12. 152/1160 replay tool requests were unserved; coverage is limited by the replay. Judge model claude-haiku-4-5.",
              "additionalSourceUrls": [
                "https://www.codereviewbench.com/"
              ]
            },
            "evidenceExcerpt": "Pinned light-v1 scorecard: 28/95 distinct known bugs matched; 30/52 findings classified true by Haiku 4.5, 22 classified false. These are different issue and finding denominators. Original scorecard precisionMicro ratio retained."
          }
        ],
        "taskStudies": [
          {
            "id": "mechanical-chore",
            "label": "Mechanical chore (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; Luna has no qualifying cohort.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "doc-edit",
            "label": "Documentation edit (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; canonical policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.6-luna",
      "vendor": "openai",
      "cli": "codex",
      "tier": "light",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 0.2,
        "outputPerMTok": 1.2,
        "cachedInputPerMTok": 0.02,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-07-30T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh",
        "ultra"
      ],
      "suitability": {
        "heavyDesign": "underpowered",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "underpowered",
        "mechanicalChore": "ideal",
        "docEdit": "ideal",
        "research": "underpowered",
        "review": null,
        "htmlUiImplementation": "underpowered",
        "sourceCodeReview": "underpowered",
        "securityAssessment": "underpowered",
        "redundancyDetection": "underpowered",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-5.6 Terra",
      "releaseDate": "2026-06-26",
      "releaseDateSource": "https://openai.com/index/previewing-gpt-5-6-sol/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Eight historical Terra/medium records had no known grades and no trustworthy terminal cohort.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [
          {
            "effort": "medium",
            "role": "coreTask",
            "minimumTaskScore": 21,
            "maximumTaskScore": 50
          }
        ],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-gpt-5.6-terra-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 21.52,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/e53da412-5e92-408c-b369-c767924c7c1c",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 100681,
              "costPerTaskUsd": 5.253091,
              "latencyMilliseconds": 2514600
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:20.912429+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.25,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 1733.52,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "71/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 71/330 attempts resolved (21.52%, 95% CI half-width 3.25 percentage points), using Codex at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 1733.52; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "kodus-light-v1-vendor-default-recall-gpt-5.6-terra",
            "benchmarkId": "kodus-light-v1-vendor-default-recall",
            "name": "CodeReviewBench light-v1 / micro recall",
            "version": "light-v1 / kodus dev / vendor defaults / scored 2026-08-13",
            "unit": "known-issue recall ratio",
            "maximumScore": 1,
            "effort": "unspecified",
            "score": 0.23157894736842105,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://github.com/kodustech/codereviewbench/blob/531297bf50e5f065e3888d7e07b55dcacbf8df64/scorecards/gpt-5.6-terra.json",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "Kodus",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-12",
              "sourcePublisher": "Kodus",
              "harness": "Kodus deterministic tool replay",
              "harnessVersion": "dev",
              "taskCount": 30,
              "sampleNotes": "Run 2026-08-04T20:50:48.271Z; scored 2026-08-13T23:30:03.958Z. One replay per model via codex_subscription; effortRequested is null (vendor default), not a known model effort. First observed public snapshot is September 12. 294/1606 replay tool requests were unserved; coverage is limited by the replay. Judge model claude-haiku-4-5.",
              "additionalSourceUrls": [
                "https://www.codereviewbench.com/"
              ]
            },
            "evidenceExcerpt": "Pinned light-v1 scorecard: 22/95 distinct known bugs matched; 22/50 findings classified true by Haiku 4.5, 28 classified false. These are different issue and finding denominators. Original scorecard recallMicro ratio retained."
          },
          {
            "id": "kodus-light-v1-vendor-default-precision-gpt-5.6-terra",
            "benchmarkId": "kodus-light-v1-vendor-default-precision",
            "name": "CodeReviewBench light-v1 / micro precision",
            "version": "light-v1 / kodus dev / vendor defaults / scored 2026-08-13",
            "unit": "finding precision ratio",
            "maximumScore": 1,
            "effort": "unspecified",
            "score": 0.44,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://github.com/kodustech/codereviewbench/blob/531297bf50e5f065e3888d7e07b55dcacbf8df64/scorecards/gpt-5.6-terra.json",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "Kodus",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-12",
              "sourcePublisher": "Kodus",
              "harness": "Kodus deterministic tool replay",
              "harnessVersion": "dev",
              "taskCount": 30,
              "sampleNotes": "Run 2026-08-04T20:50:48.271Z; scored 2026-08-13T23:30:03.958Z. One replay per model via codex_subscription; effortRequested is null (vendor default), not a known model effort. First observed public snapshot is September 12. 294/1606 replay tool requests were unserved; coverage is limited by the replay. Judge model claude-haiku-4-5.",
              "additionalSourceUrls": [
                "https://www.codereviewbench.com/"
              ]
            },
            "evidenceExcerpt": "Pinned light-v1 scorecard: 22/95 distinct known bugs matched; 22/50 findings classified true by Haiku 4.5, 28 classified false. These are different issue and finding denominators. Original scorecard precisionMicro ratio retained."
          },
          {
            "id": "token-economy-palindrome-terra-medium-20260809",
            "benchmarkId": "token-economy-controlled-setups-v1",
            "name": "Token Economy controlled setups",
            "version": "1",
            "unit": "percent successful attempts",
            "maximumScore": 100,
            "effort": "medium",
            "score": 100,
            "publishedAt": "2026-08-09",
            "sourceUrl": "https://github.com/agent-orc/token-economy/blob/main/benchmarks/results/palindrome-repair/20260809T092240107Z.report.json",
            "confidence": "ownRun",
            "evidenceStatus": "unknown",
            "sampleContext": "palindrome-repair: terra-medium passed 3 of 3 isolated attempts (100%).",
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "palindrome-repair: terra-medium passed 3 of 3 isolated attempts (100%)."
          }
        ],
        "taskStudies": [
          {
            "id": "feature",
            "label": "Feature and bug implementation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "research",
            "label": "Research and investigation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "html-ui-implementation",
            "label": "HTML/UI implementation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "source-code-review",
            "label": "Source-code review (historical baseline)",
            "status": "controlledPilot",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 2,
            "outcomeRate": 1.0,
            "costPerSuccessfulOutcome": {
              "amountUsd": 0.030434,
              "pricedAtUtc": "2026-08-12T08:29:10Z",
              "includesVerification": true,
              "note": "Catalog list-price estimate for Terra subject usage per passed outcome. Verification is deterministic local scoring and adds no model-token cost."
            },
            "benchmarkStatus": "Controlled two-scenario pilot complete; below the 20-attempt and five-scenario qualification gates.",
            "references": [
              "benchmarks/results/source-code-review-tenant-cache-v2/20260812T082910114Z.json",
              "benchmarks/results/source-code-review-tenant-transfer-v2/20260812T083344557Z.json",
              "benchmarks/task-class-studies/source-code-review-v1.json",
              "results/routing-evidence/review/v1/review-evidence.json"
            ]
          },
          {
            "id": "redundancy-detection",
            "label": "Redundancy detection",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.6-terra",
      "vendor": "openai",
      "cli": "codex",
      "tier": "balanced",
      "costClass": "standard",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 2,
        "outputPerMTok": 12,
        "cachedInputPerMTok": 0.2,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-07-30T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh",
        "ultra"
      ],
      "suitability": {
        "heavyDesign": "capable",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "ideal",
        "mechanicalChore": "capable",
        "docEdit": "capable",
        "research": "ideal",
        "review": null,
        "htmlUiImplementation": "capable",
        "sourceCodeReview": "ideal",
        "securityAssessment": "underpowered",
        "redundancyDetection": "capable",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-5.6 Sol",
      "releaseDate": "2026-06-26",
      "releaseDateSource": "https://openai.com/index/previewing-gpt-5-6-sol/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Sol/medium has the strongest favorable historical signal; Sol/xhigh remains a correctness floor, not a blanket default.",
        "workflowRoles": [
          "coreTask",
          "boundedPipelineDecision"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-gpt-5.6-sol-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 37.27,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/0e349bde-b264-494d-a853-fada9c696192",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 71065,
              "costPerTaskUsd": 7.702121,
              "latencyMilliseconds": 2388400
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:13.502808+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.78,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 2541.7,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "123/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 123/330 attempts resolved (37.27%, 95% CI half-width 3.78 percentage points), using Codex at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 2541.7; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "deepswe-owner-gpt-5.6-sol-max-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 73,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 60000,
              "costPerTaskUsd": 6.46
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 3,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 73% with reported +/-3 percentage points; 113 tasks; average cost USD 6.46, displayed output 60k and 61 steps. These are rounded leaderboard measurements, not a new local evaluation."
          },
          {
            "id": "aa-ii-v4.3-sol-max-2026-09-11",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 47,
            "publishedAt": "2026-09-11",
            "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-low-vs-gpt-5-6-sol",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 29000,
              "costPerTaskUsd": 1.99
            },
            "context": null,
            "evidenceExcerpt": "GPT-5.6 Sol (max): Intelligence Index 47; cost per task $1.99; output tokens per task 29k."
          },
          {
            "id": "aa-ii-v4.3-sol-high-2026-09-11",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "high",
            "score": 42,
            "publishedAt": "2026-09-11",
            "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-low-vs-gpt-5-6-sol-high",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 13000,
              "costPerTaskUsd": 0.81,
              "latencyMilliseconds": 41690
            },
            "context": null,
            "evidenceExcerpt": "Intelligence Index: 46 vs 42; cost per task: $0.82 vs $0.81; output tokens: 4k vs 13k."
          },
          {
            "id": "deepswe-v1-aa-sol-max",
            "benchmarkId": "deepswe-v1-aa-harness",
            "name": "DeepSWE",
            "version": "v1, Artificial Analysis harness",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 72,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Its lead comes from other components, partly offset by a lower score on DeepSWE: 68% vs 72%."
          },
          {
            "id": "aa-coding-v15-gpt-5.6-sol-max-2026-09-09",
            "benchmarkId": "artificial-analysis-coding-agent-index-v1.5-native-agents",
            "name": "Artificial Analysis Coding Agent Index 1.5",
            "version": "1.5 / September 9 comparison",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 55,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Codex",
              "taskCount": 303,
              "trialCount": 909,
              "attemptsPerTask": 3,
              "sampleNotes": "Scores are equal-weight component aggregates, not 909-trial pooled accuracy. Native-agent versions are not supplied in the article. No new composite was calculated here.",
              "additionalSourceUrls": [
                "https://artificialanalysis.ai/methodology/coding-agents-benchmarking/"
              ]
            },
            "evidenceExcerpt": "Artificial Analysis September 9 comparison: 55 Coding Agent Index points using Codex at max. Version 1.5 has 303 distinct tasks and 909 attempts across three equally weighted benchmark components."
          },
          {
            "id": "aa-coding-2026-09-09-sol-max",
            "benchmarkId": "artificial-analysis-coding-agent-index-2026-09-09",
            "name": "Artificial Analysis Coding Agent Index",
            "version": "2026-09-09 snapshot",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 55,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "GPT-6 Astra scores 62 in the Index, ahead of GPT-5.6 Sol at 55."
          },
          {
            "id": "aa-ii-v4.2-sol-high-2026-09-07",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.2",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.2",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "high",
            "score": 48,
            "publishedAt": "2026-09-07",
            "sourceUrl": "https://thenewstack.io/astra-reasoning-effort-cost/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Artificial Analysis currently scores Astra-low at 49 on its Intelligence Index, narrowly ahead of Sol-high at 48."
          },
          {
            "id": "coderabbit-astra-2026-09-04-overall-coverage-gpt-5.6-sol",
            "benchmarkId": "coderabbit-astra-2026-09-04-overall-coverage",
            "name": "CodeRabbit / overall actionable bug coverage",
            "version": "2026-09-04 internal evaluation / overall",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 59,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / overall",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "overall actionable bug coverage: 59% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "coderabbit-astra-2026-09-04-cross-file-coverage-gpt-5.6-sol",
            "benchmarkId": "coderabbit-astra-2026-09-04-cross-file-coverage",
            "name": "CodeRabbit / cross-file actionable bug coverage",
            "version": "2026-09-04 internal evaluation / cross-file",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 47.6,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / cross-file",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "cross-file actionable bug coverage: 47.6% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "terminal-bench-v4-openai-sol-unspecified",
            "benchmarkId": "terminal-bench-v4.0",
            "name": "Terminal-Bench",
            "version": "4.0",
            "unit": "percent passed",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 37.3,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Terminal-Bench 4.0: GPT-6 Astra 57.9%; GPT-5.6 Sol 37.3%."
          },
          {
            "id": "deepswe-v1.1-openai-sol-unspecified",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 72.7,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "DeepSWE v1.1: GPT-6 Astra 74.1%; GPT-5.6 Sol 72.7%."
          },
          {
            "id": "deepswe-v1.1-datacurve-sol-max",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 73,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 60000,
              "costPerTaskUsd": 6.46
            },
            "context": null,
            "evidenceExcerpt": "gpt-5.6-sol[max]: 73%\u00b13%; average cost $6.46; output tokens 60k."
          },
          {
            "id": "token-economy-palindrome-sol-medium-20260809",
            "benchmarkId": "token-economy-controlled-setups-v1",
            "name": "Token Economy controlled setups",
            "version": "1",
            "unit": "percent successful attempts",
            "maximumScore": 100,
            "effort": "medium",
            "score": 100,
            "publishedAt": "2026-08-09",
            "sourceUrl": "https://github.com/agent-orc/token-economy/blob/main/benchmarks/results/palindrome-repair/20260809T092240107Z.report.json",
            "confidence": "ownRun",
            "evidenceStatus": "unknown",
            "sampleContext": "palindrome-repair: sol-medium passed 3 of 3 isolated attempts (100%).",
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "palindrome-repair: sol-medium passed 3 of 3 isolated attempts (100%)."
          },
          {
            "id": "coderabbit-sol-2026-07-09-recall-gpt-5.6-sol",
            "benchmarkId": "coderabbit-sol-2026-07-09-recall",
            "name": "CodeRabbit Sol / actionable known-issue coverage",
            "version": "2026-07-09 / Sol model lane",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 69.7,
            "publishedAt": "2026-07-09",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-5-6-sol-and-terra-benchmark",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit July 9 Sol review lane",
              "sampleNotes": "99 known error patterns; 69 actionable and 74 full-stream issue hits. PR count, repeat count, reasoning effort, exact judge/pipeline version unknown. 231 is raw comment volume, not a valid precision denominator for reconstructing true-positive counts."
            },
            "evidenceExcerpt": "69 of 99 known issues caught actionably. Published Sol lane; 61 nitpicks are also reported. The production baseline is a separate model ensemble."
          },
          {
            "id": "coderabbit-sol-2026-07-09-precision-gpt-5.6-sol",
            "benchmarkId": "coderabbit-sol-2026-07-09-precision",
            "name": "CodeRabbit Sol / actionable comment precision",
            "version": "2026-07-09 / Sol model lane",
            "unit": "% actionable-comment precision",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 31.6,
            "publishedAt": "2026-07-09",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-5-6-sol-and-terra-benchmark",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit July 9 Sol review lane",
              "sampleNotes": "99 known error patterns; 69 actionable and 74 full-stream issue hits. PR count, repeat count, reasoning effort, exact judge/pipeline version unknown. 231 is raw comment volume, not a valid precision denominator for reconstructing true-positive counts."
            },
            "evidenceExcerpt": "Share of actionable comments correct enough to keep; the separately stated 231 comments are RAW model output before filtering. Published Sol lane; 61 nitpicks are also reported. The production baseline is a separate model ensemble."
          }
        ],
        "taskStudies": [
          {
            "id": "heavy-design",
            "label": "Heavy design (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; canonical policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "planning",
            "label": "Planning (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; operator strong-capability floor is active.",
            "references": [
              "docs/operations/evidence-base/recommendation-semantics.md",
              "docs/operations/evidence-base/evidence-map.md"
            ]
          },
          {
            "id": "decision-making",
            "label": "Decision-making (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; operator strong-capability floor is active.",
            "references": [
              "docs/operations/evidence-base/recommendation-semantics.md",
              "docs/operations/evidence-base/evidence-map.md"
            ]
          },
          {
            "id": "feature",
            "label": "Feature and bug implementation (historical baseline)",
            "status": "controlledCodingEvidence",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 4,
            "outcomeRate": 0.75,
            "costPerSuccessfulOutcome": {
              "amountUsd": 0.254174,
              "pricedAtUtc": "2026-08-08T16:17:05Z",
              "includesVerification": false,
              "note": "Catalog list-price estimate from retained subject-model usage; local deterministic test CPU is excluded."
            },
            "benchmarkStatus": "Controlled pilot complete; below the 20-sample validation gate.",
            "references": [
              "benchmarks/results/curated-hard-coding-off-by-one/20260808T161705055Z.json",
              "benchmarks/results/curated-hard-coding-unicode-locale/20260808T161855489Z.json",
              "benchmarks/results/curated-hard-coding-cross-file/20260808T162154930Z.json",
              "benchmarks/results/curated-hard-coding-underspecified/20260808T162707845Z.json"
            ]
          },
          {
            "id": "research",
            "label": "Research and investigation (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned class-specific suite; canonical policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "quality-studio-review",
            "label": "Quality Studio review (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "No eligible operational QS cohort; policy baseline only.",
            "references": [
              "results/routing-evidence/review/v1/review-evidence.json"
            ]
          },
          {
            "id": "html-ui-implementation",
            "label": "HTML/UI implementation (historical baseline)",
            "status": "controlledPilot",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 2,
            "outcomeRate": 1.0,
            "costPerSuccessfulOutcome": {
              "amountUsd": 0.212954,
              "pricedAtUtc": "2026-08-12T08:35:46Z",
              "includesVerification": false,
              "note": "Catalog list-price estimate for subject-model usage per passed rendered outcome. The fixed jury consumed 31,074 additional tokens across the two Sol/medium cases; component-level jury usage was not retained, so its dollar cost remains explicit unknown rather than an invented zero."
            },
            "benchmarkStatus": "Controlled two-scenario pilot complete; below the 20-attempt and five-scenario qualification gates.",
            "references": [
              "benchmarks/results/html-ui-incident-console/20260812T083546713Z.json",
              "benchmarks/results/html-ui-billing-onboarding/20260812T084429463Z.json",
              "benchmarks/task-class-studies/html-ui-v1.json"
            ]
          },
          {
            "id": "security-assessment",
            "label": "Security assessment (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "xHigh"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned slice; correctness hard floor is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "redundancy-detection",
            "label": "Redundancy detection (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned slice; canonical policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.6-sol",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 4,
        "outputPerMTok": 20,
        "cachedInputPerMTok": 0.4,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-08-21T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh",
        "ultra"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "observational",
      "provisional": false
    },
    {
      "displayName": "GPT-6 Astra",
      "releaseDate": "2026-09-03",
      "releaseDateSource": "https://openai.com/index/safety-overview-gpt-6-astra/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Selectable for explicit evaluation and operator pins, provisional. Operator observed two access_programs.cyber HTTP 400s in about eight runs on 2026-09-18. No comparable local completion cohort; not a default tier or equivalent provider fallback.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 57.88,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/16db8ad5-84aa-4588-b660-1ce68c0d45e2",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 45304,
              "costPerTaskUsd": 7.122758,
              "latencyMilliseconds": 2208500
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:53:40.128995+00:00",
              "sourceUpdatedAt": "2026-09-10T21:57:57.437008+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.72,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 2350.51,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "191/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 191/330 attempts resolved (57.88%, 95% CI half-width 2.72 percentage points), using Codex at xHigh. PublishedAt is first observed availability, not an asserted run date. Total cost USD 2350.51; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb40-official-gpt-6-astra-medium-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "medium",
            "score": 54.24,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/f3c3d5a6-6424-4acb-bfcc-615c3f79f6dd",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 32638,
              "costPerTaskUsd": 5.802424,
              "latencyMilliseconds": 1885900
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:53:40.128995+00:00",
              "sourceUpdatedAt": "2026-09-10T21:57:52.562258+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.66,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 1914.8,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "179/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 179/330 attempts resolved (54.24%, 95% CI half-width 2.66 percentage points), using Codex at medium. PublishedAt is first observed availability, not an asserted run date. Total cost USD 1914.8; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb40-official-gpt-6-astra-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 58.18,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/5c537be4-7fc3-449b-8bfc-ceb9061c2535",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 72694,
              "costPerTaskUsd": 9.900545,
              "latencyMilliseconds": 2796300
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:53:40.128995+00:00",
              "sourceUpdatedAt": "2026-09-10T21:58:00.001022+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.79,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 3267.18,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "192/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 192/330 attempts resolved (58.18%, 95% CI half-width 2.79 percentage points), using Codex at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 3267.18; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb40-official-gpt-6-astra-low-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "low",
            "score": 50.61,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/b3ad58f3-b311-4d4d-875b-3158cf0309d6",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 24207,
              "costPerTaskUsd": 4.719091,
              "latencyMilliseconds": 1683700
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:53:40.128995+00:00",
              "sourceUpdatedAt": "2026-09-10T21:57:49.894535+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.75,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 1557.3,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "167/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 167/330 attempts resolved (50.61%, 95% CI half-width 2.75 percentage points), using Codex at low. PublishedAt is first observed availability, not an asserted run date. Total cost USD 1557.3; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb40-official-gpt-6-astra-high-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "high",
            "score": 57.88,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/3475050c-bf5e-4261-a3f6-0af5350af13f",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 41527,
              "costPerTaskUsd": 6.87703,
              "latencyMilliseconds": 2113700
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:53:40.128995+00:00",
              "sourceUpdatedAt": "2026-09-10T21:57:54.867871+00:00",
              "harness": "Codex",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 2.97,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 2269.42,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "191/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 191/330 attempts resolved (57.88%, 95% CI half-width 2.97 percentage points), using Codex at high. PublishedAt is first observed availability, not an asserted run date. Total cost USD 2269.42; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb21-official-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "terminal-bench-v2.1-official-native-agents",
            "name": "Terminal-Bench 2.1 / official native-agent leaderboard",
            "version": "2.1",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 85.84,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1/1/leaderboards/main/rows/3d9a7a70-ad72-4197-b895-c73698601637",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 11882,
              "costPerTaskUsd": 1.996719,
              "latencyMilliseconds": 487300
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:54:37.791805+00:00",
              "sourceUpdatedAt": "2026-09-03T20:54:37.791805+00:00",
              "harness": "Codex",
              "taskCount": 89,
              "trialCount": 445,
              "confidenceIntervalHalfWidth": 1.43,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 888.54,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "undefined/445 successful attempts. Dataset task count is distinct from trial count. 445/89=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: undefined/445 attempts resolved (85.84%, 95% CI half-width 1.43 percentage points), using Codex at xHigh. PublishedAt is first observed availability, not an asserted run date. Total cost USD 888.54; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb21-official-gpt-6-astra-medium-2026-09-12",
            "benchmarkId": "terminal-bench-v2.1-official-native-agents",
            "name": "Terminal-Bench 2.1 / official native-agent leaderboard",
            "version": "2.1",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "medium",
            "score": 86.97,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1/1/leaderboards/main/rows/17be4c09-ab31-4202-9d56-31a2ccf37e1b",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 6517,
              "costPerTaskUsd": 1.391169,
              "latencyMilliseconds": 390400
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:54:37.791805+00:00",
              "sourceUpdatedAt": "2026-09-03T20:54:37.791805+00:00",
              "harness": "Codex",
              "taskCount": 89,
              "trialCount": 445,
              "confidenceIntervalHalfWidth": 1.89,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 619.07,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "undefined/445 successful attempts. Dataset task count is distinct from trial count. 445/89=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: undefined/445 attempts resolved (86.97%, 95% CI half-width 1.89 percentage points), using Codex at medium. PublishedAt is first observed availability, not an asserted run date. Total cost USD 619.07; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb21-official-gpt-6-astra-max-2026-09-12",
            "benchmarkId": "terminal-bench-v2.1-official-native-agents",
            "name": "Terminal-Bench 2.1 / official native-agent leaderboard",
            "version": "2.1",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 86.74,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1/1/leaderboards/main/rows/f8d623c8-5f1b-4f9c-8deb-d110fa91e1d2",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 17028,
              "costPerTaskUsd": 2.403933,
              "latencyMilliseconds": 576000
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:54:37.791805+00:00",
              "sourceUpdatedAt": "2026-09-03T20:54:37.791805+00:00",
              "harness": "Codex",
              "taskCount": 89,
              "trialCount": 445,
              "confidenceIntervalHalfWidth": 1.36,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 1069.75,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "undefined/445 successful attempts. Dataset task count is distinct from trial count. 445/89=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: undefined/445 attempts resolved (86.74%, 95% CI half-width 1.36 percentage points), using Codex at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 1069.75; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb21-official-gpt-6-astra-low-2026-09-12",
            "benchmarkId": "terminal-bench-v2.1-official-native-agents",
            "name": "Terminal-Bench 2.1 / official native-agent leaderboard",
            "version": "2.1",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "low",
            "score": 86.74,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1/1/leaderboards/main/rows/93fc9e32-98c9-4e4a-a6a0-89dd3da598af",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 4400,
              "costPerTaskUsd": 1.041461,
              "latencyMilliseconds": 348900
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:54:37.791805+00:00",
              "sourceUpdatedAt": "2026-09-03T20:54:37.791805+00:00",
              "harness": "Codex",
              "taskCount": 89,
              "trialCount": 445,
              "confidenceIntervalHalfWidth": 1.82,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 463.45,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "undefined/445 successful attempts. Dataset task count is distinct from trial count. 445/89=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: undefined/445 attempts resolved (86.74%, 95% CI half-width 1.82 percentage points), using Codex at low. PublishedAt is first observed availability, not an asserted run date. Total cost USD 463.45; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "tb21-official-gpt-6-astra-high-2026-09-12",
            "benchmarkId": "terminal-bench-v2.1-official-native-agents",
            "name": "Terminal-Bench 2.1 / official native-agent leaderboard",
            "version": "2.1",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "high",
            "score": 87.42,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1/1/leaderboards/main/rows/03ff7953-4955-4fc5-9ddd-911cab72d7f1",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 9485,
              "costPerTaskUsd": 1.738202,
              "latencyMilliseconds": 434100
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T20:54:37.791805+00:00",
              "sourceUpdatedAt": "2026-09-03T20:54:37.791805+00:00",
              "harness": "Codex",
              "taskCount": 89,
              "trialCount": 445,
              "confidenceIntervalHalfWidth": 1.79,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 773.5,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "undefined/445 successful attempts. Dataset task count is distinct from trial count. 445/89=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: undefined/445 attempts resolved (87.42%, 95% CI half-width 1.79 percentage points), using Codex at high. PublishedAt is first observed availability, not an asserted run date. Total cost USD 773.5; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "scale-test-writing-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-test-writing-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Test Writing / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 50.74,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-tw",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Codex",
              "taskCount": 90,
              "confidenceIntervalHalfWidth": 5.91,
              "attemptsPerTask": 3,
              "trialCount": 270
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 50.74% resolve rate with reported +/-5.91 percentage points at xHigh. 90 tasks, three trials per task. Manifest, mutation-test and rubric checks determine resolution; Opus 4.5 judges rubric checks. The page records a July 28 dataset/harness update, not publication of September model rows."
          },
          {
            "id": "scale-refactoring-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-refactoring-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Refactoring / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 59.05,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-refactoring",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Codex",
              "taskCount": 70,
              "confidenceIntervalHalfWidth": 6.43
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 59.05% resolve rate with reported +/-6.43 percentage points at xHigh. 70 tasks from 10 production repositories. Existing tests and all required rubrics must pass; Opus 4.5 judges rubric checks. Total attempts and CI level not established. No Opus 5 result was present."
          },
          {
            "id": "scale-qna-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-qna-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Codebase QnA / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 59.14,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-qna",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Codex",
              "taskCount": 124,
              "confidenceIntervalHalfWidth": 4.88
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 59.14% resolve rate with reported +/-4.88 percentage points at xHigh. 124 tasks from 11 production repositories. All rubric criteria must pass, with Opus 4.5 as judge. Sample count is tasks; total trials are not asserted from another evaluator methodology."
          },
          {
            "id": "deepswe-owner-gpt-6-astra-xhigh-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 74,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 30000,
              "costPerTaskUsd": 6.52
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 3,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 74% with reported +/-3 percentage points; 113 tasks; average cost USD 6.52, displayed output 30k and 29 steps. These are rounded leaderboard measurements, not a new local evaluation."
          },
          {
            "id": "aa-ii-v4.3-astra-low-2026-09-11",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "low",
            "score": 46,
            "publishedAt": "2026-09-11",
            "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-low-vs-gpt-5-6-sol-high",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 4000,
              "costPerTaskUsd": 0.82,
              "latencyMilliseconds": 2620
            },
            "context": null,
            "evidenceExcerpt": "Intelligence Index: 46 vs 42; cost per task: $0.82 vs $0.81; output tokens: 4k vs 13k."
          },
          {
            "id": "deepswe-v1-aa-astra-max",
            "benchmarkId": "deepswe-v1-aa-harness",
            "name": "DeepSWE",
            "version": "v1, Artificial Analysis harness",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 68,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Its lead comes from other components, partly offset by a lower score on DeepSWE: 68% vs 72%."
          },
          {
            "id": "aa-ii-v4.3-astra-max-2026-09-09",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 53,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 27000,
              "costPerTaskUsd": 3.26
            },
            "context": null,
            "evidenceExcerpt": "GPT-6 Astra (max) scores 53; at max it costs $3.26 and uses 27k output tokens per task."
          },
          {
            "id": "aa-coding-v15-gpt-6-astra-max-2026-09-09",
            "benchmarkId": "artificial-analysis-coding-agent-index-v1.5-native-agents",
            "name": "Artificial Analysis Coding Agent Index 1.5",
            "version": "1.5 / September 9 comparison",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 62,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 7.09
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Codex",
              "taskCount": 303,
              "trialCount": 909,
              "attemptsPerTask": 3,
              "sampleNotes": "Scores are equal-weight component aggregates, not 909-trial pooled accuracy. Native-agent versions are not supplied in the article. No new composite was calculated here.",
              "additionalSourceUrls": [
                "https://artificialanalysis.ai/methodology/coding-agents-benchmarking/"
              ]
            },
            "evidenceExcerpt": "Artificial Analysis September 9 comparison: 62 Coding Agent Index points using Codex at max. Version 1.5 has 303 distinct tasks and 909 attempts across three equally weighted benchmark components."
          },
          {
            "id": "aa-coding-2026-09-09-astra-max",
            "benchmarkId": "artificial-analysis-coding-agent-index-2026-09-09",
            "name": "Artificial Analysis Coding Agent Index",
            "version": "2026-09-09 snapshot",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 62,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 7.09
            },
            "context": null,
            "evidenceExcerpt": "In Codex, GPT-6 Astra scores 62 in the Index; at max effort it costs $7.09 per task."
          },
          {
            "id": "aa-ii-v4.2-astra-low-2026-09-07",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.2",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.2",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "low",
            "score": 49,
            "publishedAt": "2026-09-07",
            "sourceUrl": "https://thenewstack.io/astra-reasoning-effort-cost/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Artificial Analysis currently scores Astra-low at 49 on its Intelligence Index, narrowly ahead of Sol-high at 48."
          },
          {
            "id": "coderabbit-astra-2026-09-04-overall-coverage-gpt-6-astra",
            "benchmarkId": "coderabbit-astra-2026-09-04-overall-coverage",
            "name": "CodeRabbit / overall actionable bug coverage",
            "version": "2026-09-04 internal evaluation / overall",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 61.3,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / overall",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "overall actionable bug coverage: 61.3% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "coderabbit-astra-2026-09-04-cross-file-coverage-gpt-6-astra",
            "benchmarkId": "coderabbit-astra-2026-09-04-cross-file-coverage",
            "name": "CodeRabbit / cross-file actionable bug coverage",
            "version": "2026-09-04 internal evaluation / cross-file",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 57.1,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / cross-file",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "cross-file actionable bug coverage: 57.1% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "terminal-bench-v4-openai-astra-unspecified",
            "benchmarkId": "terminal-bench-v4.0",
            "name": "Terminal-Bench",
            "version": "4.0",
            "unit": "percent passed",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 57.9,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "Terminal-Bench 4.0: GPT-6 Astra 57.9%; GPT-5.6 Sol 37.3%."
          },
          {
            "id": "deepswe-v1.1-datacurve-astra-xhigh",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 74,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 30000,
              "costPerTaskUsd": 6.52
            },
            "context": null,
            "evidenceExcerpt": "gpt-6-astra[xhigh]: 74%\u00b13%; average cost $6.52; output tokens 30k."
          },
          {
            "id": "deepswe-v1.1-openai-astra-unspecified",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 74.1,
            "publishedAt": "2026-09-03",
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "confidence": "publisherReported",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "DeepSWE v1.1: GPT-6 Astra 74.1%; GPT-5.6 Sol 72.7%."
          }
        ],
        "taskStudies": [],
        "assessment": {
          "modelId": "gpt-6-astra",
          "summary": "A strong cost-aware starting choice for terminal coding and difficult repository work. The clearest effort evidence supports high for terminal tasks and xHigh for the measured long-task and codebase configurations.",
          "strengths": [
            {
              "text": "Terminal-Bench 4.0 provides an actual effort sweep in Codex: high resolves 191/330 attempts (57.88%) and max 192/330 (58.18%). Total reported cost is $2,269.42 versus $3,267.18: high uses 30.5% less money for nearly the same observed result, with overlapping 95% intervals.",
              "sourceUrls": [
                "https://www.tbench.ai/",
                "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/3475050c-bf5e-4261-a3f6-0af5350af13f",
                "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/5c537be4-7fc3-449b-8bfc-ceb9061c2535"
              ],
              "evidenceIds": [
                "tb40-official-gpt-6-astra-high-2026-09-12",
                "tb40-official-gpt-6-astra-max-2026-09-12"
              ]
            },
            {
              "text": "DeepSWE 1.1 measures Astra/xHigh at 74% and $6.52 per task in the shared owner harness; Opus 5/max is also 74% at $11.84. AA's separate Coding Agent Index v1.5 scores Astra/max and Fable 5.1/max at 62, with Opus 5/max at 60.",
              "sourceUrls": [
                "https://deepswe.datacurve.ai/",
                "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra"
              ],
              "evidenceIds": [
                "deepswe-owner-gpt-6-astra-xhigh-2026-09-12",
                "aa-coding-v15-gpt-6-astra-max-2026-09-09"
              ]
            },
            {
              "text": "Scale's Refactoring board reports 59.05% for Astra/xHigh and 56.67% for Fable 5.1/xHigh. Their published intervals overlap; both are supported candidates for that task class.",
              "sourceUrls": [
                "https://labs.scale.com/leaderboard/sweatlas-refactoring"
              ],
              "evidenceIds": [
                "scale-refactoring-gpt-6-astra-xhigh-2026-09-12"
              ]
            }
          ],
          "tradeoffs": [
            {
              "text": "Scale's Test Writing board reports Astra/xHigh at 50.74%, below the Fable 5.1/xHigh point estimate of 67.04% and Opus 5/xHigh at 62.22%. A terminal-coding advantage does not imply the same ordering for production test creation.",
              "sourceUrls": [
                "https://labs.scale.com/leaderboard/sweatlas-tw"
              ],
              "evidenceIds": [
                "scale-test-writing-gpt-6-astra-xhigh-2026-09-12"
              ]
            }
          ],
          "effortGuidance": {
            "text": "Start at high for terminal coding: the official same-agent sweep shows little observed benefit from max at substantially higher cost. Use xHigh as a candidate for long repository changes, codebase questions and refactoring, where that exact level is measured. Escalating everything to max is not supported by the terminal sweep.",
            "sourceUrls": [
              "https://www.tbench.ai/",
              "https://deepswe.datacurve.ai/",
              "https://labs.scale.com/leaderboard/sweatlas-qna",
              "https://labs.scale.com/leaderboard/sweatlas-refactoring"
            ]
          },
          "gaps": [
            "The terminal sweep does not prove that high matches max on all reasoning or coding tasks.",
            "No exact Astra row was found in the inspected SWE-bench Verified or SWE-bench Pro leaderboards.",
            "Official leaderboard records do not identify the execution organization or exact Codex version; model/agent vendor fields are not proof of an independent run."
          ],
          "displayName": "GPT-6 Astra"
        },
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-6-astra",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 10,
        "outputPerMTok": 50,
        "cachedInputPerMTok": 1,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-09-03T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max",
        "ultra"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-6 Sol",
      "releaseDate": "2026-09-22",
      "releaseDateSource": "https://developers.openai.com/api/docs/changelog",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Operator evidence dated 2026-09-25: Codex CLI 0.155.0 exposes minimal/low/medium/high/xhigh. Sol runs on the workstation and agent-runner-01 since September 24; CLI default xhigh, policy explicitly selects medium or xhigh. Source ultra pins must be explicitly downgraded to xhigh, never silently migrated. Default core route by operator decision backed by dated prices and vendor claims; local completion and review cohorts remain unqualified. API max benchmarks do not establish Codex max support.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [
          {
            "effort": "medium",
            "role": "coreTask",
            "minimumTaskScore": 51,
            "maximumTaskScore": 69
          },
          {
            "effort": "xhigh",
            "role": "coreTask",
            "minimumTaskScore": 70,
            "maximumTaskScore": 100
          }
        ],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "aa-ii-v4.3.2-gpt-6-sol-max-2026-09-25",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3.2",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3.2",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 48,
            "publishedAt": "2026-09-25",
            "sourceUrl": "https://artificialanalysis.ai/models/gpt-6-sol",
            "confidence": "publisherReported",
            "evidenceStatus": "provisional",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 1.06
            },
            "context": {
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-25",
              "harness": "Artificial Analysis Intelligence Index",
              "harnessVersion": "4.3.2",
              "sampleNotes": "Live page first observed on this date; actual run date, sample count and uncertainty interval unknown. API max is not the local Codex ladder.",
              "costBasis": "Publisher weighted cost per Intelligence Index task; not local card cost."
            },
            "evidenceExcerpt": "September 25 public snapshot: gpt-6-sol at max scores 48 on Intelligence Index v4.3.2; $1.06 per index task."
          },
          {
            "id": "openai-deepswe-v1.1-gpt-6-sol-max-2026-09-22",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 68.8,
            "publishedAt": "2026-09-22",
            "sourceUrl": "https://openai.com/index/introducing-gpt-6-sol-and-luna/",
            "confidence": "publisherReported",
            "evidenceStatus": "provisional",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "modelProvider",
              "sourcePublisher": "OpenAI",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-25",
              "harness": "OpenAI reported DeepSWE v1.1 evaluation; exact execution configuration not retained",
              "harnessVersion": "1.1",
              "sampleNotes": "Local completion cohort absent. Denominator and uncertainty interval not established from the announcement; no noRegression claim.",
              "additionalSourceUrls": [
                "https://developers.openai.com/api/docs/models/gpt-6-sol"
              ]
            },
            "evidenceExcerpt": "OpenAI reports DeepSWE v1.1 at API max: gpt-6-sol 68.8%. This is not a local Codex medium/xhigh measurement."
          }
        ],
        "taskStudies": [
          {
            "id": "heavy-design",
            "label": "Heavy design",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "planning",
            "label": "Planning",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "decision-making",
            "label": "Decision-making",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "feature",
            "label": "Feature and bug implementation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "research",
            "label": "Research and investigation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "quality-studio-review",
            "label": "Quality Studio review",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "html-ui-implementation",
            "label": "HTML/UI implementation",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "source-code-review",
            "label": "Source-code review",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "security-assessment",
            "label": "Security assessment",
            "status": "policyBaseline",
            "efforts": [
              "xHigh"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "redundancy-detection",
            "label": "Redundancy detection",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-6-sol",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "standard",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 2,
        "outputPerMTok": 10,
        "cachedInputPerMTok": 0.2,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-09-22T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-6 Luna",
      "releaseDate": "2026-09-22",
      "releaseDateSource": "https://developers.openai.com/api/docs/changelog",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Operator evidence dated 2026-09-25: Codex CLI 0.155.0 exposes minimal/low/medium/high/xhigh. Luna execution succeeded on agent-runner-01 on September 25; the old unverified availability note is superseded. Default core route by operator decision backed by dated prices and vendor claims; local completion and review cohorts remain unqualified. API max benchmarks do not establish Codex max support.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [
          {
            "effort": "medium",
            "role": "coreTask",
            "minimumTaskScore": 0,
            "maximumTaskScore": 20
          }
        ],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "aa-ii-v4.3.2-gpt-6-luna-max-2026-09-25",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3.2",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3.2",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 37,
            "publishedAt": "2026-09-25",
            "sourceUrl": "https://artificialanalysis.ai/models/gpt-6-luna",
            "confidence": "publisherReported",
            "evidenceStatus": "provisional",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 0.07
            },
            "context": {
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "dateBasis": "firstObservedPublicSnapshot",
              "observedAt": "2026-09-25",
              "harness": "Artificial Analysis Intelligence Index",
              "harnessVersion": "4.3.2",
              "sampleNotes": "Live page first observed on this date; actual run date, sample count and uncertainty interval unknown. API max is not the local Codex ladder.",
              "costBasis": "Publisher weighted cost per Intelligence Index task; not local card cost."
            },
            "evidenceExcerpt": "September 25 public snapshot: gpt-6-luna at max scores 37 on Intelligence Index v4.3.2; $0.07 per index task."
          },
          {
            "id": "openai-deepswe-v1.1-gpt-6-luna-max-2026-09-22",
            "benchmarkId": "deepswe-v1.1",
            "name": "DeepSWE v1.1",
            "version": "1.1 (113 tasks)",
            "unit": "percent resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 66.6,
            "publishedAt": "2026-09-22",
            "sourceUrl": "https://openai.com/index/introducing-gpt-6-sol-and-luna/",
            "confidence": "publisherReported",
            "evidenceStatus": "provisional",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "modelProvider",
              "sourcePublisher": "OpenAI",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-25",
              "harness": "OpenAI reported DeepSWE v1.1 evaluation; exact execution configuration not retained",
              "harnessVersion": "1.1",
              "sampleNotes": "Local completion cohort absent. Denominator and uncertainty interval not established from the announcement; no noRegression claim.",
              "additionalSourceUrls": [
                "https://developers.openai.com/api/docs/models/gpt-6-luna"
              ]
            },
            "evidenceExcerpt": "OpenAI reports DeepSWE v1.1 at API max: gpt-6-luna 66.6%. This is not a local Codex medium/xhigh measurement."
          }
        ],
        "taskStudies": [
          {
            "id": "mechanical-chore",
            "label": "Mechanical chore",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "doc-edit",
            "label": "Documentation edit",
            "status": "policyBaseline",
            "efforts": [
              "medium"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-6-luna",
      "vendor": "openai",
      "cli": "codex",
      "tier": "light",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 0.1,
        "outputPerMTok": 0.5,
        "cachedInputPerMTok": 0.01,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-09-22T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "underpowered",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "underpowered",
        "mechanicalChore": "ideal",
        "docEdit": "ideal",
        "research": "underpowered",
        "review": null,
        "htmlUiImplementation": "underpowered",
        "sourceCodeReview": "underpowered",
        "securityAssessment": "underpowered",
        "redundancyDetection": "underpowered",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "Claude Fable 5.1",
      "releaseDate": "2026-09-01",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/fable-5-1/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Provider-supported model and five-level effort ladder: https://platform.claude.com/docs/en/models/fable-5-1/overview and https://platform.claude.com/docs/en/build-with-claude/effort (retrieved 2026-09-12). Selectable for explicit evaluation, compatibility comparison, and operator-pinned core tasks. Local completion and review fit remain unvalidated; support does not add a default tier or declare provider-fallback equivalence.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-claude-fable-5-1-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 57.88,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/c741608e-c94e-417d-b7a8-e67111a0c887",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 191277,
              "costPerTaskUsd": 18.919697,
              "latencyMilliseconds": 3893500
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-09-03T01:51:14.437314+00:00",
              "sourceUpdatedAt": "2026-09-03T02:56:08.149542+00:00",
              "harness": "Claude Code",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.76,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 6243.5,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "191/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 191/330 attempts resolved (57.88%, 95% CI half-width 3.76 percentage points), using Claude Code at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 6243.5; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "scale-test-writing-claude-fable-5-1-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-test-writing-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Test Writing / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 67.04,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-tw",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Claude Code",
              "taskCount": 90,
              "confidenceIntervalHalfWidth": 5.33,
              "attemptsPerTask": 3,
              "trialCount": 270
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 67.04% resolve rate with reported +/-5.33 percentage points at xHigh. 90 tasks, three trials per task. Manifest, mutation-test and rubric checks determine resolution; Opus 4.5 judges rubric checks. The page records a July 28 dataset/harness update, not publication of September model rows."
          },
          {
            "id": "scale-refactoring-claude-fable-5-1-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-refactoring-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Refactoring / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 56.67,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-refactoring",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Claude Code",
              "taskCount": 70,
              "confidenceIntervalHalfWidth": 6.52
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 56.67% resolve rate with reported +/-6.52 percentage points at xHigh. 70 tasks from 10 production repositories. Existing tests and all required rubrics must pass; Opus 4.5 judges rubric checks. Total attempts and CI level not established. No Opus 5 result was present."
          },
          {
            "id": "scale-qna-claude-fable-5-1-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-qna-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Codebase QnA / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 59.95,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-qna",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Claude Code",
              "taskCount": 124,
              "confidenceIntervalHalfWidth": 4.85
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 59.95% resolve rate with reported +/-4.85 percentage points at xHigh. 124 tasks from 11 production repositories. All rubric criteria must pass, with Opus 4.5 as judge. Sample count is tasks; total trials are not asserted from another evaluator methodology."
          },
          {
            "id": "aa-coding-v15-claude-fable-5-1-max-2026-09-09",
            "benchmarkId": "artificial-analysis-coding-agent-index-v1.5-native-agents",
            "name": "Artificial Analysis Coding Agent Index 1.5",
            "version": "1.5 / September 9 comparison",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 62,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Claude Code",
              "taskCount": 303,
              "trialCount": 909,
              "attemptsPerTask": 3,
              "sampleNotes": "Scores are equal-weight component aggregates, not 909-trial pooled accuracy. Native-agent versions are not supplied in the article. No new composite was calculated here.",
              "additionalSourceUrls": [
                "https://artificialanalysis.ai/methodology/coding-agents-benchmarking/"
              ],
              "fallbackPolicy": "max with default production fallback enabled"
            },
            "evidenceExcerpt": "Artificial Analysis September 9 comparison: 62 Coding Agent Index points using Claude Code at max. Version 1.5 has 303 distinct tasks and 909 attempts across three equally weighted benchmark components."
          },
          {
            "id": "aa-ii-v43-claude-fable-5-1-max-2026-09-07",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 53,
            "publishedAt": "2026-09-07",
            "sourceUrl": "https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-3",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Artificial Analysis Intelligence Index evaluation harnesses",
              "sampleNotes": "Composite Intelligence Index v4.3; index points are not percent accuracy and are not interchangeable with older index versions.",
              "fallbackPolicy": "Default production fallback enabled; routes to Opus 4.8 or Opus 5 can contribute. Not a pure single-model result."
            },
            "evidenceExcerpt": "The dated September 7 Intelligence Index v4.3 announcement reports 53 index points at max. Fable 5.1 uses the default fallback configuration."
          },
          {
            "id": "tb21-aa-claude-fable-5-1-max-2026-09-01",
            "benchmarkId": "terminal-bench-v2.1-aa-intelligence-harness",
            "name": "Terminal-Bench 2.1 / Artificial Analysis",
            "version": "2.1 / Artificial Analysis historical runs",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 91.4,
            "publishedAt": "2026-09-01",
            "sourceUrl": "https://artificialanalysis.ai/articles/claude-fable-5-1",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Artificial Analysis evaluation harness; exact historical configuration unspecified",
              "taskCount": 89,
              "fallbackPolicy": "Production fallback enabled in the launch evaluation; exact per-benchmark contribution unspecified.",
              "sampleNotes": "Independent pre-release evaluation in collaboration with Anthropic. Article does not give exact trial count or error bars for this result."
            },
            "evidenceExcerpt": "Dated Artificial Analysis launch assessment reports 91.4% on Terminal-Bench 2.1 at max. This is an older suite and evaluator run; it must not be merged with TB4 or the official native-agent leaderboard."
          },
          {
            "id": "coderabbit-fable51-2026-09-01-low-recall-claude-fable-5-1",
            "benchmarkId": "coderabbit-fable51-2026-09-01-low-recall",
            "name": "CodeRabbit Fable 5.1 low setting / known-issue recall",
            "version": "2026-09-01 / internal low reasoning setting",
            "unit": "% known-issue recall",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 61,
            "publishedAt": "2026-09-01",
            "sourceUrl": "https://www.coderabbit.ai/blog/fable-5-1-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "latencyMilliseconds": 1118000
            },
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit September 1 review pipeline / internal low setting",
              "taskCount": 45,
              "sampleNotes": "Known-issue denominator 105; 64 issues found; 92 review-file model calls include retries and split-file batches. 166 final comments and 79 separate nitpicks. No reliable token totals, repeated-seed count, judge identity or exact pipeline version. Low/High do not establish provider effort enum values."
            },
            "evidenceExcerpt": "Share of the 105 known-issue points with at least one valid comment. Internal low setting: 61%; 45 review tasks, 105 known issues, 166 final comments, 79 separately counted nitpicks."
          },
          {
            "id": "coderabbit-fable51-2026-09-01-low-precision-claude-fable-5-1",
            "benchmarkId": "coderabbit-fable51-2026-09-01-low-precision",
            "name": "CodeRabbit Fable 5.1 low setting / processed comment precision",
            "version": "2026-09-01 / internal low reasoning setting",
            "unit": "% processed-comment precision",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 37.3,
            "publishedAt": "2026-09-01",
            "sourceUrl": "https://www.coderabbit.ai/blog/fable-5-1-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "latencyMilliseconds": 1118000
            },
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit September 1 review pipeline / internal low setting",
              "taskCount": 45,
              "sampleNotes": "Known-issue denominator 105; 64 issues found; 92 review-file model calls include retries and split-file batches. 166 final comments and 79 separate nitpicks. No reliable token totals, repeated-seed count, judge identity or exact pipeline version. Low/High do not establish provider effort enum values."
            },
            "evidenceExcerpt": "Share of final post-pipeline comments judged valid; not a claim about all possible code defects. Internal low setting: 37.3%; 45 review tasks, 105 known issues, 166 final comments, 79 separately counted nitpicks."
          },
          {
            "id": "coderabbit-fable51-2026-09-01-high-recall-claude-fable-5-1",
            "benchmarkId": "coderabbit-fable51-2026-09-01-high-recall",
            "name": "CodeRabbit Fable 5.1 high setting / known-issue recall",
            "version": "2026-09-01 / internal high reasoning setting",
            "unit": "% known-issue recall",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 57.1,
            "publishedAt": "2026-09-01",
            "sourceUrl": "https://www.coderabbit.ai/blog/fable-5-1-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "latencyMilliseconds": 1296000
            },
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit September 1 review pipeline / internal high setting",
              "taskCount": 45,
              "sampleNotes": "Known-issue denominator 105; 92 review-file model calls include retries and split-file batches. 165 final comments and 88 separate nitpicks. No reliable token totals, repeated-seed count, judge identity or exact pipeline version. Low/High do not establish provider effort enum values."
            },
            "evidenceExcerpt": "Share of the 105 known-issue points with at least one valid comment. Internal high setting: 57.1%; 45 review tasks, 105 known issues, 165 final comments, 88 separately counted nitpicks."
          },
          {
            "id": "coderabbit-fable51-2026-09-01-high-precision-claude-fable-5-1",
            "benchmarkId": "coderabbit-fable51-2026-09-01-high-precision",
            "name": "CodeRabbit Fable 5.1 high setting / processed comment precision",
            "version": "2026-09-01 / internal high reasoning setting",
            "unit": "% processed-comment precision",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 36.4,
            "publishedAt": "2026-09-01",
            "sourceUrl": "https://www.coderabbit.ai/blog/fable-5-1-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "latencyMilliseconds": 1296000
            },
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit September 1 review pipeline / internal high setting",
              "taskCount": 45,
              "sampleNotes": "Known-issue denominator 105; 92 review-file model calls include retries and split-file batches. 165 final comments and 88 separate nitpicks. No reliable token totals, repeated-seed count, judge identity or exact pipeline version. Low/High do not establish provider effort enum values."
            },
            "evidenceExcerpt": "Share of final post-pipeline comments judged valid; not a claim about all possible code defects. Internal high setting: 36.4%; 45 review tasks, 105 known issues, 165 final comments, 88 separately counted nitpicks."
          }
        ],
        "taskStudies": [],
        "assessment": {
          "modelId": "claude-fable-5-1",
          "summary": "A strong specialist candidate for production test writing and demanding coding work. Its benchmark results are competitive, while measured cost and production fallback behavior matter when choosing it as a default.",
          "strengths": [
            {
              "text": "Scale's Test Writing board measures Fable 5.1/xHigh at 67.04%, Opus 5/xHigh at 62.22% and Astra/xHigh at 50.74%, across 90 tasks with three trials each. This directly supports evaluating Fable for test-generation work; the Fable and Opus error bands overlap.",
              "sourceUrls": [
                "https://labs.scale.com/leaderboard/sweatlas-tw"
              ],
              "evidenceIds": [
                "scale-test-writing-claude-fable-5-1-xhigh-2026-09-12"
              ]
            },
            {
              "text": "At max, Fable 5.1 reaches 57.88% on the official Terminal-Bench 4.0 native-agent board and 62 points on AA Coding Agent Index v1.5. Astra scores 58.18% and 62 respectively; these results support comparable candidates, not a universal winner.",
              "sourceUrls": [
                "https://www.tbench.ai/",
                "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra"
              ],
              "evidenceIds": [
                "tb40-official-claude-fable-5-1-max-2026-09-12",
                "aa-coding-v15-claude-fable-5-1-max-2026-09-09"
              ]
            }
          ],
          "tradeoffs": [
            {
              "text": "AA's dated Astra comparison reports both models at 53 Intelligence Index v4.3 points at max, but weighted task cost is $7.63 for Fable 5.1 versus $3.26 for Astra. That is one suite's cost comparison, not a universal bill multiplier.",
              "sourceUrls": [
                "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra"
              ]
            },
            {
              "text": "AA evaluates Fable 5.1 with default fallback enabled. Its launch assessment says roughly 4% of output tokens across the Intelligence Index came from fallback models, including Opus 4.8 and Opus 5. This is measured production behavior, not a pure Fable-only configuration.",
              "sourceUrls": [
                "https://artificialanalysis.ai/articles/claude-fable-5-1"
              ]
            }
          ],
          "effortGuidance": {
            "text": "Use xHigh as an evidence-backed candidate for test writing and refactoring, where Scale measures that level. Max is directly measured on Terminal-Bench and AA's indexes, but the retrieved sources do not establish its incremental coding benefit over xHigh in a controlled same-harness sweep.",
            "sourceUrls": [
              "https://labs.scale.com/leaderboard/sweatlas-tw",
              "https://labs.scale.com/leaderboard/sweatlas-refactoring",
              "https://www.tbench.ai/",
              "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra"
            ]
          },
          "gaps": [
            "No exact Fable 5.1 row was found in the inspected DeepSWE owner, SWE-bench Verified or SWE-bench Pro boards; AA and provider runs remain separate evidence.",
            "The retrieved native-agent leaderboards do not give an exact agent version or fallback share for each result.",
            "Benchmark API dollars cannot be converted into Claude subscription quota percentages without observed plan-specific usage data."
          ],
          "displayName": "Claude Fable 5.1"
        },
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-fable-5-1",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 10,
        "outputPerMTok": 50,
        "cachedInputPerMTok": 0.25,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-09-01T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "Claude Opus 5.5",
      "releaseDate": "2026-09-22",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/opus-5-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Anthropic documents low/medium/high/xhigh/max with provider default medium for Claude Opus 5.5. Claude Code 2.1.281 execution reported the canonical model in modelUsage; 2.1.270 silently fell back after an unrecognized-model warning, so 2.1.281 is the minimum observed CLI version. Selectable for explicit evaluation, compatibility comparison, and operator-pinned core tasks; no comparable local completion or review cohort is available. This support adds no default tier, fallback equivalence, or automatic migration.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-5-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 4,
        "outputPerMTok": 20,
        "cachedInputPerMTok": 0.2,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-09-22T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-5.4 Mini",
      "releaseDate": "2026-03-17",
      "releaseDateSource": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Role exception for bounded decisions over compact structured evidence; not a core-task route.",
        "workflowRoles": [
          "boundedPipelineDecision"
        ],
        "routes": [
          {
            "effort": "high",
            "role": "boundedPipelineDecision",
            "minimumTaskScore": null,
            "maximumTaskScore": null
          }
        ],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [
          {
            "id": "graphical-quality-judgment",
            "label": "Graphical-quality judgment",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "graphical-quality-judgment",
            "label": "Graphical-quality judgment (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned slice; bounded-decision policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "consistency-checking",
            "label": "Consistency checking",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "consistency-checking",
            "label": "Consistency checking (historical baseline)",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": true,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "Planned slice; bounded-decision policy baseline is active.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.4-mini",
      "vendor": "openai",
      "cli": "codex",
      "tier": "light",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 0.75,
        "outputPerMTok": 4.5,
        "cachedInputPerMTok": 0.075,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-03-17T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high"
      ],
      "suitability": {
        "heavyDesign": "underpowered",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "underpowered",
        "mechanicalChore": "ideal",
        "docEdit": "ideal",
        "research": "underpowered",
        "review": null,
        "htmlUiImplementation": "underpowered",
        "sourceCodeReview": "underpowered",
        "securityAssessment": "underpowered",
        "redundancyDetection": "underpowered",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "Claude Sonnet 5",
      "releaseDate": "2026-06-30",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/sonnet-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Small favorable Claude Sonnet 5/high feature cohort; use only as an equivalent-capability provider fallback.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": [
          {
            "effort": "high",
            "forRoutes": [
              "terra-medium",
              "sol-medium"
            ]
          }
        ]
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-claude-sonnet-5-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 12.42,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/8180b9e4-4990-43d0-b906-cd8b43adaaa2",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 353330,
              "costPerTaskUsd": 29.102606,
              "latencyMilliseconds": 6510300
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:50.318045+00:00",
              "harness": "Claude Code",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.06,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 9603.86,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "41/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 41/330 attempts resolved (12.42%, 95% CI half-width 3.06 percentage points), using Claude Code at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 9603.86; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "deepswe-owner-claude-sonnet-5-max-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 54,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 214000,
              "costPerTaskUsd": 26.4
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 4,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 54% with reported +/-4 percentage points; 113 tasks; average cost USD 26.4, displayed output 214k and 268 steps. These are rounded leaderboard measurements, not a new local evaluation."
          },
          {
            "id": "token-economy-palindrome-sonnet5-high-20260809",
            "benchmarkId": "token-economy-controlled-setups-v1",
            "name": "Token Economy controlled setups",
            "version": "1",
            "unit": "percent successful attempts",
            "maximumScore": 100,
            "effort": "high",
            "score": 100,
            "publishedAt": "2026-08-09",
            "sourceUrl": "https://github.com/agent-orc/token-economy/blob/main/benchmarks/results/palindrome-repair/20260809T092240107Z.report.json",
            "confidence": "ownRun",
            "evidenceStatus": "unknown",
            "sampleContext": "palindrome-repair: claude-sonnet-5/high passed 3 of 3 isolated attempts (100%).",
            "secondaryMetrics": {},
            "context": null,
            "evidenceExcerpt": "palindrome-repair: claude-sonnet-5/high passed 3 of 3 isolated attempts (100%)."
          }
        ],
        "taskStudies": [
          {
            "id": "heavy-design",
            "label": "Heavy design",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "planning",
            "label": "Planning",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "decision-making",
            "label": "Decision-making",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "feature",
            "label": "Feature and bug implementation",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "research",
            "label": "Research and investigation",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "html-ui-implementation",
            "label": "HTML/UI implementation",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          },
          {
            "id": "redundancy-detection",
            "label": "Redundancy detection",
            "status": "policyBaseline",
            "efforts": [
              "high"
            ],
            "primaryRecommendation": false,
            "scenarioCount": 0,
            "outcomeRate": null,
            "costPerSuccessfulOutcome": null,
            "benchmarkStatus": "GPT-6 prior is provisional; no qualifying local completion cohort. Historical measurements remain separately attributed.",
            "references": [
              "docs/system/domains/model-routing-policy.md"
            ]
          }
        ],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-sonnet-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "standard",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 2,
        "outputPerMTok": 10,
        "cachedInputPerMTok": 0.2,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-06-30T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "fallbackOnly",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "Claude Opus 4.1",
      "releaseDate": "2025-08-05",
      "releaseDateSource": "https://platform.claude.com/docs/en/release-notes/overview#august-5-2025",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Retires 2026-08-05.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-4-1",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 15,
        "outputPerMTok": 75,
        "cachedInputPerMTok": 1.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-08-05T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": true,
      "costUnconfirmed": false,
      "selectionStatus": "deprecated",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Fable 5",
      "releaseDate": "2026-06-09",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/fable-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-claude-fable-5-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 44.55,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/36c077e0-4879-4444-b315-8532d66401d6",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 177636,
              "costPerTaskUsd": 22.015182,
              "latencyMilliseconds": 4202800
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:00.046808+00:00",
              "harness": "Claude Code",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.85,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 7265.01,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "147/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 147/330 attempts resolved (44.55%, 95% CI half-width 3.85 percentage points), using Claude Code at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 7265.01; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "deepswe-owner-claude-fable-5-xhigh-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 70,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 80000,
              "costPerTaskUsd": 13.41
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 3,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 70% with reported +/-3 percentage points; 113 tasks; average cost USD 13.41, displayed output 80k and 68 steps. These are rounded leaderboard measurements, not a new local evaluation."
          }
        ],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-fable-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 10,
        "outputPerMTok": 50,
        "cachedInputPerMTok": 1,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-06-09T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Opus 5",
      "releaseDate": "2026-07-24",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/opus-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Retained Anthropic vendor override for explicit operator pins, subject to task correctness floors. No automatic equivalence to Sol/xhigh is established.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-claude-opus-5-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 51.82,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/d71ac3d0-da36-49cb-89d9-323136e77111",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 200037,
              "costPerTaskUsd": 18.088212,
              "latencyMilliseconds": 4792800
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:44.281605+00:00",
              "harness": "Claude Code",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.39,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 5969.11,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "171/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 171/330 attempts resolved (51.82%, 95% CI half-width 3.39 percentage points), using Claude Code at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 5969.11; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "scale-test-writing-claude-opus-5-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-test-writing-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Test Writing / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 62.22,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-tw",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Claude Code",
              "taskCount": 90,
              "confidenceIntervalHalfWidth": 5.58,
              "attemptsPerTask": 3,
              "trialCount": 270
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 62.22% resolve rate with reported +/-5.58 percentage points at xHigh. 90 tasks, three trials per task. Manifest, mutation-test and rubric checks determine resolution; Opus 4.5 judges rubric checks. The page records a July 28 dataset/harness update, not publication of September model rows."
          },
          {
            "id": "scale-qna-claude-opus-5-xhigh-2026-09-12",
            "benchmarkId": "swe-atlas-qna-scale-native-agents-2026-09-12",
            "name": "SWE Atlas Codebase QnA / Scale native agents",
            "version": "2026-09-12 snapshot; dataset revision unspecified",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 63.17,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://labs.scale.com/leaderboard/sweatlas-qna",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Scale AI",
              "runnerOrganization": "Scale AI",
              "dateBasis": "firstObservedPublicSnapshot",
              "sampleNotes": "Current leaderboard snapshot; exact row publication date and harness version are not supplied. Error bars are reproduced in percentage points; the confidence level was not established. Native Claude Code and Codex runs differ in harness. Overlapping intervals do not establish a clear winner.",
              "harness": "Claude Code",
              "taskCount": 124,
              "confidenceIntervalHalfWidth": 5.01
            },
            "evidenceExcerpt": "Scale leaderboard snapshot: 63.17% resolve rate with reported +/-5.01 percentage points at xHigh. 124 tasks from 11 production repositories. All rubric criteria must pass, with Opus 4.5 as judge. Sample count is tasks; total trials are not asserted from another evaluator methodology."
          },
          {
            "id": "deepswe-owner-claude-opus-5-max-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 74,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 118000,
              "costPerTaskUsd": 11.84
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 4,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 74% with reported +/-4 percentage points; 113 tasks; average cost USD 11.84, displayed output 118k and 99 steps. These are rounded leaderboard measurements, not a new local evaluation."
          },
          {
            "id": "aa-ii-v43-opus5-xhigh-2026-09-12",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 50,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://artificialanalysis.ai/models/claude-opus-5-xhigh",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 4.88
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "Artificial Analysis Intelligence Index evaluation harnesses",
              "sampleNotes": "Composite Intelligence Index v4.3; index points are not percent accuracy and are not interchangeable with older index versions.",
              "costBasis": "Artificial Analysis weighted cost per Intelligence Index v4.3 task at this snapshot.",
              "fallbackPolicy": "Production fallback behavior is part of the evaluated API configuration; exact fallback share not provided on this model page."
            },
            "evidenceExcerpt": "Current AA v4.3 snapshot: Opus 5 xHigh scores 50 index points at USD 4.88 per weighted index task. Cost is specific to this suite and cannot be compared directly with an unrelated coding benchmark."
          },
          {
            "id": "aa-ii-v43-opus5-max-2026-09-12",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 51,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://artificialanalysis.ai/models/claude-opus-5",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 5.86
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "Artificial Analysis Intelligence Index evaluation harnesses",
              "sampleNotes": "Composite Intelligence Index v4.3; index points are not percent accuracy and are not interchangeable with older index versions.",
              "costBasis": "Artificial Analysis weighted cost per Intelligence Index v4.3 task at this snapshot.",
              "fallbackPolicy": "Production fallback behavior is part of the evaluated API configuration; exact fallback share not provided on this model page."
            },
            "evidenceExcerpt": "Current AA v4.3 snapshot: Opus 5 max scores 51 index points at USD 5.86 per weighted index task. Cost is specific to this suite and cannot be compared directly with an unrelated coding benchmark."
          },
          {
            "id": "aa-ii-v43-opus5-high-2026-09-12",
            "benchmarkId": "artificial-analysis-intelligence-index-v4.3",
            "name": "Artificial Analysis Intelligence Index",
            "version": "4.3",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "high",
            "score": 48,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://artificialanalysis.ai/models/claude-opus-5-high",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "costPerTaskUsd": 3.61
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "Artificial Analysis Intelligence Index evaluation harnesses",
              "sampleNotes": "Composite Intelligence Index v4.3; index points are not percent accuracy and are not interchangeable with older index versions.",
              "costBasis": "Artificial Analysis weighted cost per Intelligence Index v4.3 task at this snapshot.",
              "fallbackPolicy": "Production fallback behavior is part of the evaluated API configuration; exact fallback share not provided on this model page."
            },
            "evidenceExcerpt": "Current AA v4.3 snapshot: Opus 5 high scores 48 index points at USD 3.61 per weighted index task. Cost is specific to this suite and cannot be compared directly with an unrelated coding benchmark."
          },
          {
            "id": "aa-coding-v15-claude-opus-5-max-2026-09-09",
            "benchmarkId": "artificial-analysis-coding-agent-index-v1.5-native-agents",
            "name": "Artificial Analysis Coding Agent Index 1.5",
            "version": "1.5 / September 9 comparison",
            "unit": "index points",
            "maximumScore": 100,
            "effort": "max",
            "score": 60,
            "publishedAt": "2026-09-09",
            "sourceUrl": "https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Claude Code",
              "taskCount": 303,
              "trialCount": 909,
              "attemptsPerTask": 3,
              "sampleNotes": "Scores are equal-weight component aggregates, not 909-trial pooled accuracy. Native-agent versions are not supplied in the article. No new composite was calculated here.",
              "additionalSourceUrls": [
                "https://artificialanalysis.ai/methodology/coding-agents-benchmarking/"
              ]
            },
            "evidenceExcerpt": "Artificial Analysis September 9 comparison: 60 Coding Agent Index points using Claude Code at max. Version 1.5 has 303 distinct tasks and 909 attempts across three equally weighted benchmark components."
          },
          {
            "id": "coderabbit-astra-2026-09-04-overall-coverage-claude-opus-5",
            "benchmarkId": "coderabbit-astra-2026-09-04-overall-coverage",
            "name": "CodeRabbit / overall actionable bug coverage",
            "version": "2026-09-04 internal evaluation / overall",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 50.2,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / overall",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "overall actionable bug coverage: 50.2% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "coderabbit-astra-2026-09-04-cross-file-coverage-claude-opus-5",
            "benchmarkId": "coderabbit-astra-2026-09-04-cross-file-coverage",
            "name": "CodeRabbit / cross-file actionable bug coverage",
            "version": "2026-09-04 internal evaluation / cross-file",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "unspecified",
            "score": 42.9,
            "publishedAt": "2026-09-04",
            "sourceUrl": "https://www.coderabbit.ai/blog/gpt-6-astra-code-review-evaluation",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit internal review pipeline / September 4 evaluation / cross-file",
              "sampleNotes": "PR count, known-issue count, repeat count, reasoning effort, judge identity and exact pipeline version were not published. Values transcribed from the labeled chart; compare only within this protocol and subset."
            },
            "evidenceExcerpt": "cross-file actionable bug coverage: 42.9% in CodeRabbit's September 4 evaluation. Percentage of labeled bugs surfaced through actionable findings; false-positive/comment precision is not reported."
          },
          {
            "id": "tb21-aa-claude-opus-5-max-2026-07-24",
            "benchmarkId": "terminal-bench-v2.1-aa-intelligence-harness",
            "name": "Terminal-Bench 2.1 / Artificial Analysis",
            "version": "2.1 / Artificial Analysis historical runs",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 89,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://artificialanalysis.ai/articles/opus-5",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "independentEvaluator",
              "sourcePublisher": "Artificial Analysis",
              "runnerOrganization": "Artificial Analysis",
              "dateBasis": "publishedDate",
              "harness": "Artificial Analysis evaluation harness; exact historical configuration unspecified",
              "taskCount": 89,
              "fallbackPolicy": "Production fallback enabled in the launch evaluation; exact per-benchmark contribution unspecified.",
              "sampleNotes": "Independent pre-release evaluation in collaboration with Anthropic. Article does not give exact trial count or error bars for this result."
            },
            "evidenceExcerpt": "Dated Artificial Analysis launch assessment reports 89% on Terminal-Bench 2.1 at max. This is an older suite and evaluator run; it must not be merged with TB4 or the official native-agent leaderboard."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-full-stream-precision-claude-opus-5-xhigh",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-full-stream-precision",
            "name": "CodeRabbit Opus 5 senior / full-stream comment precision",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% full-stream comment precision",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 28.6,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 166 actionable comments and 92 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share accepted after every scored post-pipeline comment class is included. Do not substitute actionable precision for this noisier stream. Senior xHigh: 28.6%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-full-stream-precision-claude-opus-5-high",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-full-stream-precision",
            "name": "CodeRabbit Opus 5 senior / full-stream comment precision",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% full-stream comment precision",
            "maximumScore": 100,
            "effort": "high",
            "score": 27.3,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 176 actionable comments and 91 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share accepted after every scored post-pipeline comment class is included. Do not substitute actionable precision for this noisier stream. Senior high: 27.3%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-full-stream-coverage-claude-opus-5-xhigh",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-full-stream-coverage",
            "name": "CodeRabbit Opus 5 senior / full-stream known-issue coverage",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 60.8,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 166 actionable comments and 92 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of known error patterns caught in any post-pipeline comment class, including outside-diff and low-confidence nitpicks. Senior xHigh: 60.8%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-full-stream-coverage-claude-opus-5-high",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-full-stream-coverage",
            "name": "CodeRabbit Opus 5 senior / full-stream known-issue coverage",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "high",
            "score": 62.8,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 176 actionable comments and 91 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of known error patterns caught in any post-pipeline comment class, including outside-diff and low-confidence nitpicks. Senior high: 62.8%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-actionable-precision-claude-opus-5-xhigh",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-actionable-precision",
            "name": "CodeRabbit Opus 5 senior / actionable comment precision",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% actionable-comment precision",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 39.3,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 166 actionable comments and 92 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of actionable post-pipeline comments accepted by the evaluation judge. Comment denominator differs from the known-issue coverage denominator. Senior xHigh: 39.3%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-actionable-precision-claude-opus-5-high",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-actionable-precision",
            "name": "CodeRabbit Opus 5 senior / actionable comment precision",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% actionable-comment precision",
            "maximumScore": 100,
            "effort": "high",
            "score": 35.6,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 176 actionable comments and 91 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of actionable post-pipeline comments accepted by the evaluation judge. Comment denominator differs from the known-issue coverage denominator. Senior high: 35.6%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-actionable-coverage-claude-opus-5-xhigh",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-actionable-coverage",
            "name": "CodeRabbit Opus 5 senior / actionable known-issue coverage",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "xHigh",
            "score": 55.2,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 166 actionable comments and 92 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of 96 known error patterns caught by an actionable comment; averaged across three repeats. Senior xHigh: 55.2%. Three-run average over 96 error patterns."
          },
          {
            "id": "coderabbit-opus5-2026-07-24-senior-actionable-coverage-claude-opus-5-high",
            "benchmarkId": "coderabbit-opus5-2026-07-24-senior-actionable-coverage",
            "name": "CodeRabbit Opus 5 senior / actionable known-issue coverage",
            "version": "2026-07-24 / senior reviewer / 96 patterns / three repeats",
            "unit": "% known-issue coverage",
            "maximumScore": 100,
            "effort": "high",
            "score": 55.6,
            "publishedAt": "2026-07-24",
            "sourceUrl": "https://www.coderabbit.ai/blog/opus-5-model-review",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {},
            "context": {
              "sourceKind": "benchmarkOwner",
              "runnerOrganization": "CodeRabbit",
              "dateBasis": "publishedDate",
              "observedAt": "2026-09-12",
              "sourcePublisher": "CodeRabbit",
              "harness": "CodeRabbit senior-reviewer profile / verification, deduplication and assertive filtering",
              "sampleNotes": "96 evaluation patterns; three complete configuration repeats. PR count and exact judge/pipeline revision not published. 176 actionable comments and 91 nitpicks are average output volumes, not confirmed unique defects.",
              "additionalSourceUrls": [
                "https://www.coderabbit.ai/content/assets/opus-5-results-table.png"
              ]
            },
            "evidenceExcerpt": "Share of 96 known error patterns caught by an actionable comment; averaged across three repeats. Senior high: 55.6%. Three-run average over 96 error patterns."
          }
        ],
        "taskStudies": [],
        "assessment": {
          "modelId": "claude-opus-5",
          "summary": "A well-supported choice for understanding large codebases and difficult repository changes. Its measured strengths justify considering it alongside Astra and Fable 5.1; maximum effort is expensive and does not win every coding benchmark.",
          "strengths": [
            {
              "text": "DeepSWE 1.1: Opus 5 at max resolves 74% of the 113-task suite in Datacurve's shared mini-swe-agent harness, versus 59% for Opus 4.8 at max. The reported average cost is also lower: $11.84 versus $13.22 per task.",
              "sourceUrls": [
                "https://deepswe.datacurve.ai/",
                "https://deepswe.datacurve.ai/changelog"
              ],
              "evidenceIds": [
                "deepswe-owner-claude-opus-5-max-2026-09-12",
                "deepswe-owner-claude-opus-4-8-max-2026-09-12"
              ]
            },
            {
              "text": "Scale's Codebase QnA board reports 63.17% at xHigh with Claude Code, compared with 59.95% for Fable 5.1 and 59.14% for Astra in their native agents. Opus has the highest point estimate of the three; the published intervals overlap.",
              "sourceUrls": [
                "https://labs.scale.com/leaderboard/sweatlas-qna"
              ],
              "evidenceIds": [
                "scale-qna-claude-opus-5-xhigh-2026-09-12"
              ]
            }
          ],
          "tradeoffs": [
            {
              "text": "Astra at xHigh also scores 74% on the same DeepSWE owner harness, with a reported $6.52 average cost. Equal rounded scores and overlapping error bars do not establish an Opus quality advantage.",
              "sourceUrls": [
                "https://deepswe.datacurve.ai/"
              ],
              "evidenceIds": [
                "deepswe-owner-gpt-6-astra-xhigh-2026-09-12"
              ]
            },
            {
              "text": "On the official Terminal-Bench 4.0 native-agent board, Opus 5/max scores 51.82%, Fable 5.1/max 57.88%, and Astra/max 58.18%. These compare model-plus-agent configurations, not isolated models.",
              "sourceUrls": [
                "https://www.tbench.ai/"
              ],
              "evidenceIds": [
                "tb40-official-claude-opus-5-max-2026-09-12",
                "tb40-official-claude-fable-5-1-max-2026-09-12",
                "tb40-official-gpt-6-astra-max-2026-09-12"
              ]
            }
          ],
          "effortGuidance": {
            "text": "Use xHigh as an evidence-backed candidate for codebase questions and test writing, where Scale publishes that exact configuration. Consider max for hard repository work, where DeepSWE measures it directly. For general reasoning, AA v4.3 reports high at 48 points/$3.61 per task, xHigh at 50/$4.88, and max at 51/$5.86; high is a reasonable cost-conscious starting point for that suite, not a measured coding default.",
            "sourceUrls": [
              "https://labs.scale.com/leaderboard/sweatlas-qna",
              "https://labs.scale.com/leaderboard/sweatlas-tw",
              "https://deepswe.datacurve.ai/",
              "https://artificialanalysis.ai/models/claude-opus-5-high",
              "https://artificialanalysis.ai/models/claude-opus-5-xhigh",
              "https://artificialanalysis.ai/models/claude-opus-5"
            ]
          },
          "gaps": [
            "No exact Opus 5 row was found in the inspected SWE-bench Verified, SWE-bench Pro or Scale Refactoring leaderboards. Its DeepSWE score is not a substitute refactoring score.",
            "The retrieved coding sources do not provide a controlled Opus 5 high/xHigh/max sweep with one harness and one pricing window.",
            "Local repository behavior, retry cost and subscription-limit consumption have not been measured by this public-source review. AGT is an additional environment-fit signal."
          ],
          "displayName": "Claude Opus 5"
        },
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 25,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-07-24T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "Claude Opus 4.8",
      "releaseDate": "2026-05-28",
      "releaseDateSource": "https://platform.claude.com/docs/en/release-notes/overview#may-28-2026",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [
          {
            "id": "tb40-official-claude-opus-4-8-max-2026-09-12",
            "benchmarkId": "terminal-bench-v4.0-official-native-agents",
            "name": "Terminal-Bench 4.0 / official native-agent leaderboard",
            "version": "4.0",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 23.64,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/4/leaderboards/4-0-0/rows/952b4217-421f-4d14-849b-9968fcb2063b",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 269131,
              "costPerTaskUsd": 19.640182,
              "latencyMilliseconds": 5163000
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "officialLeaderboard",
              "sourcePublisher": "Terminal-Bench / Harbor",
              "dateBasis": "firstObservedPublicSnapshot",
              "sourceCreatedAt": "2026-08-27T18:30:27.559933+00:00",
              "sourceUpdatedAt": "2026-09-03T00:09:38.524394+00:00",
              "harness": "Claude Code",
              "taskCount": 66,
              "trialCount": 330,
              "confidenceIntervalHalfWidth": 3.56,
              "confidenceIntervalLevel": 0.95,
              "totalCostUsd": 6481.26,
              "costBasis": "Publisher total USD divided by all trials, including failures. Observed ledger pricing; not recomputed at current tariff.",
              "sampleNotes": "78/330 successful attempts. Dataset task count is distinct from trial count. 330/66=5 is a derived ratio; balanced per-task allocation was not independently checked. Runner, agent version and fallback settings are not supplied. metadata.date is the model release date and is not used as the run or publication date.",
              "additionalSourceUrls": [
                "https://www.tbench.ai/",
                "https://ofhuhcpkvzjlejydnvyd.supabase.co/functions/v1/leaderboard-read"
              ]
            },
            "evidenceExcerpt": "Public snapshot 2026-09-12: 78/330 attempts resolved (23.64%, 95% CI half-width 3.56 percentage points), using Claude Code at max. PublishedAt is first observed availability, not an asserted run date. Total cost USD 6481.26; secondary metrics are per attempt and output tokens are rounded."
          },
          {
            "id": "deepswe-owner-claude-opus-4-8-max-2026-09-12",
            "benchmarkId": "deepswe-v1.1-datacurve-mini-swe-agent",
            "name": "DeepSWE 1.1 / benchmark-owner harness",
            "version": "1.1 / Datacurve mini-swe-agent",
            "unit": "% resolved",
            "maximumScore": 100,
            "effort": "max",
            "score": 59,
            "publishedAt": "2026-09-12",
            "sourceUrl": "https://deepswe.datacurve.ai/",
            "confidence": "thirdParty",
            "evidenceStatus": "unknown",
            "sampleContext": null,
            "secondaryMetrics": {
              "outputTokensPerTask": 135000,
              "costPerTaskUsd": 13.22
            },
            "context": {
              "observedAt": "2026-09-12",
              "sourceKind": "benchmarkOwner",
              "sourcePublisher": "Datacurve",
              "runnerOrganization": "Datacurve",
              "dateBasis": "firstObservedPublicSnapshot",
              "harness": "mini-swe-agent",
              "taskCount": 113,
              "reportedErrorHalfWidth": 2,
              "costBasis": "Owner leaderboard average cost per task, captured on 2026-09-12; historical tariff window unspecified.",
              "sampleNotes": "Shared bash toolkit and prompt. The source does not specify the confidence level, number of attempts, or exact harness revision. Displayed k-token values are rounded. The current table says updated September 3; Opus 5 results were first added July 25, but current cost and score values are only asserted for this snapshot.",
              "additionalSourceUrls": [
                "https://deepswe.datacurve.ai/changelog",
                "https://deepswe.datacurve.ai/blog/deepswe"
              ]
            },
            "evidenceExcerpt": "Owner leaderboard snapshot: 59% with reported +/-2 percentage points; 113 tasks; average cost USD 13.22, displayed output 135k and 120 steps. These are rounded leaderboard measurements, not a new local evaluation."
          }
        ],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-4-8",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 25,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-05-28T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Opus 4.7",
      "releaseDate": "2026-04-16",
      "releaseDateSource": "https://platform.claude.com/docs/en/release-notes/overview#april-16-2026",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-4-7",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 25,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-04-16T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Opus 4.6",
      "releaseDate": "2026-02-05",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/opus-4-6/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-4-6",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 25,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-02-05T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Opus 4.5",
      "releaseDate": "2025-11-24",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/opus-4-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-opus-4-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 25,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-11-24T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "max"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Sonnet 4.6",
      "releaseDate": "2026-02-17",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/sonnet-4-6/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-sonnet-4-6",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "balanced",
      "costClass": "standard",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 3,
        "outputPerMTok": 15,
        "cachedInputPerMTok": 0.3,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-02-17T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "capable",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "ideal",
        "mechanicalChore": "capable",
        "docEdit": "capable",
        "research": "ideal",
        "review": null,
        "htmlUiImplementation": "capable",
        "sourceCodeReview": "ideal",
        "securityAssessment": "underpowered",
        "redundancyDetection": "capable",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Sonnet 4.5",
      "releaseDate": "2025-09-29",
      "releaseDateSource": "https://platform.claude.com/docs/en/release-notes/overview#september-29-2025",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the price catalog but not qualified by the authoritative routing policy.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-sonnet-4-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "balanced",
      "costClass": "standard",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 3,
        "outputPerMTok": 15,
        "cachedInputPerMTok": 0.3,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-09-29T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high",
        "xHigh",
        "max"
      ],
      "suitability": {
        "heavyDesign": "capable",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "ideal",
        "mechanicalChore": "capable",
        "docEdit": "capable",
        "research": "ideal",
        "review": null,
        "htmlUiImplementation": "capable",
        "sourceCodeReview": "ideal",
        "securityAssessment": "underpowered",
        "redundancyDetection": "capable",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "Claude Haiku 4.5",
      "releaseDate": "2025-10-15",
      "releaseDateSource": "https://platform.claude.com/docs/en/models/haiku-4-5/overview",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Retained Anthropic vendor override for explicit operator pins, subject to task correctness floors. No automatic equivalence to Sol/xhigh is established.",
        "workflowRoles": [
          "coreTask"
        ],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "claude-haiku-4-5",
      "vendor": "anthropic",
      "cli": "claude",
      "tier": "light",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 1,
        "outputPerMTok": 5,
        "cachedInputPerMTok": 0.1,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-10-15T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium"
      ],
      "suitability": {
        "heavyDesign": "underpowered",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "underpowered",
        "mechanicalChore": "ideal",
        "docEdit": "ideal",
        "research": "underpowered",
        "review": null,
        "htmlUiImplementation": "underpowered",
        "sourceCodeReview": "underpowered",
        "securityAssessment": "underpowered",
        "redundancyDetection": "underpowered",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "selectable",
      "evidenceStatus": "provisional",
      "provisional": true
    },
    {
      "displayName": "GPT-5.5",
      "releaseDate": "2026-04-23",
      "releaseDateSource": "https://openai.com/index/introducing-gpt-5-5/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the CLI/price catalog but not a route in policy version 2026-07-24.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.5",
      "vendor": "openai",
      "cli": "codex",
      "tier": "balanced",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 5,
        "outputPerMTok": 30,
        "cachedInputPerMTok": 0.5,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-04-24T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "capable",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "ideal",
        "mechanicalChore": "capable",
        "docEdit": "capable",
        "research": "ideal",
        "review": null,
        "htmlUiImplementation": "capable",
        "sourceCodeReview": "ideal",
        "securityAssessment": "underpowered",
        "redundancyDetection": "capable",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "GPT-5.5 Pro",
      "releaseDate": "2026-04-23",
      "releaseDateSource": "https://openai.com/index/introducing-gpt-5-5/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the API/price catalog but not a route in policy version 2026-07-24.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.5-pro",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 30,
        "outputPerMTok": 180,
        "cachedInputPerMTok": 30,
        "cachedInputUsesInputFallback": true,
        "validFromUtc": "2026-04-24T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "GPT-5.5 Cyber (Preview)",
      "releaseDate": "2026-05-07",
      "releaseDateSource": "https://openai.com/index/gpt-5-5-with-trusted-access-for-cyber/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Limited-preview model for separately approved defenders; catalog coverage does not make it a selectable route in policy version 2026-07-24.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5.5-cyber-preview",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "premium",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 12.5,
        "outputPerMTok": 75.0,
        "cachedInputPerMTok": 1.25,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2026-05-07T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": true,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "restricted",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "GPT-5",
      "releaseDate": "2025-08-07",
      "releaseDateSource": "https://openai.com/index/introducing-gpt-5/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the CLI/price catalog but not a route in policy version 2026-07-24.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5",
      "vendor": "openai",
      "cli": "codex",
      "tier": "balanced",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 1.25,
        "outputPerMTok": 10,
        "cachedInputPerMTok": 0.125,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-08-07T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xHigh"
      ],
      "suitability": {
        "heavyDesign": "capable",
        "planning": "underpowered",
        "decisionMaking": "underpowered",
        "feature": "ideal",
        "mechanicalChore": "capable",
        "docEdit": "capable",
        "research": "ideal",
        "review": null,
        "htmlUiImplementation": "capable",
        "sourceCodeReview": "ideal",
        "securityAssessment": "underpowered",
        "redundancyDetection": "capable",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    },
    {
      "displayName": "GPT-5 Codex",
      "releaseDate": "2025-09-15",
      "releaseDateSource": "https://openai.com/index/introducing-upgrades-to-codex/",
      "policy": {
        "version": "2026-09-25",
        "evidenceAsOfDate": "2026-09-25",
        "reason": "Known to the CLI/price catalog but not a route in policy version 2026-07-24.",
        "workflowRoles": [],
        "routes": [],
        "fallbacks": []
      },
      "evidence": {
        "external": [],
        "taskStudies": [],
        "assessment": null,
        "taskStudiesAsOfDate": "2026-09-25"
      },
      "modelId": "gpt-5-codex",
      "vendor": "openai",
      "cli": "codex",
      "tier": "frontier",
      "costClass": "economy",
      "price": {
        "status": "resolved",
        "currency": "USD",
        "inputPerMTok": 1.25,
        "outputPerMTok": 10,
        "cachedInputPerMTok": 0.125,
        "cachedInputUsesInputFallback": false,
        "validFromUtc": "2025-09-23T00:00:00Z",
        "unconfirmed": false
      },
      "effortLevels": [
        "low",
        "medium",
        "high"
      ],
      "suitability": {
        "heavyDesign": "ideal",
        "planning": "ideal",
        "decisionMaking": "ideal",
        "feature": "capable",
        "mechanicalChore": "overkill",
        "docEdit": "overkill",
        "research": "capable",
        "review": null,
        "htmlUiImplementation": "ideal",
        "sourceCodeReview": "overkill",
        "securityAssessment": "ideal",
        "redundancyDetection": "ideal",
        "graphicalQualityJudgment": null,
        "consistencyChecking": null
      },
      "restricted": false,
      "deprecated": false,
      "costUnconfirmed": false,
      "selectionStatus": "unsupported",
      "evidenceStatus": "unknown",
      "provisional": false
    }
  ]
}
