{
  "schemaVersion": 2,
  "evidenceDate": "2026-09-04",
  "benchmark": {
    "name": "Laminarity vs OpenRouter controlled replay",
    "promptCount": 500,
    "fixedArmCount": 25,
    "trainingArmCount": 26,
    "scoreScale": "0_to_1",
    "costUnit": "usd_per_1000_prompts",
    "evaluationType": "cross_fitted_historical_quality_on_common_openrouter_prompt_pairs",
    "benchmarks": [
      {
        "id": "aime",
        "label": "AIME",
        "promptCount": 24,
        "avengersScore": 0.9166666666666666,
        "autoScore": 0.875,
        "autoBetaScore": 0.2916666666666667
      },
      {
        "id": "arena_hard_v2",
        "label": "Arena-Hard v2",
        "promptCount": 15,
        "avengersScore": 0.7833333333333333,
        "autoScore": 0.8166666666666667,
        "autoBetaScore": 0.8
      },
      {
        "id": "bbeh",
        "label": "BBEH",
        "promptCount": 37,
        "avengersScore": 0.05405405405405406,
        "autoScore": 0.13513513513513514,
        "autoBetaScore": 0.02702702702702703
      },
      {
        "id": "evalplus",
        "label": "EvalPlus",
        "promptCount": 67,
        "avengersScore": 0.8805970149253731,
        "autoScore": 0.8059701492537313,
        "autoBetaScore": 0.8208955223880597
      },
      {
        "id": "finqa",
        "label": "FinQA",
        "promptCount": 29,
        "avengersScore": 0,
        "autoScore": 0,
        "autoBetaScore": 0
      },
      {
        "id": "hlce",
        "label": "HLCE",
        "promptCount": 49,
        "avengersScore": 0.40816326530612246,
        "autoScore": 0.02040816326530612,
        "autoBetaScore": 0.02040816326530612
      },
      {
        "id": "hle_text_mc",
        "label": "HLE Text MC",
        "promptCount": 27,
        "avengersScore": 0.5555555555555556,
        "autoScore": 0.2222222222222222,
        "autoBetaScore": 0.07407407407407407
      },
      {
        "id": "ifeval",
        "label": "IFEval",
        "promptCount": 56,
        "avengersScore": 0.9821428571428571,
        "autoScore": 0.8928571428571429,
        "autoBetaScore": 0.9107142857142857
      },
      {
        "id": "ifeval_hard",
        "label": "IFEval Hard",
        "promptCount": 34,
        "avengersScore": 0.9117647058823529,
        "autoScore": 0.7941176470588235,
        "autoBetaScore": 0.7941176470588235
      },
      {
        "id": "math_500",
        "label": "MATH 500",
        "promptCount": 43,
        "avengersScore": 0.9302325581395349,
        "autoScore": 0.9302325581395349,
        "autoBetaScore": 0.8837209302325582
      },
      {
        "id": "mmlu_pro",
        "label": "MMLU-Pro",
        "promptCount": 56,
        "avengersScore": 0.7678571428571429,
        "autoScore": 0.6964285714285714,
        "autoBetaScore": 0.6785714285714286
      },
      {
        "id": "olympiadbench",
        "label": "OlympiadBench",
        "promptCount": 37,
        "avengersScore": 0,
        "autoScore": 0,
        "autoBetaScore": 0.4594594594594595
      },
      {
        "id": "simpleqa_verified",
        "label": "SimpleQA Verified",
        "promptCount": 13,
        "avengersScore": 0.6923076923076923,
        "autoScore": 0.23076923076923078,
        "autoBetaScore": 0.23076923076923078
      },
      {
        "id": "swebench_verified",
        "label": "SWE-bench Verified",
        "promptCount": 0,
        "avengersScore": null,
        "autoScore": null,
        "autoBetaScore": null
      },
      {
        "id": "wildbench",
        "label": "WildBench",
        "promptCount": 13,
        "avengersScore": 0.6999999926640437,
        "autoScore": 0.5999999986245081,
        "autoBetaScore": 0.5538461529291593
      }
    ]
  },
  "laminarity": {
    "alpha": 0.8,
    "score": 0.6336999998092652,
    "exact": null,
    "costPer1000Usd": 1.147517425
  },
  "baseline": {
    "armId": "openai/gpt-5.5",
    "modelId": "openai/gpt-5.5",
    "name": "GPT-5.5",
    "provider": "openai",
    "score": 0.5693999999165535,
    "exact": null,
    "costPer1000Usd": 2.87816,
    "availability": "available",
    "routing": "automatic"
  },
  "comparison": {
    "costSavingPercent": 60.13017257553437,
    "baselineCostMultiple": 2.508162348820106,
    "scoreDeltaPercentagePoints": 6.429999989271162,
    "relativeScoreGainPercent": 11.292588672661562,
    "pairedScoreDelta95Ci": null
  },
  "links": {
    "research": "/research",
    "quickstart": "/docs/guides/quickstart",
    "evidence": "/research/home-evidence.json"
  },
  "method": {
    "score": "Mean objective score from frozen historical answers on 500 common prompt pairs. Laminarity uses the cross-fitted default policy, alpha=0.8. Exact-match accuracy is not reported for this replay.",
    "cost": "Historical realized batch model cost in USD per 1,000 prompts. Live probe acquisition cost is separate. These are benchmark costs, not current token price quotes or guaranteed production savings.",
    "comparison": "The same default Laminarity point is compared with each fixed model. Cost savings use the fixed-model cost as the denominator. Homepage quality change is relative; the score difference is also supplied in percentage points."
  },
  "provenance": {
    "repositoryUrl": "https://github.com/laminarityai/laminarity-frontend",
    "importedRevision": "c66a886e1abe28892f77dd44ae17da1d284c6770",
    "snapshotPath": "src/lib/research-data.json",
    "snapshotSha256": "a7e9ee062625f698063a6d4634e0787272713127f0410e7cabfa56d0e1fbf8f0",
    "sourceUrl": "https://github.com/laminarityai/laminarity-frontend/blob/c66a886e1abe28892f77dd44ae17da1d284c6770/src/lib/research-data.json"
  },
  "notes": [
    "This September controlled replay replaces the August 997-prompt panel. The two evaluations have different workloads and methods and must not be combined.",
    "The 25 benchmark models are separate from the production catalog. GPT-6 Astra and Claude Fable 5.1 have sparse research coverage and are not measured in this panel.",
    "OpenRouter Auto and Auto Beta act as selectors in this replay; their response text is not scored. The default Laminarity point has higher quality and higher cost than both selectors.",
    "Matched-quality OpenRouter operating points were selected after evaluation. They are descriptive test-set comparisons, not the default policy or a guarantee of equivalent quality."
  ]
}
