{
  "schema": "cellara.proof.pipeline-cohort-summary/v1",
  "generated_at": "2026-08-10T20:30:05.388Z",
  "benchmark_family": "pipeline-reality-check-2026-08-10-v3",
  "experiment": "Three opaque-ID/order variants of one synthetic 389-record sales-pipeline pack.",
  "treatment": "Fixture-authored fixed-policy current view plus governing evidence over a hash-bound source archive.",
  "pair_count": 6,
  "cellara_exact": 6,
  "raw_exact": 3,
  "surfaces": [
    {
      "provider": "grok",
      "model": "grok-4.5-build",
      "product_surface": "Grok Build CLI",
      "client_version": "1.0.0",
      "variants": 3,
      "cellara_exact": 3,
      "raw_exact": 2,
      "cellara_faster": 3,
      "median_elapsed_time_reduction_percent": 51.2,
      "elapsed_time_reduction_range_percent": [
        24.1,
        73.4
      ],
      "median_reported_token_reduction_percent": 34.3,
      "reported_token_reduction_range_percent": [
        -138.7,
        78.4
      ],
      "median_reported_cost_reduction_percent": 27.9,
      "accuracy_wins": 1
    },
    {
      "provider": "codex",
      "model": "gpt-5.4-mini",
      "product_surface": "Codex CLI",
      "client_version": "0.139.0",
      "variants": 3,
      "cellara_exact": 3,
      "raw_exact": 1,
      "cellara_faster": 3,
      "median_elapsed_time_reduction_percent": 82.1,
      "elapsed_time_reduction_range_percent": [
        80.5,
        85.2
      ],
      "median_reported_token_reduction_percent": 84.3,
      "reported_token_reduction_range_percent": [
        43,
        84.8
      ],
      "median_reported_cost_reduction_percent": null,
      "accuracy_wins": 2
    }
  ],
  "strongest_exact_run": {
    "receipt": "grok-grok-4-5-build-v02.json",
    "raw_committed_total_usd": 113000,
    "correct_committed_total_usd": 211000,
    "raw_committed_dollar_error": 98000,
    "raw_fatal_false_commit_ids": [
      "D-08"
    ],
    "cellara_committed_total_usd": 211000,
    "cellara_committed_dollar_error": 0
  },
  "page_decision": {
    "descriptive_page_earned": true,
    "general_advantage_claim_earned": false,
    "reason": "Six exact paired receipts support a bounded, surface-specific descriptive page; one semantic fixture family does not support a general performance claim."
  },
  "limitations": [
    "Synthetic records only; no customer data or real forecast outcome.",
    "Three randomized presentations of one semantic fixture, not three independent business cases.",
    "The fixed-policy current view and links are fixture-authored; automatic ingestion, linking, and policy inference were not tested.",
    "CLI surfaces only. These results do not stand for ChatGPT or Grok consumer products.",
    "Runs overlapped in time; elapsed-time comparisons are exact within each pair but are not general latency benchmarks.",
    "Grok efficiency was variable, and one Cellara run cited less of the expected evidence despite an exact decision."
  ],
  "artifacts": [
    {
      "path": "grok-grok-4-5-build-v01.json",
      "sha256": "677959638425c0c7598f1dee64da6a3e87f72b1f9f1addd2373c4c0ee9daa31a",
      "bytes": 19456
    },
    {
      "path": "grok-grok-4-5-build-v02.json",
      "sha256": "d0ee21790870c51b9ed2f94210f41b8aae7c0875cef2591157ccc3dd047c51d3",
      "bytes": 18353
    },
    {
      "path": "grok-grok-4-5-build-v03.json",
      "sha256": "218dd6c51f1e9a1be99c6b0f838795f08e2c35cd3db3c8821cebeec6553df1af",
      "bytes": 19557
    },
    {
      "path": "codex-gpt-5-4-mini-v01.json",
      "sha256": "fdfe4212cafe72f901be069e640c5686478e4803b3799d5094350f8e4dd35921",
      "bytes": 18230
    },
    {
      "path": "codex-gpt-5-4-mini-v02.json",
      "sha256": "e442baf3ed5d4e75d0f72cddfa3030112587e3b4dd740b9a1c527e5b32aa53ac",
      "bytes": 17923
    },
    {
      "path": "codex-gpt-5-4-mini-v03.json",
      "sha256": "8879df0df78f1441746b901ee4d4eb325c5f8841857474bcdb647faddd90a4d6",
      "bytes": 17082
    }
  ]
}
