{
  "schemaVersion": 3,
  "benchmarkId": "devshift-operational-twin-2026-08-13",
  "executedAt": "2026-08-13",
  "publishedAt": "2026-08-14T08:04:14Z",
  "methodology": {
    "orchestrator": "Cursor Agent CLI 2026.08.11-e8db854",
    "samePrompt": true,
    "parallel": true,
    "processStartSkewMs": 0.13925,
    "promptWriteSkewMs": 0.132791,
    "machine": "MacBookPro18,2, Apple silicon, 64 GiB, 10 logical CPUs",
    "subscription": "Cursor Individual Ultra",
    "oneRunLimitation": true,
    "blindHumanReview": {
      "status": "not-completed",
      "reviewerCount": 0,
      "winner": null
    }
  },
  "pricing": {
    "kind": "cursor-rate-card-equivalent",
    "currency": "USD",
    "individualUltraTokenFeeUsd": 0,
    "comparisonBasis": "standard-list-excluding-temporary-promotions",
    "grokCacheReadUsdPerMillionTokens": 0.5,
    "cursorBench": {
      "source": "https://cursor.com/cursorbench",
      "grokExtraHighUsdPerTask": 2.81,
      "solMaxUsdPerTask": 5.69
    },
    "grokRateSource": "https://cursor.com/docs/models/grok-4-6",
    "solRateSource": "https://cursor.com/docs/models/gpt-5-6-sol",
    "note": "Usage equivalents use published standard list rates and exclude temporary promotions. They are not additional cash charges beyond the Individual Ultra subscription."
  },
  "results": [
    {
      "model": "Grok 4.6 Extra High",
      "modelId": "cursor-grok-4.6-xhigh",
      "cliModel": "grok-4.6[effort=xhigh,fast=false]",
      "route": "https://www.devshift.biz/en/labs/operational-twin/grok-4-6-extra-high",
      "contestantCommit": "6449c394a25ba66ae9a8324554c2457cec104782",
      "execution": {
        "kind": "one-shot",
        "status": "completed",
        "wallDurationMs": 5280186.456417,
        "agentDurationMs": 5273709,
        "completedModelTurns": 185,
        "completedToolCalls": 612,
        "fileEdits": 152,
        "deniedActions": 5
      },
      "usage": {
        "inputTokens": 1300427,
        "cacheReadTokens": 28980224,
        "cacheWriteTokens": 0,
        "outputTokens": 255553,
        "totalTokens": 30536204,
        "standardListRatedUsageUsd": 18.624284
      },
      "checks": {
        "publicAdapter": "pass",
        "hiddenEvaluator": {
          "status": "fail",
          "passed": 122452,
          "failed": 826,
          "total": 123278,
          "dominantFailure": "Committed order fulfillment delta was not conserved"
        },
        "rawTypecheck": "pass",
        "rawLint": "fail",
        "rawProductionBuild": "pass",
        "objectiveRerankJourney": "fail",
        "webgl2Production": "pass",
        "required2dFallback": "fail",
        "ordinaryRouteHeavyPackageLeakage": "none"
      },
      "eligibility": {
        "status": "DNF",
        "reasons": [
          "hidden business-logic evaluator failed",
          "objective controls did not materially rerank plans",
          "required 2D fallback was not exposed after WebGL loss",
          "raw lint failed",
          "strict isolated-profile evidence was invalidated by Cursor runtime HOME writes"
        ]
      }
    },
    {
      "model": "Claude Opus 5 Max",
      "modelId": "claude-opus-5-thinking-max",
      "cliModel": "claude-opus-5[thinking=true,context=300k,effort=max,fast=false]",
      "route": "https://www.devshift.biz/en/labs/operational-twin/opus-5-max",
      "contestantCommit": "a00965d6560b477a834d21da0a283d9ff1736c84",
      "execution": {
        "kind": "failed-one-shot-plus-session-recovery",
        "status": "recovered-not-comparable",
        "combinedWallDurationMs": 4525846.423042,
        "completedModelTurns": 114,
        "completedToolCalls": 240,
        "fileEdits": 59,
        "deniedActions": 7,
        "segments": [
          {
            "kind": "original-parallel-run",
            "status": "failed",
            "wallDurationMs": 3139940.033084,
            "completedModelTurns": 61,
            "completedToolCalls": 135,
            "terminalUsage": null,
            "failure": "resource_exhausted"
          },
          {
            "kind": "exact-session-resume",
            "status": "completed",
            "wallDurationMs": 1385906.389958,
            "agentDurationMs": 1378948,
            "completedModelTurns": 53,
            "completedToolCalls": 105,
            "usage": {
              "inputTokens": 156,
              "cacheReadTokens": 14624998,
              "cacheWriteTokens": 257827,
              "outputTokens": 66700,
              "totalTokens": 14949681,
              "ratedUsageUsd": 10.59219775
            }
          }
        ]
      },
      "usage": {
        "originalRun": null,
        "recoveryRatedUsageUsd": 10.59219775,
        "recoveryModelTurns": 53,
        "combinedRatedUsageUsd": 22.783218,
        "combinedRatedUsageMethod": "turn-normalized-reconstruction",
        "unavailableReason": "The failed original run emitted no terminal usage record; the combined figure is reconstructed from measured recovery usage per model turn"
      },
      "checks": {
        "publicAdapter": "pass",
        "hiddenEvaluator": {
          "status": "fail",
          "passed": 53437,
          "failed": 177,
          "total": 53614,
          "dominantFailure": "Committed order fulfillment delta was not conserved"
        },
        "rawTypecheck": "pass",
        "rawLint": "fail",
        "rawProductionBuild": "pass",
        "objectiveRerankJourney": "fail",
        "webgl2Production": "pass",
        "required2dFallback": "fail",
        "ordinaryRouteHeavyPackageLeakage": "none"
      },
      "eligibility": {
        "status": "DNF",
        "reasons": [
          "original parallel run failed without terminal usage",
          "recovery completion is not comparable to a one-shot run",
          "hidden business-logic evaluator failed",
          "objective controls did not materially rerank plans",
          "required 2D fallback was not exposed after WebGL loss",
          "raw lint failed",
          "strict isolated-profile evidence was invalidated by Cursor runtime HOME writes"
        ]
      }
    },
    {
      "model": "GPT-5.6 SOL Max",
      "modelId": "gpt-5.6-sol-max",
      "cliModel": "gpt-5.6-sol[effort=max,fast=false]",
      "route": "https://www.devshift.biz/en/labs/operational-twin/gpt-5-6-sol-max",
      "contestantCommit": "0d738875ae2940c4c74c882828628538edadc6a7",
      "execution": {
        "kind": "one-shot",
        "status": "completed",
        "wallDurationMs": 2685942.038166,
        "agentDurationMs": 2679194,
        "completedModelTurns": 88,
        "completedToolCalls": 238,
        "fileEdits": 35,
        "deniedActions": 4
      },
      "usage": {
        "inputTokens": 510,
        "cacheReadTokens": 23562162,
        "cacheWriteTokens": 371456,
        "outputTokens": 110531,
        "totalTokens": 24044659,
        "standardListRatedUsageUsd": 17.421161
      },
      "checks": {
        "publicAdapter": "pass",
        "hiddenEvaluator": {
          "status": "fail",
          "passed": 32287,
          "failed": 182,
          "total": 32469,
          "dominantFailure": "Committed order fulfillment delta was not conserved"
        },
        "rawTypecheck": "fail",
        "rawLint": "fail",
        "rawProductionBuild": "fail",
        "deploymentCompatibilityRepair": "two React state types widened without behavior changes",
        "objectiveRerankJourney": "fail",
        "webgl2Production": "pass",
        "required2dFallback": "fail",
        "ordinaryRouteHeavyPackageLeakage": "none"
      },
      "eligibility": {
        "status": "DNF",
        "reasons": [
          "raw production build failed",
          "hidden business-logic evaluator failed",
          "objective controls did not materially rerank plans",
          "required 2D fallback was not exposed after WebGL loss",
          "raw lint failed",
          "forbidden model fallback and command attempts were blocked by the harness",
          "strict isolated-profile evidence was invalidated by Cursor runtime HOME writes"
        ]
      }
    }
  ],
  "conclusion": {
    "eligibleWinner": null,
    "lowestComparableOneShotStandardListUsage": "GPT-5.6 SOL Max",
    "fastestComparableOneShot": "GPT-5.6 SOL Max",
    "note": "All three outputs are public as preserved artifacts, but all three are DNF under the frozen evaluator. Under standard list rates, Grok's 28,980,224 cache-read tokens made its one-shot usage equivalent higher than SOL's."
  },
  "deployment": {
    "project": "devshift-website-next",
    "deploymentId": "dpl_9w6dmeyvZgvv5JS2ZycFRLY3FqJV",
    "sourceCommit": "596afcdd0a6fcec2a877c475cee4c360c31cfd92",
    "status": "READY",
    "liveVerifiedAt": "2026-08-14T08:04:14Z",
    "productionWebgl2": true,
    "productionRuntimeErrors": 0
  }
}
