{
  "schemaVersion": "1.0.0",
  "resultId": "builderbench-pilot-2026-09-24",
  "slug": "2026-09-24-pilot",
  "title": "BuilderBench pilot: Opus 5.5 vs GPT-6 Sol",
  "runDate": "2026-09-24",
  "displayDate": "24 September 2026",
  "timezone": "Australia/Perth",
  "status": "cleared_for_web",
  "benchmarkVersion": "v1 pilot",
  "packVersion": "1.0.0",
  "packSha256": "0d88351719d3de54bb6b0980e7e4d6d6a24a845590e6dcd7875099178f411fff",
  "methodologyVersion": "v1",
  "scope": "One owner-scored pilot of each model, harness and tool setup. This is not a universal model ranking or a statistical claim.",
  "durationSeconds": 14400,
  "scoringWeights": {
    "quality": 70,
    "autonomy": 20,
    "proactivity": 10
  },
  "settings": {
    "reasoning": "Extra High (xhigh)",
    "speed": "Standard",
    "billing": "Existing subscriptions"
  },
  "review": {
    "reviewer": "Lennox Saint",
    "sampleSizePerSetup": 1,
    "scoresLocked": true,
    "identitiesRevealed": true,
    "scoreDigestSha256": "bec5cbd7c21e72815fb901521185ac415081f68f3228892734987b03411b3dd6"
  },
  "candidates": [
    {
      "name": "Opus 5.5",
      "harness": "Claude",
      "modelVersion": null,
      "harnessVersion": null,
      "quality": 61.8,
      "autonomy": 80,
      "proactivity": 90,
      "overall": 68.26,
      "tasks": [
        { "label": "Writing", "score": 49, "weight": 20 },
        { "label": "Video editing", "score": 40, "weight": 15 },
        { "label": "Frontend design", "score": 60, "weight": 15 },
        { "label": "Motion graphics", "score": 60, "weight": 10 },
        { "label": "Browser & computer", "score": 70, "weight": 15 },
        { "label": "Research & decisions", "score": 70, "weight": 10 },
        { "label": "Intake automation", "score": 95, "weight": 10 },
        { "label": "Founder conversation", "score": 80, "weight": 5 }
      ],
      "cost": {
        "actualIncrementalSpendUsd": null,
        "apiEquivalentEstimateUsd": 49.672092,
        "coverage": "Provider-reported API estimates for the recorded task rows."
      }
    },
    {
      "name": "GPT-6 Sol",
      "harness": "Codex",
      "modelVersion": null,
      "harnessVersion": null,
      "quality": 46.5,
      "autonomy": 40,
      "proactivity": 50,
      "overall": 45.55,
      "tasks": [
        { "label": "Writing", "score": 30, "weight": 20 },
        { "label": "Video editing", "score": 10, "weight": 15 },
        { "label": "Frontend design", "score": 40, "weight": 15 },
        { "label": "Motion graphics", "score": 30, "weight": 10 },
        { "label": "Browser & computer", "score": 90, "weight": 15 },
        { "label": "Research & decisions", "score": 55, "weight": 10 },
        { "label": "Intake automation", "score": 90, "weight": 10 },
        { "label": "Founder conversation", "score": 40, "weight": 5 }
      ],
      "cost": {
        "actualIncrementalSpendUsd": null,
        "apiEquivalentEstimateUsd": 12.7549352,
        "coverage": "Partial recorded usage only; unknown attempts and shared overhead are excluded."
      }
    }
  ],
  "costNote": "API-equivalent estimates are not bills. Actual incremental spend is unknown. The methods and coverage differ, so these figures do not support an exact cost ratio.",
  "methodologyNote": "The owner scored quality, autonomy and proactivity after a blind review. Quality is the weighted score across eight task families. The score lock happened before identities were revealed.",
  "limitations": [
    "This was one pilot scored by one owner, not a universal ranking.",
    "The score belongs to each full setup: model, harness, tools and settings.",
    "The two cost estimates use different methods and coverage.",
    "Actual incremental spend remains unknown because the run used existing subscriptions.",
    "Complete normal-speed watched-and-listened review of both full video outputs was not established."
  ],
  "assets": {
    "resultGraphic": {
      "src": "/builderbench/2026-09-24-pilot/pilot-result.png",
      "sha256": "4bd656d5ca66a42d7bb99125667b0b3ebb9217785db2c8d7b6b4561199445071",
      "width": 1600,
      "height": 900,
      "alt": "BuilderBench pilot result: Opus 5.5 scored 68.26 and GPT-6 Sol scored 45.55, with API-equivalent estimate caveats."
    }
  },
  "sourceResultsSha256": "6f212a2c460939063e690de0d6ae44276a17191e90bc77d3ec3f074e6565e518",
  "integrity": {
    "algorithm": "sha256",
    "digest": "8b6a406aa48fefa222fe10a1ea8745e2c0693c34cb87c814cacf4b16b0153e1b"
  }
}
