{
  "schemaVersion": "remakebench.public-benchmark-result.v1",
  "resultId": "shrine-approach-unity-opus-5-max",
  "category": "Unity technical demo / benchmark result",
  "title": "Shrine Approach v1",
  "model": "Claude Opus 5 Max",
  "workflow": "Claude Code, Max effort",
  "runMode": "r_and_d",
  "requestCount": 1,
  "separatelyBilledSubagents": {
    "count": 7,
    "reportedTokens": 1888337,
    "tokenMix": "unavailable",
    "costUsd": null
  },
  "measurements": {
    "wallClockSeconds": 27696.5,
    "wallClockDisplay": "7h 41m 36.5s",
    "parentProcessedTokens": 327042040,
    "parentApiListPriceEquivalentUsd": 204.1729205,
    "displayCost": "$204.17+",
    "exactCompleteCostUsd": null,
    "explanation": "The displayed lower bound is the parent-session Claude API list-price equivalent. Seven separately billed subagents reported tokens without a costable token mix, so the complete API-equivalent cost is unknown and higher. It is not a Claude Code invoice or subscription charge."
  },
  "visualAcceptance": {
    "passed": false,
    "scoreHistory": [
      12,
      14,
      14
    ],
    "final": 14,
    "maximum": 40,
    "suppressesResult": false
  },
  "publication": {
    "benchmarkVisible": true,
    "blindBattleEligible": true,
    "libraryListed": true
  },
  "caveats": [
    "This is one run; no reliability rate can be inferred.",
    "Wall clock is end-to-end session latency, including tool execution, Unity and Blender work, captures, self-tests and subagent waits.",
    "Output tokens include hidden reasoning, code and tool-call JSON.",
    "The original harness ledger predates the later publication package; current player and source identities come from the separate active release records."
  ],
  "missingEvidence": [
    "Subagent token mix and exact subagent cost",
    "Exact complete API-equivalent run cost",
    "A multi-run reliability estimate",
    "Automated cross-platform gameplay evidence for WebGL and Windows"
  ]
}
