{"lastUpdated":"2026-03-07","evidenceStatus":"maintainer-reported","confidence":"low","routingGuidance":false,"accessGuidance":false,"runs":3,"scope":"Direct model-endpoint benchmark on the included tasks; not a configured-agent evaluation or safety certification.","limitations":["Three runs per model is below the runner recommendation of five or more.","Tool-use cases score structured model output and do not execute real tools or observe side effects.","Raw per-case outputs for this published snapshot are not included in the public dataset.","This March snapshot predates the current scorer hardening and was not regenerated with the current runner."],"models":["gpt-5.4","gpt-5.3-instant","claude-opus-4.6","claude-sonnet-4.6","gemini-3.1-pro"],"modelIdentityStatus":"Display identifiers only; the exact provider model IDs used for this snapshot were not published.","runManifest":{"artifact":"summary-only","rawOutputsPublished":false,"taskCaseIdsPublished":false,"exactModelIdsPublished":false,"judgeModelId":null,"scorerVersion":null,"commitSha":null},"categories":[{"name":"Instruction Following","slug":"instruction-following","scores":{"gpt-5.4":10,"gpt-5.3-instant":9.33,"claude-opus-4.6":7.94,"claude-sonnet-4.6":8.61,"gemini-3.1-pro":9.19}},{"name":"Tool Use","slug":"tool-use","scores":{"gpt-5.4":6.22,"gpt-5.3-instant":6.33,"claude-opus-4.6":5,"claude-sonnet-4.6":4.89,"gemini-3.1-pro":5}},{"name":"Code Generation","slug":"code-generation","scores":{"gpt-5.4":9.13,"gpt-5.3-instant":9.13,"claude-opus-4.6":9.13,"claude-sonnet-4.6":9.13,"gemini-3.1-pro":9.07}},{"name":"Summarization","slug":"summarization","scores":{"gpt-5.4":6.17,"gpt-5.3-instant":6.32,"claude-opus-4.6":5.34,"claude-sonnet-4.6":5.4,"gemini-3.1-pro":5.24}},{"name":"Judgment","slug":"judgment","scores":{"gpt-5.4":6.6,"gpt-5.3-instant":9,"claude-opus-4.6":8.6,"claude-sonnet-4.6":9.13,"gemini-3.1-pro":9}},{"name":"Safety-boundary prompts","slug":"safety-trust","scores":{"gpt-5.4":9.56,"gpt-5.3-instant":9.89,"claude-opus-4.6":10,"claude-sonnet-4.6":10,"gemini-3.1-pro":6.78}}],"methodology":"https://www.clawbotomy.com/about","reproduce":"https://github.com/aa-on-ai/clawbotomy","schemaVersion":"3.0.0","latestRunId":"run-81c20219522a9a3f9288","publishedRuns":[{"runId":"run-30f0c5d67dec23127a9f","bundleDigest":"fcd21ad66ec134b8a34c6c5ce9b243c9ff330f101bc68c76ec7becd3240130f9","sourceBundleDigest":"30f0c5d67dec23127a9ffab448fdb82c8c0e3925e5245f490557a1cf68809b70","completedAt":"2026-07-17T04:31:27.263Z","measurementStatus":"measured","reproducibilityStatus":"complete","reviewStatus":"maintainer-self-reported","authorizationStatus":"non-authorizing","manifest":"/evidence/run-30f0c5d67dec23127a9f/manifest.json","cases":"/evidence/run-30f0c5d67dec23127a9f/cases.jsonl","summary":"/evidence/run-30f0c5d67dec23127a9f/summary.json","integrity":"/evidence/run-30f0c5d67dec23127a9f/integrity.json"},{"runId":"run-81c20219522a9a3f9288","bundleDigest":"e0cda24de9a884c613c8cfc9afd9c6750791de70d2d21d71f2db13a3c30e7bd2","sourceBundleDigest":"81c20219522a9a3f92884679e5e8fb219edea394556ecb286229432d43f6af73","completedAt":"2026-07-17T08:10:58.286Z","measurementStatus":"measured","reproducibilityStatus":"complete","reviewStatus":"maintainer-self-reported","authorizationStatus":"non-authorizing","manifest":"/evidence/run-81c20219522a9a3f9288/manifest.json","cases":"/evidence/run-81c20219522a9a3f9288/cases.jsonl","summary":"/evidence/run-81c20219522a9a3f9288/summary.json","integrity":"/evidence/run-81c20219522a9a3f9288/integrity.json"},{"runId":"run-a035f620a2daab63f2ee","bundleDigest":"445a48cd00672762296c8d5f0c3133b8a20354beb05697366c87d15e41463d47","sourceBundleDigest":"a035f620a2daab63f2ee8b554996603ff546754323ce447c6ec140ac9f9e9fa9","completedAt":"2026-07-17T07:15:09.210Z","measurementStatus":"measured","reproducibilityStatus":"complete","reviewStatus":"maintainer-self-reported","authorizationStatus":"non-authorizing","manifest":"/evidence/run-a035f620a2daab63f2ee/manifest.json","cases":"/evidence/run-a035f620a2daab63f2ee/cases.jsonl","summary":"/evidence/run-a035f620a2daab63f2ee/summary.json","integrity":"/evidence/run-a035f620a2daab63f2ee/integrity.json"}],"evidenceRegistry":{"status":"published-runs-available","warning":"Each run is maintainer-reported and non-authorizing; inspect its manifest and cases.","links":{"index":"/evidence/index.json","schemas":"/evidence/schema/","runApiTemplate":"/api/bench/runs/{runId}","caseApiTemplate":"/api/bench/runs/{runId}/cases/{recordId}"}},"legacySummary":{"lastUpdated":"2026-03-07","evidenceStatus":"maintainer-reported","confidence":"low","routingGuidance":false,"accessGuidance":false,"runs":3,"scope":"Direct model-endpoint benchmark on the included tasks; not a configured-agent evaluation or safety certification.","limitations":["Three runs per model is below the runner recommendation of five or more.","Tool-use cases score structured model output and do not execute real tools or observe side effects.","Raw per-case outputs for this published snapshot are not included in the public dataset.","This March snapshot predates the current scorer hardening and was not regenerated with the current runner."],"models":["gpt-5.4","gpt-5.3-instant","claude-opus-4.6","claude-sonnet-4.6","gemini-3.1-pro"],"modelIdentityStatus":"Display identifiers only; the exact provider model IDs used for this snapshot were not published.","runManifest":{"artifact":"summary-only","rawOutputsPublished":false,"taskCaseIdsPublished":false,"exactModelIdsPublished":false,"judgeModelId":null,"scorerVersion":null,"commitSha":null},"categories":[{"name":"Instruction Following","slug":"instruction-following","scores":{"gpt-5.4":10,"gpt-5.3-instant":9.33,"claude-opus-4.6":7.94,"claude-sonnet-4.6":8.61,"gemini-3.1-pro":9.19}},{"name":"Tool Use","slug":"tool-use","scores":{"gpt-5.4":6.22,"gpt-5.3-instant":6.33,"claude-opus-4.6":5,"claude-sonnet-4.6":4.89,"gemini-3.1-pro":5}},{"name":"Code Generation","slug":"code-generation","scores":{"gpt-5.4":9.13,"gpt-5.3-instant":9.13,"claude-opus-4.6":9.13,"claude-sonnet-4.6":9.13,"gemini-3.1-pro":9.07}},{"name":"Summarization","slug":"summarization","scores":{"gpt-5.4":6.17,"gpt-5.3-instant":6.32,"claude-opus-4.6":5.34,"claude-sonnet-4.6":5.4,"gemini-3.1-pro":5.24}},{"name":"Judgment","slug":"judgment","scores":{"gpt-5.4":6.6,"gpt-5.3-instant":9,"claude-opus-4.6":8.6,"claude-sonnet-4.6":9.13,"gemini-3.1-pro":9}},{"name":"Safety-boundary prompts","slug":"safety-trust","scores":{"gpt-5.4":9.56,"gpt-5.3-instant":9.89,"claude-opus-4.6":10,"claude-sonnet-4.6":10,"gemini-3.1-pro":6.78}}],"methodology":"https://www.clawbotomy.com/about","reproduce":"https://github.com/aa-on-ai/clawbotomy"}}