Human Frontier
83.9
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$10.00/M
Input Price
$5.00/M
Output Price
$25.00/M
Speed
—
TTFT
—
Axes
Filters
Providers
v3 · public-demo · Claude Opus 4.8 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:anthropic-opus-4-8-high
{
"provider": "Anthropic",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "anthropic-opus-4-8",
"modelNotes": "Evaluation conducted with 'high' output effort.",
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "anthropic-opus-4-8-high",
"environmentCount": 25,
"modelReleaseDate": "2026-06-01T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 10000,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Claude Opus 4.8 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:anthropic-opus-4-8-high
{
"provider": "Anthropic",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "anthropic-opus-4-8",
"modelNotes": "Evaluation conducted with 'high' output effort.",
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "anthropic-opus-4-8-high",
"environmentCount": 25,
"modelReleaseDate": "2026-06-01T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 10000,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Claude Opus 4.8 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:anthropic-opus-4-8-high
{
"provider": "Anthropic",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "anthropic-opus-4-8",
"modelNotes": "Evaluation conducted with 'high' output effort.",
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "anthropic-opus-4-8-high",
"environmentCount": 25,
"modelReleaseDate": "2026-06-01T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 10000,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · Claude Opus 4.8 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:anthropic-opus-4-8-high
{
"provider": "Anthropic",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "anthropic-opus-4-8",
"modelNotes": "Evaluation conducted with 'high' output effort.",
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "anthropic-opus-4-8-high",
"environmentCount": 25,
"modelReleaseDate": "2026-06-01T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 10000,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
vlive-leaderboard · enigma_eval · claude-opus-4-8-xhigh · effort xhigh · 1 attempt · pass@1 · scale-seal · provider claude-opus-4-8-xhigh · record enigma_eval:claude-opus-4-8-xhigh
{
"rank": 4,
"company": "anthropic",
"benchmarkId": "enigma_eval",
"identityNote": "Scale SEAL published configuration label preserved",
"benchmarkName": "EnigmaEval"
}Evidence observed 2026-08-01
vlive-leaderboard · enigma_eval · claude-opus-4-8-xhigh · effort xhigh · 1 attempt · pass@1 · scale-seal · provider claude-opus-4-8-xhigh · record enigma_eval:claude-opus-4-8-xhigh
{
"rank": 4,
"company": "anthropic",
"benchmarkId": "enigma_eval",
"identityNote": "Scale SEAL published configuration label preserved",
"benchmarkName": "EnigmaEval"
}Evidence observed 2026-07-31
vlive-leaderboard · enigma_eval · claude-opus-4-8-xhigh · effort xhigh · 1 attempt · pass@1 · scale-seal · provider claude-opus-4-8-xhigh · record enigma_eval:claude-opus-4-8-xhigh
{
"rank": 4,
"company": "anthropic",
"benchmarkId": "enigma_eval",
"identityNote": "Scale SEAL published configuration label preserved",
"benchmarkName": "EnigmaEval"
}Evidence observed 2026-07-30
vlive-leaderboard · enigma_eval · claude-opus-4-8-xhigh · effort xhigh · 1 attempt · pass@1 · scale-seal · provider claude-opus-4-8-xhigh · record enigma_eval:claude-opus-4-8-xhigh
{
"rank": 4,
"company": "anthropic",
"benchmarkId": "enigma_eval",
"identityNote": "Scale SEAL published configuration label preserved",
"benchmarkName": "EnigmaEval"
}Evidence observed 2026-07-29
v2026.06 · official-100-task · Claude Opus 4.8 (max) · harness mini-swe-agent · effort max · 1 attempt · pass@1 · harbor:snorkel-ai/senior-swe-bench-v2026.06 · provider anthropic/claude-opus-4-8 · record opus-48-mini-swe · run 4ee23dca8f25cd97660a6a5e6781ba4c8a820780bc8018dd43fcf22a96c06b11
{
"taskCount": 100,
"agentGroup": "mini-swe-agent:claude-opus-4-8",
"basicPassAt1": 39.77272727272727,
"datasetPackage": "snorkel-ai/senior-swe-bench-v2026.06",
"publicTaskCount": 50,
"tastefulPassAt1": 25,
"agentDescription": "Claude Opus 4.8 via mini-swe-agent harness (Portkey→Bedrock)",
"privateTaskCount": 50,
"upstreamProvider": "anthropic",
"upstreamRunCount": 80,
"averageAgentSteps": 132.36363636363637,
"averageOutputTokens": 115169.26136363637,
"averageTotalCostUsd": 8.03432186590909,
"averageOutputCostUsd": 2.879231534090909
}Evidence observed 2026-06-30
vcontinuous-leaderboard · tasks-created:2026-03-01:2026-05-15 · Claude Opus 4.8-xhigh · harness swe-rebench-react@tools · effort xhigh · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Claude Opus 4.8-xhigh__tools · record Claude Opus 4.8-xhigh__tools:1772323200000:1778803200000
{
"passAt5": 67.27272727272727,
"developer": "anthropic",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 2479387.470909091,
"modelReleaseDate": "2026-05-28",
"scoreStandardError": 1.199173268933903,
"cachedTokenPercentage": 95.27838279131717,
"modelReleaseTimestamp": 1779926400000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 2.0202793086363635,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1772323200000
},
"publishedTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1772323200000
}
}Evidence observed 2026-05-15
v2.1 · official-leaderboard · Opus 4.8 (high) · harness claude-code@2.1.205 · effort high · 5 attempts · harbor:terminal-bench/terminal-bench-2-1 · provider anthropic/claude-opus-4-8 · record leaderboard/submissions/2026-07-09-anthropic-claude-opus-4-8-high-claude-code.json · run a3019ec2-bc78-5ff6-9cae-d22d62470515
{
"passAt2": 0.8652,
"passAt3": 0.9011,
"passAt4": 0.9258,
"passAt5": 0.9438,
"taskCount": 89,
"actualTokens": {
"output": 8089069,
"cachedInput": 161931526,
"uncachedInput": 12873472
},
"datasetPackage": "terminal-bench/terminal-bench-2-1",
"leaderboardUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
"pullRequestUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/92",
"agentDisplayUrl": "https://claude.com/claude-code",
"modelDisplayUrl": "https://www.anthropic.com/news/claude-opus-4-8",
"actualTrialCount": 445,
"agentOrganization": "Anthropic",
"modelOrganization": "Anthropic",
"actualTotalCostUsd": 286.94,
"rewardHacksPercent": 0,
"accuracyStandardError": 1.31,
"submissionDownloadUrl": "https://raw.githubusercontent.com/harbor-framework/terminal-bench-2-1/main/leaderboard/submissions/2026-07-09-anthropic-claude-opus-4-8-high-claude-code.json",
"disqualifiedTrialCount": 0,
"expectedAttemptsPerTask": 5,
"sourceFilterReasoningEffort": "high",
"actualAverageTrialDurationSeconds": 551.4
}Evidence observed 2026-07-09