Human Frontier
90.4
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$3.00/M
Input Price
$2.00/M
Output Price
$6.00/M
Speed
53 tok/s
TTFT
10.15s
Axes
Filters
Providers
v3 · public-demo · Grok 4.5 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-high
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-high",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 6892.87,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-medium
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8458.3,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (Low) · effort low · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-low
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-low",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8718.48,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-high
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-high",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 6892.87,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-medium
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8458.3,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (Low) · effort low · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-low
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "xai-grok-4-5-low",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8718.48,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · Grok 4.5 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-high
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "xai-grok-4-5-high",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 6892.87,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · Grok 4.5 (Low) · effort low · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-low
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "xai-grok-4-5-low",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8718.48,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · Grok 4.5 (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-medium
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "xai-grok-4-5-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8458.3,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · Grok 4.5 (Low) · effort low · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-low
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "xai-grok-4-5-low",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8718.48,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
v3 · public-demo · Grok 4.5 (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-medium
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "xai-grok-4-5-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 8458.3,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
v3 · public-demo · Grok 4.5 (High) · effort high · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:xai-grok-4-5-high
{
"provider": "xAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "xai-grok-4-5",
"modelNotes": null,
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "xai-grok-4-5-high",
"environmentCount": 25,
"modelReleaseDate": "2026-07-16T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 6892.87,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
v2026.06 · official-100-task · Grok 4.5 (high) · harness mini-swe-agent · effort high · 1 attempt · pass@1 · harbor:snorkel-ai/senior-swe-bench-v2026.06 · provider openai/grok-4.5 · record grok-4-5-mini-swe · run 1b426ea2b95513900b31967c6460a77b395f178aced1c84f631044114146fd16
{
"taskCount": 100,
"agentGroup": "mini-swe-agent:grok-4.5",
"basicPassAt1": 49.42528735632184,
"datasetPackage": "snorkel-ai/senior-swe-bench-v2026.06",
"publicTaskCount": 50,
"tastefulPassAt1": 17.24137931034483,
"agentDescription": "xAI Grok 4.5 via mini-swe-agent harness (Portkey→xAI)",
"privateTaskCount": 50,
"upstreamProvider": "xai",
"upstreamRunCount": 50,
"averageAgentSteps": 79.86363636363636,
"averageOutputTokens": 22893.670454545456,
"averageTotalCostUsd": 1.2253361936363636,
"averageOutputCostUsd": 0.13736202272727271
}Evidence observed 2026-07-09
vcontinuous-leaderboard · tasks-created:2026-05-15:2026-07-01 · Grok 4.5 [high] · harness swe-rebench-react@tools · effort high · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Grok 4.5 [high]__tools · record Grok 4.5 [high]__tools:1778803200000:1782864000000
{
"passAt5": 77.47747747747748,
"developer": "xai",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 2429423.823423424,
"modelReleaseDate": "2026-07-08",
"scoreStandardError": 0.5975900523162895,
"cachedTokenPercentage": 92.59017890332878,
"modelReleaseTimestamp": 1783468800000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 1.4705187747747748,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
},
"publishedTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
}
}Evidence observed 2026-07-01
vcontinuous-leaderboard · tasks-created:2026-05-15:2026-07-01 · Grok 4.5 [high] · harness swe-rebench-react@tools · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Grok 4.5 [high]__tools · record Grok 4.5 [high]__tools:1778803200000:1782864000000
{
"passAt5": 77.47747747747748,
"developer": "xai",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 2429423.823423424,
"modelReleaseDate": "2026-07-08",
"scoreStandardError": 0.5975900523162895,
"cachedTokenPercentage": 92.59017890332878,
"modelReleaseTimestamp": 1783468800000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 1.4705187747747748,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
},
"publishedTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
}
}Evidence observed 2026-07-01
v2.1 · official-leaderboard · Grok 4.5 (high) · harness cursor-cli@2026.07.08-0c04a8a · effort high · 5 attempts · harbor:terminal-bench/terminal-bench-2-1 · provider cursor/grok-4.5 · record leaderboard/submissions/2026-07-09-cursor-grok-4-5-none-cursor-cli.json · run d478d2af-5348-575c-b20a-e5a2434dbff7
{
"passAt2": 0.8888,
"passAt3": 0.9236,
"passAt4": 0.9416,
"passAt5": 0.9551,
"taskCount": 89,
"actualTokens": {
"output": 4734062,
"cachedInput": 161079936,
"uncachedInput": 12570445
},
"datasetPackage": "terminal-bench/terminal-bench-2-1",
"leaderboardUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
"pullRequestUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/86",
"agentDisplayUrl": "https://cursor.com/docs/cli/overview",
"modelDisplayUrl": "https://docs.x.ai/developers/models/grok-4.5",
"actualTrialCount": 445,
"agentOrganization": "Cursor",
"modelOrganization": "xAI",
"actualTotalCostUsd": 134.09,
"rewardHacksPercent": 8.99,
"accuracyStandardError": 1.46,
"submissionDownloadUrl": "https://raw.githubusercontent.com/harbor-framework/terminal-bench-2-1/main/leaderboard/submissions/2026-07-09-cursor-grok-4-5-none-cursor-cli.json",
"disqualifiedTrialCount": 40,
"expectedAttemptsPerTask": 5,
"sourceFilterReasoningEffort": null,
"actualAverageTrialDurationSeconds": 443
}Evidence observed 2026-07-09