Human Frontier
—
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$0.45/M
Input Price
$0.20/M
Output Price
$1.20/M
Speed
161 tok/s
TTFT
1.50s
Axes
Filters
Providers
v3 · public-demo · GPT-5.6 Luna (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-luna-medium
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-luna",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "openai-gpt-5-6-luna-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 2379.13,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · GPT-5.6 Luna (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-luna-medium
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-luna",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "openai-gpt-5-6-luna-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 2379.13,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · GPT-5.6 Luna (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-luna-medium
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-luna",
"modelNotes": null,
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "openai-gpt-5-6-luna-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 2379.13,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · GPT-5.6 Luna (Medium) · effort medium · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-luna-medium
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-luna",
"modelNotes": null,
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "openai-gpt-5-6-luna-medium",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 2379.13,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
vcontinuous-leaderboard · tasks-created:2026-05-15:2026-07-01 · GPT-5.6 Luna [medium] · harness swe-rebench-react@tools · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider GPT-5.6 Luna [medium]__tools · record GPT-5.6 Luna [medium]__tools:1778803200000:1782864000000
{
"passAt5": 59.45945945945946,
"developer": "openai",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 395521.6576576577,
"modelReleaseDate": "2026-06-26",
"scoreStandardError": 1.4693249036306386,
"cachedTokenPercentage": 85.15466063171635,
"modelReleaseTimestamp": 1782432000000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.10816659891891893,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
},
"publishedTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
}
}Evidence observed 2026-07-01
vcontinuous-leaderboard · tasks-created:2026-05-15:2026-07-01 · GPT-5.6 Luna [medium] · harness swe-rebench-react@tools · effort medium · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider GPT-5.6 Luna [medium]__tools · record GPT-5.6 Luna [medium]__tools:1778803200000:1782864000000
{
"passAt5": 59.45945945945946,
"developer": "openai",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 395521.6576576577,
"modelReleaseDate": "2026-06-26",
"scoreStandardError": 1.4693249036306386,
"cachedTokenPercentage": 85.15466063171635,
"modelReleaseTimestamp": 1782432000000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.10816659891891893,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
},
"publishedTaskRangeTimestamp": {
"to": 1782864000000,
"from": 1778803200000
}
}Evidence observed 2026-07-01