Human Frontier
78.4
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$2.63/M
Input Price
$0.70/M
Output Price
$8.40/M
Speed
—
TTFT
—
Axes
Filters
Providers
vcontinuous-leaderboard · tasks-created:2025-08-01:2025-09-01 · Qwen3-32B · harness swe-rebench-react@tools · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Qwen3-32B__tools · record Qwen3-32B__tools:1754006400000:1756684800000
{
"passAt5": 19.230769230769234,
"developer": "alibaba",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 436355.0153846154,
"modelReleaseDate": "2025-04-29",
"scoreStandardError": 2.0532842792367907,
"cachedTokenPercentage": null,
"modelReleaseTimestamp": 1745884800000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.04574375538461538,
"potentialContamination": false,
"aggregateTaskRangeTimestamp": {
"to": 1756684800000,
"from": 1754006400000
},
"publishedTaskRangeTimestamp": {
"to": 1756684800000,
"from": 1754006400000
}
}Evidence observed 2025-09-01
vcontinuous-leaderboard · tasks-created:2025-01-01:2025-08-01 · Qwen3-32B no-thinking · harness swe-rebench-react@text · effort no-thinking · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Qwen3-32B no-thinking__text · record Qwen3-32B no-thinking__text:1735689600000:1754006400000
{
"passAt5": 26.19647355163728,
"developer": "alibaba",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 0,
"modelReleaseDate": "2025-04-28",
"scoreStandardError": 0.518671543626549,
"cachedTokenPercentage": null,
"modelReleaseTimestamp": 1745798400000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
},
"publishedTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
}
}Evidence observed 2025-08-01
vcontinuous-leaderboard · tasks-created:2025-01-01:2025-08-01 · Qwen3-32B thinking · harness swe-rebench-react@text · effort thinking · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Qwen3-32B thinking__text · record Qwen3-32B thinking__text:1735689600000:1754006400000
{
"passAt5": 25.692695214105793,
"developer": "alibaba",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 0,
"modelReleaseDate": "2025-04-28",
"scoreStandardError": 0.17083954617444017,
"cachedTokenPercentage": null,
"modelReleaseTimestamp": 1745798400000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
},
"publishedTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
}
}Evidence observed 2025-08-01
vcontinuous-leaderboard · tasks-created:2025-01-01:2025-08-01 · Qwen3-32B thinking · harness swe-rebench-react@text · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider Qwen3-32B thinking__text · record Qwen3-32B thinking__text:1735689600000:1754006400000
{
"passAt5": 25.692695214105793,
"developer": "alibaba",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 0,
"modelReleaseDate": "2025-04-28",
"scoreStandardError": 0.17083954617444017,
"cachedTokenPercentage": null,
"modelReleaseTimestamp": 1745798400000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
},
"publishedTaskRangeTimestamp": {
"to": 1754006400000,
"from": 1735689600000
}
}Evidence observed 2025-08-01