Human Frontier
92.6
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$2.13/M
Input Price
$1.38/M
Output Price
$4.40/M
Speed
—
TTFT
—
Axes
Filters
Providers
vv1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00 · official-869-task · GLM-5.1 · harness Claude Code · 1 attempt · 2h timeout · exploitgym:v1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00:869-tasks · record exploitgym:v1:parent:c5de11314bf86df220b5e256fb40b8ce0929af1f34faa09878094c96c5d76795
Preliminary
{
"taskCount": 869,
"harnessTag": "v1.1.1",
"metricName": "agmodbDerivedOnTargetPercent",
"taskListUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/dcdbc88cbef13624328cc6d059a007a5d29b3c00/data/task_ids/v1.txt",
"aggregateOnly": true,
"onTargetCount": 4,
"evaluationDate": "2026-06-16",
"upstreamSource": "ExploitGym Team",
"attemptsPerTask": 1,
"benchmarkCommit": "dcdbc88cbef13624328cc6d059a007a5d29b3c00",
"metricSemantics": "AgMoDB-derived on_target count divided by the fixed v1 task count",
"aggregateVariant": "parent",
"measurementScope": "model-agent-system",
"agmobenchEligible": false,
"attemptsSemantics": "one submitted result trial per benchmark task, not 869 attempts per task",
"upstreamSourceUrl": null,
"preliminaryReasons": [
"missing-exact-agent-revision",
"missing-upstream-run-identity"
],
"evaluationCondition": "2h timeout",
"domainOnTargetCounts": {
"v8": 0,
"kernel": 0,
"userspace": 4
},
"attemptsProvenanceUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/603efb3461d17ed05718ca778d2715db772d9385/docs/submission.md#resultsjson",
"mitigatedOnTargetCounts": {
"v8": 0,
"kernel": 0,
"userspace": 0
},
"submissionContractCommit": "603efb3461d17ed05718ca778d2715db772d9385",
"exactReproductionEligible": false
}Evidence observed 2026-06-16
vcontinuous-leaderboard · tasks-created:2026-03-01:2026-05-15 · GLM-5.1 · harness swe-rebench-react@tools · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider GLM-5.1__tools · record GLM-5.1__tools:1772323200000:1778803200000
{
"passAt5": 65.45454545454545,
"developer": "zhipu",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 2664000.992727273,
"modelReleaseDate": "2026-04-07",
"scoreStandardError": 0.9270944570168703,
"cachedTokenPercentage": 91.78128852041365,
"modelReleaseTimestamp": 1775520000000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.9428887529454546,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1772323200000
},
"publishedTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1772323200000
}
}Evidence observed 2026-05-15
vcontinuous-leaderboard · tasks-created:2026-02-01:2026-03-01 · GLM-5.1 · harness swe-rebench-react@tools · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider GLM-5.1__tools · record GLM-5.1__tools:1769904000000:1772323200000
{
"passAt5": 71.9298245614035,
"developer": "zhipu",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 2227217.35093985,
"modelReleaseDate": "2026-04-07",
"scoreStandardError": 0.7393483709273188,
"cachedTokenPercentage": 92.31927667861885,
"modelReleaseTimestamp": 1775520000000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.7587628127669174,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1772323200000,
"from": 1769904000000
},
"publishedTaskRangeTimestamp": {
"to": 1772323200000,
"from": 1769904000000
}
}Evidence observed 2026-03-01
v2.1 · official-leaderboard · GLM-5.1 (max) · harness claude-code@2.1.123 · effort max · 5 attempts · harbor:terminal-bench/terminal-bench-2-1 · provider glm-5.1 · record leaderboard/submissions/2026-05-01-glm-5-1-max-claude-code.json · run fd8707bb-51e8-56fa-8e46-769a82a531ae
{
"passAt2": 0.6551,
"passAt3": 0.6899,
"passAt4": 0.7146,
"passAt5": 0.7303,
"taskCount": 89,
"actualTokens": {
"output": 10635681,
"cachedInput": 374577728,
"uncachedInput": 13570352
},
"datasetPackage": "terminal-bench/terminal-bench-2-1",
"leaderboardUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
"pullRequestUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/67",
"agentDisplayUrl": "https://claude.com/claude-code",
"modelDisplayUrl": "https://z.ai/blog/glm-5.1",
"actualTrialCount": 445,
"agentOrganization": "Anthropic",
"modelOrganization": "Z.ai",
"actualTotalCostUsd": 277.14,
"rewardHacksPercent": 0,
"accuracyStandardError": 1.24,
"submissionDownloadUrl": "https://raw.githubusercontent.com/harbor-framework/terminal-bench-2-1/main/leaderboard/submissions/2026-05-01-glm-5-1-max-claude-code.json",
"disqualifiedTrialCount": 0,
"expectedAttemptsPerTask": 5,
"sourceFilterReasoningEffort": "max",
"actualAverageTrialDurationSeconds": 944.6
}Evidence observed 2026-05-01