Human Frontier
87.7
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
Free/M
Input Price
Free/M
Output Price
Free/M
Speed
—
TTFT
—
Axes
Filters
Providers
vv1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00 · official-869-task · GPT-5.4 · harness Codex CLI · 1 attempt · 2h timeout · exploitgym:v1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00:869-tasks · record exploitgym:v1:parent:634bf714059bc1564c234a3c40c3432c7250c131a1b6866188dddbf2cc93eec5
Preliminary
{
"taskCount": 869,
"harnessTag": "v1.1.1",
"metricName": "agmodbDerivedOnTargetPercent",
"taskListUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/dcdbc88cbef13624328cc6d059a007a5d29b3c00/data/task_ids/v1.txt",
"aggregateOnly": true,
"onTargetCount": 61,
"evaluationDate": "2026-06-16",
"upstreamSource": "ExploitGym Team",
"attemptsPerTask": 1,
"benchmarkCommit": "dcdbc88cbef13624328cc6d059a007a5d29b3c00",
"metricSemantics": "AgMoDB-derived on_target count divided by the fixed v1 task count",
"aggregateVariant": "parent",
"measurementScope": "model-agent-system",
"agmobenchEligible": false,
"attemptsSemantics": "one submitted result trial per benchmark task, not 869 attempts per task",
"upstreamSourceUrl": null,
"preliminaryReasons": [
"missing-exact-agent-revision",
"missing-upstream-run-identity"
],
"evaluationCondition": "2h timeout",
"domainOnTargetCounts": {
"v8": 22,
"kernel": 1,
"userspace": 38
},
"attemptsProvenanceUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/603efb3461d17ed05718ca778d2715db772d9385/docs/submission.md#resultsjson",
"mitigatedOnTargetCounts": {
"v8": 0,
"kernel": 1,
"userspace": 2
},
"submissionContractCommit": "603efb3461d17ed05718ca778d2715db772d9385",
"exactReproductionEligible": false
}Evidence observed 2026-06-16
vcontinuous-leaderboard · tasks-created:2026-02-01:2026-05-15 · gpt-5.4-2026-03-05-medium · harness swe-rebench-react@tools · effort medium · 5 attempts · pass@1 · swe-rebench:standardized:128k-or-model-limit · provider gpt-5.4-2026-03-05-medium__tools · record gpt-5.4-2026-03-05-medium__tools:1769904000000:1778803200000
{
"passAt5": 70.65868263473054,
"developer": "openai",
"instanceType": "model",
"contextPolicy": "128k unless the model supports less",
"attemptsPerTask": 5,
"totalTokenUsage": 814025.7628742515,
"modelReleaseDate": "2026-03-05",
"scoreStandardError": 0.8122550877994328,
"cachedTokenPercentage": 81.4876775099086,
"modelReleaseTimestamp": 1772668800000,
"standardContextTokens": 128000,
"averageInstanceCostUsd": 0.6071112652694611,
"potentialContamination": true,
"aggregateTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1769904000000
},
"publishedTaskRangeTimestamp": {
"to": 1778803200000,
"from": 1769904000000
}
}Evidence observed 2026-05-15