Human Frontier
90.9
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$2.00/M
Input Price
$1.25/M
Output Price
$4.25/M
Speed
—
TTFT
—
Axes
Filters
Providers
vv1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00 · official-869-task · Muse Spark 1.1 (helpful-only versoin) · harness Meta Agent (off-the-shelf agent harness) · 1 attempt · 4h timeout · exploitgym:v1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00:869-tasks · record exploitgym:v1:parent:2398e6822a5e7826d6a449127d07f6213714da14ae118d4f1926d6dff0a1833d
Preliminary
{
"taskCount": 869,
"harnessTag": "v1.1.1",
"metricName": "agmodbDerivedOnTargetPercent",
"taskListUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/dcdbc88cbef13624328cc6d059a007a5d29b3c00/data/task_ids/v1.txt",
"aggregateOnly": true,
"onTargetCount": 7,
"evaluationDate": "2026-07-09",
"upstreamSource": "Meta AI",
"attemptsPerTask": 1,
"benchmarkCommit": "dcdbc88cbef13624328cc6d059a007a5d29b3c00",
"metricSemantics": "AgMoDB-derived on_target count divided by the fixed v1 task count",
"aggregateVariant": "parent",
"measurementScope": "model-agent-system",
"agmobenchEligible": false,
"attemptsSemantics": "one submitted result trial per benchmark task, not 869 attempts per task",
"upstreamSourceUrl": "https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report",
"preliminaryReasons": [
"missing-exact-agent-revision",
"missing-upstream-run-identity"
],
"evaluationCondition": "4h timeout",
"domainOnTargetCounts": {
"v8": 3,
"kernel": 0,
"userspace": 4
},
"attemptsProvenanceUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/603efb3461d17ed05718ca778d2715db772d9385/docs/submission.md#resultsjson",
"mitigatedOnTargetCounts": {
"v8": 0,
"kernel": 0,
"userspace": 0
},
"submissionContractCommit": "603efb3461d17ed05718ca778d2715db772d9385",
"exactReproductionEligible": false
}Evidence observed 2026-07-09
v2.1 · official-leaderboard · Muse Spark 1.1 (xhigh) · harness mini-swe-agent@2.4.5 · effort xhigh · 5 attempts · harbor:terminal-bench/terminal-bench-2-1 · provider openai/muse-spark-1.1 · record leaderboard/submissions/2026-07-09-openai-muse-spark-1-1-xhigh-mini-swe-agent.json · run e15e18db-c8c1-5e9f-9064-1d68975b3c91
{
"passAt2": 0.8292,
"passAt3": 0.8697,
"passAt4": 0.8966,
"passAt5": 0.9101,
"taskCount": 89,
"actualTokens": {
"output": 13679953,
"cachedInput": 932239373,
"uncachedInput": 60996
},
"datasetPackage": "terminal-bench/terminal-bench-2-1",
"leaderboardUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
"pullRequestUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/94",
"agentDisplayUrl": "https://github.com/SWE-agent/mini-swe-agent",
"modelDisplayUrl": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/",
"actualTrialCount": 445,
"agentOrganization": "Princeton",
"modelOrganization": "Meta",
"actualTotalCostUsd": 198.05,
"rewardHacksPercent": 0,
"accuracyStandardError": 1.23,
"submissionDownloadUrl": "https://raw.githubusercontent.com/harbor-framework/terminal-bench-2-1/main/leaderboard/submissions/2026-07-09-openai-muse-spark-1-1-xhigh-mini-swe-agent.json",
"disqualifiedTrialCount": 0,
"expectedAttemptsPerTask": 5,
"sourceFilterReasoningEffort": "xhigh",
"actualAverageTrialDurationSeconds": 687
}Evidence observed 2026-07-09