Human Frontier
—
Human-calibrated frontier signal, backed by Arena-style preference evidence and separate from raw AgMoBench benchmark composite scores.
Blended Price
$11.25/M
Input Price
$5.00/M
Output Price
$30.00/M
Speed
64 tok/s
TTFT
111.31s
Axes
Filters
Providers
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"1c13f7ca05a41dddb3e8dbf221993d7e\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"a339ccee4fdf8453a492fc58c884889b\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-08-07
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"6bd4d2a79abbba43993aa72466c90f28\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"909105c66efc7bcdbd1ddf40343ec110\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-08-06
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"fe316c967e356fc8f0ae7405b6a374ea\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"1102d97f1e4c2cf4cd67b86832b6396b\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-08-05
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"2fc19b6caba8b52765ca17eab04fc437\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"9485c7ef206d9b12cdfdec6a631df56b\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-07-31
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"cfc9b87c645d3ef2103e359fd0f9f521\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"0d65f6e0efe0d5cae15e8e2fe5a0994b\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-07-30
v3 · public-demo · GPT-5.6 Sol (Max) · effort max · 1 attempt · 5x-human-baseline-median-per-level · arc-agi-3:public-demo · record v3_Semi_Private:openai-gpt-5-6-sol-max
{
"provider": "OpenAI",
"modelType": "CoT",
"modelsUrl": "https://arcprize.org/media/data/models.json",
"modelGroup": "openai-gpt-5-6-sol",
"modelNotes": null,
"modelsEtag": "W/\"bbdf5725ef7a703a7f0970e913262b6f\"",
"scoreMetric": "relative-human-action-efficiency",
"evaluationsUrl": "https://arcprize.org/media/data/evaluations.json",
"evaluationsEtag": "W/\"6529094085f9ca7ce5f3c2e1957d3b9c\"",
"upstreamModelId": "openai-gpt-5-6-sol-max",
"environmentCount": 25,
"modelReleaseDate": "2026-07-09T00:00:00.000Z",
"upstreamDatasetId": "v3_Semi_Private",
"actionBudgetPolicy": "5x-human-baseline-median-per-level",
"technicalReportUrl": "https://arcprize.org/media/ARC_AGI_3_Technical_Report.pdf",
"catalogMatchAliases": [],
"agentHarnessPublished": false,
"verificationAuthority": "ARC Prize Foundation",
"upstreamRunIdPublished": false,
"actualEvaluationCostUsd": 25064.11,
"providerModelIdPublished": false
}Evidence observed 2026-07-24
vv1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00 · official-869-task · GPT-5.6 Sol (reasoning max) · harness Codex CLI · 1 attempt · 6h timeout · exploitgym:v1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00:869-tasks · record exploitgym:v1:parent:1fede7d2e97c3a3b102b8c2bf56d36c24d5adb3489ef5bb221778591bdb4fd8a
Preliminary
{
"taskCount": 869,
"harnessTag": "v1.1.1",
"metricName": "agmodbDerivedOnTargetPercent",
"taskListUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/dcdbc88cbef13624328cc6d059a007a5d29b3c00/data/task_ids/v1.txt",
"aggregateOnly": true,
"onTargetCount": 293,
"evaluationDate": "2026-07-13",
"upstreamSource": "OpenAI & ExploitGym Team",
"attemptsPerTask": 1,
"benchmarkCommit": "dcdbc88cbef13624328cc6d059a007a5d29b3c00",
"metricSemantics": "AgMoDB-derived on_target count divided by the fixed v1 task count",
"aggregateVariant": "parent",
"measurementScope": "model-agent-system",
"agmobenchEligible": false,
"attemptsSemantics": "one submitted result trial per benchmark task, not 869 attempts per task",
"upstreamSourceUrl": "https://deploymentsafety.openai.com/gpt-5-6",
"preliminaryReasons": [
"missing-exact-agent-revision",
"missing-upstream-run-identity"
],
"evaluationCondition": "6h timeout",
"domainOnTargetCounts": {
"v8": 71,
"kernel": 73,
"userspace": 149
},
"attemptsProvenanceUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/603efb3461d17ed05718ca778d2715db772d9385/docs/submission.md#resultsjson",
"mitigatedOnTargetCounts": {
"v8": 32,
"kernel": 44,
"userspace": 60
},
"submissionContractCommit": "603efb3461d17ed05718ca778d2715db772d9385",
"exactReproductionEligible": false
}Evidence observed 2026-07-13
vv1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00 · official-869-task · GPT-5.6 Sol (reasoning max) · harness Codex CLI · 1 attempt · 2h timeout cutoff · exploitgym:v1@harness-v1.1.1@dcdbc88cbef13624328cc6d059a007a5d29b3c00:869-tasks · record exploitgym:v1:subrow:4ad4674171c9b3e3524af3895b717091751582003735f9d58562453011b6acb4
Preliminary
{
"taskCount": 869,
"harnessTag": "v1.1.1",
"metricName": "agmodbDerivedOnTargetPercent",
"taskListUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/dcdbc88cbef13624328cc6d059a007a5d29b3c00/data/task_ids/v1.txt",
"aggregateOnly": true,
"onTargetCount": 216,
"evaluationDate": "2026-07-13",
"upstreamSource": "OpenAI & ExploitGym Team",
"attemptsPerTask": 1,
"benchmarkCommit": "dcdbc88cbef13624328cc6d059a007a5d29b3c00",
"metricSemantics": "AgMoDB-derived on_target count divided by the fixed v1 task count",
"aggregateVariant": "subrow",
"measurementScope": "model-agent-system",
"agmobenchEligible": false,
"attemptsSemantics": "one submitted result trial per benchmark task, not 869 attempts per task",
"upstreamSourceUrl": "https://deploymentsafety.openai.com/gpt-5-6",
"preliminaryReasons": [
"missing-exact-agent-revision",
"missing-upstream-run-identity"
],
"evaluationCondition": "2h timeout cutoff",
"domainOnTargetCounts": {
"v8": 59,
"kernel": 38,
"userspace": 119
},
"attemptsProvenanceUrl": "https://github.com/sunblaze-ucb/exploitgym/blob/603efb3461d17ed05718ca778d2715db772d9385/docs/submission.md#resultsjson",
"mitigatedOnTargetCounts": {
"v8": 19,
"kernel": 24,
"userspace": 41
},
"submissionContractCommit": "603efb3461d17ed05718ca778d2715db772d9385",
"exactReproductionEligible": false
}Evidence observed 2026-07-13
vv0.1.0@eb4af26c249b5770e788aa9e3acd2f6ef21f50c2 · official-leaderboard · GPT-5.6 Sol (max) · harness Codex · effort max · 1 attempt · harbor:frontier-bench/frontier-bench@2f325140-b410-452e-87b3-978ffcb4e1ea · record 4ce39786-946e-447b-bf36-9650ff7749ac
{
"taskCount": 74,
"metricName": "accuracy",
"releaseTag": "v0.1.0",
"agentDisplay": {
"url": "https://openai.com/codex/",
"label": "Codex"
},
"modelDisplay": {
"url": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
"label": "GPT-5.6 Sol"
},
"rowCreatedAt": "2026-07-23T00:48:29.293634+00:00",
"rowUpdatedAt": "2026-08-03T16:30:20.547554+00:00",
"leaderboardId": "076032b9-6e20-458a-b56d-5aea2488b3f9",
"releaseCommit": "eb4af26c249b5770e788aa9e3acd2f6ef21f50c2",
"datasetPackage": "frontier-bench/frontier-bench",
"evaluationDate": "2026-07-13",
"attemptsPerTask": 1,
"leaderboardRank": 2,
"metricSemantics": "native-harbor-accuracy-not-pass-at-k",
"datasetVersionId": "2f325140-b410-452e-87b3-978ffcb4e1ea",
"measurementScope": "model-agent-system",
"modelReleaseDate": "2026-07-09",
"actualTotalTokens": 5779656478,
"agentOrganization": {
"url": "https://openai.com",
"label": "OpenAI"
},
"agmobenchEligible": false,
"leaderboardStatus": "display",
"modelOrganization": {
"url": "https://openai.com",
"label": "OpenAI"
},
"actualTotalCostUsd": 3963.86,
"attemptsProvenance": "frontier-bench-v0.1-official-command-uses-harbor-default",
"upstreamTrialCount": 0,
"accuracyStandardError": 1.58,
"attemptsProvenanceUrls": [
"https://github.com/harbor-framework/frontier-bench/blob/eb4af26c249b5770e788aa9e3acd2f6ef21f50c2/README.md#running-the-benchmark",
"https://github.com/harbor-framework/harbor/blob/v0.14.0/src/harbor/models/job/config.py#L318-L323"
]
}Evidence observed 2026-07-23
v2.1 · official-leaderboard · GPT-5.6 Sol (max) · harness codex@0.144.0 · effort max · 5 attempts · harbor:terminal-bench/terminal-bench-2-1 · provider gpt-5.6-sol · record leaderboard/submissions/2026-07-10-gpt-5-6-sol-max-codex.json · run fa34c325-6014-59c4-b45b-4ed08c8719ae
{
"passAt2": 0.8348,
"passAt3": 0.8663,
"passAt4": 0.8809,
"passAt5": 0.8876,
"taskCount": 89,
"actualTokens": {
"output": 5935747,
"cachedInput": 540552541,
"uncachedInput": 25266704
},
"datasetPackage": "terminal-bench/terminal-bench-2-1",
"leaderboardUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
"pullRequestUrl": "https://github.com/harbor-framework/terminal-bench-2-1/pull/102",
"agentDisplayUrl": "https://openai.com/codex/",
"modelDisplayUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
"actualTrialCount": 445,
"agentOrganization": "OpenAI",
"modelOrganization": "OpenAI",
"actualTotalCostUsd": 574.68,
"rewardHacksPercent": 7.19,
"accuracyStandardError": 1.28,
"submissionDownloadUrl": "https://raw.githubusercontent.com/harbor-framework/terminal-bench-2-1/main/leaderboard/submissions/2026-07-10-gpt-5-6-sol-max-codex.json",
"disqualifiedTrialCount": 32,
"expectedAttemptsPerTask": 5,
"sourceFilterReasoningEffort": "max",
"actualAverageTrialDurationSeconds": 431
}Evidence observed 2026-07-10