chore: import upstream snapshot with attribution

This commit is contained in:
wehub-resource-sync
2026-07-13 12:28:55 +08:00
commit db42b91b75
6397 changed files with 146012 additions and 0 deletions
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-3.5-turbo"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "gpt"
release_date = "2023-03-01"
last_updated = "2023-11-06"
attachment = false
reasoning = false
temperature = true
tool_call = false
structured_output = false
knowledge = "2021-09-01"
open_weights = false
[limit]
context = 16_385
output = 4_096
[modalities]
input = ["text"]
output = ["text"]
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 10.7
metric = "index"
source = "https://openrouter.ai/openai/gpt-3.5-turbo/benchmarks"
date = "2026-03-11"
+34
View File
@@ -0,0 +1,34 @@
name = "GPT-4 Turbo"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "gpt"
release_date = "2023-11-06"
last_updated = "2024-04-09"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = false
knowledge = "2023-12"
open_weights = false
[limit]
context = 128_000
output = 4_096
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 21.5
metric = "index"
source = "https://openrouter.ai/openai/gpt-4-turbo/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "SciCode"
score = 31.9
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-4-turbo/benchmarks"
date = "2026-03-11"
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-4.1 mini"
description = "Affordable GPT-4.1 lane for fast coding help and structured extraction"
family = "gpt-mini"
release_date = "2025-04-14"
last_updated = "2025-04-14"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2024-04"
open_weights = false
[limit]
context = 1_047_576
output = 32_768
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 32.4
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-04-14"
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-4.1 nano"
description = "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks"
family = "gpt-nano"
release_date = "2025-04-14"
last_updated = "2025-04-14"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2024-04"
open_weights = false
[limit]
context = 1_047_576
output = 32_768
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 8.9
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-04-14"
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-4.1"
description = "Long-lived GPT workhorse for coding, instruction following, and production apps"
family = "gpt"
release_date = "2025-04-14"
last_updated = "2025-04-14"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2024-04"
open_weights = false
[limit]
context = 1_047_576
output = 32_768
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 52.4
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-04-14"
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-4"
description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks"
family = "gpt"
release_date = "2023-11-06"
last_updated = "2024-04-09"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = false
knowledge = "2023-11"
open_weights = false
[limit]
context = 8_192
output = 8_192
[modalities]
input = ["text"]
output = ["text"]
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 13.1
metric = "index"
source = "https://openrouter.ai/openai/gpt-4/benchmarks"
date = "2026-03-11"
+34
View File
@@ -0,0 +1,34 @@
name = "GPT-4o (2024-05-13)"
description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks"
family = "gpt"
release_date = "2024-05-13"
last_updated = "2024-05-13"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 128_000
output = 4_096
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 24.2
metric = "index"
source = "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "SciCode"
score = 30.9
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks"
date = "2026-03-11"
+48
View File
@@ -0,0 +1,48 @@
name = "GPT-4o (2024-08-06)"
description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks"
family = "gpt"
release_date = "2024-08-06"
last_updated = "2024-08-06"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 23.1
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2024-12-30"
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 16.6
metric = "index"
source = "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "SciCode"
score = 33.1
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "Terminal-Bench Hard"
score = 8.3
metric = "success rate"
source = "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks"
date = "2026-03-11"
+48
View File
@@ -0,0 +1,48 @@
name = "GPT-4o (2024-11-20)"
description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks"
family = "gpt"
release_date = "2024-11-20"
last_updated = "2024-11-20"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 18.2
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2024-12-30"
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 16.7
metric = "index"
source = "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "SciCode"
score = 33.3
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks"
date = "2026-03-11"
[[benchmarks]]
name = "Terminal-Bench Hard"
score = 8.3
metric = "success rate"
source = "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks"
date = "2026-03-11"
+34
View File
@@ -0,0 +1,34 @@
name = "GPT-4o mini"
description = "Small omni GPT for cheap multimodal assistance and production-scale traffic"
family = "gpt-mini"
release_date = "2024-07-18"
last_updated = "2024-07-18"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 3.6
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2024-12-21"
[[benchmarks]]
name = "SciCode"
score = 22.9
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-4o-mini/benchmarks"
date = "2026-03-11"
+27
View File
@@ -0,0 +1,27 @@
name = "GPT-4o"
description = "Omni-era GPT for multimodal chat, practical coding, and general assistants"
family = "gpt"
release_date = "2024-05-13"
last_updated = "2024-08-06"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 23.1
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2024-12-30"
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5 Chat (latest)"
description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows"
family = "gpt-codex"
release_date = "2025-08-07"
last_updated = "2025-08-07"
attachment = true
reasoning = true
temperature = true
tool_call = false
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+42
View File
@@ -0,0 +1,42 @@
name = "GPT-5-Codex"
description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work"
family = "gpt-codex"
release_date = "2025-09-15"
last_updated = "2025-09-15"
attachment = false
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Artificial Analysis Coding Index"
score = 38.9
metric = "index"
source = "https://openrouter.ai/openai/gpt-5-codex/benchmarks"
date = "2026-06-01"
[[benchmarks]]
name = "SciCode"
score = 40.9
metric = "percent correct"
source = "https://openrouter.ai/openai/gpt-5-codex/benchmarks"
date = "2026-06-01"
[[benchmarks]]
name = "Terminal-Bench Hard"
score = 37.9
metric = "success rate"
source = "https://openrouter.ai/openai/gpt-5-codex/benchmarks"
date = "2026-06-01"
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5 Mini"
description = "Small GPT-5 for responsive agents, coding help, and everyday automation"
family = "gpt-mini"
release_date = "2025-08-07"
last_updated = "2025-08-07"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5 Nano"
description = "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs"
family = "gpt-nano"
release_date = "2025-08-07"
last_updated = "2025-08-07"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5 Pro"
description = "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning"
family = "gpt-pro"
release_date = "2025-10-06"
last_updated = "2025-10-06"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 272_000
[modalities]
input = ["text", "image"]
output = ["text"]
+20
View File
@@ -0,0 +1,20 @@
name = "GPT-5.1 Chat"
description = "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations"
family = "gpt-codex"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.1 Codex Max"
description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work"
family = "gpt-codex"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.1 Codex mini"
description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work"
family = "gpt-codex"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.1 Codex"
description = "Codex GPT for repository edits, code review, and practical software agents"
family = "gpt-codex"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.1"
description = "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks"
family = "gpt"
release_date = "2025-11-13"
last_updated = "2025-11-13"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+20
View File
@@ -0,0 +1,20 @@
name = "GPT-5.2 Chat"
description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows"
family = "gpt-codex"
release_date = "2025-12-11"
last_updated = "2025-12-11"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image"]
output = ["text"]
+28
View File
@@ -0,0 +1,28 @@
name = "GPT-5.2 Codex"
description = "Code-specialist GPT for repository edits, reviews, and long-running software agents"
family = "gpt-codex"
release_date = "2025-12-11"
last_updated = "2025-12-11"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 41.04
metric = "resolve rate"
dataset = "public"
source = "https://labs.scale.com/leaderboard/swe_bench_pro_public"
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.2 Pro"
description = "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows"
family = "gpt-pro"
release_date = "2025-12-11"
last_updated = "2025-12-11"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = false
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+28
View File
@@ -0,0 +1,28 @@
name = "GPT-5.2"
description = "Reliable GPT generation for broad coding, writing, and tool-assisted product work"
family = "gpt"
release_date = "2025-12-11"
last_updated = "2025-12-11"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 29.94
metric = "resolve rate"
dataset = "public"
source = "https://labs.scale.com/leaderboard/swe_bench_pro_public"
+20
View File
@@ -0,0 +1,20 @@
name = "GPT-5.3 Chat (latest)"
description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows"
family = "gpt"
release_date = "2026-03-03"
last_updated = "2026-03-03"
attachment = true
reasoning = false
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 128_000
output = 16_384
[modalities]
input = ["text", "image"]
output = ["text"]
+42
View File
@@ -0,0 +1,42 @@
name = "GPT-5.3 Codex"
description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work"
family = "gpt-codex"
release_date = "2026-02-05"
last_updated = "2026-02-05"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 32.6
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-qna"
[[benchmarks]]
name = "SWE-Atlas Refactoring"
score = 42.38
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-refactoring"
[[benchmarks]]
name = "SWE-Atlas Test Writing"
score = 38.98
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-tw"
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.4 mini"
description = "Strong small GPT for coding subagents, quick tool use, and high-volume work"
family = "gpt-mini"
release_date = "2026-03-17"
last_updated = "2026-03-17"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-5.4 nano"
description = "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation"
family = "gpt-nano"
release_date = "2026-03-17"
last_updated = "2026-03-17"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
+105
View File
@@ -0,0 +1,105 @@
name = "GPT-5.4 Pro"
description = "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks"
family = "gpt-pro"
release_date = "2026-03-05"
last_updated = "2026-03-05"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = false
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "GPQA Diamond"
score = 94.4
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 42.7
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 58.7
metric = "accuracy"
variant = "with tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "BrowseComp"
score = 89.3
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GDPval"
score = 82.0
metric = "wins or ties"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 50.0
metric = "accuracy"
dataset = "Tier 1-3"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 38.0
metric = "accuracy"
dataset = "Tier 4"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "ARC-AGI-1"
score = 94.5
metric = "accuracy"
variant = "Verified"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "ARC-AGI-2"
score = 83.3
metric = "accuracy"
variant = "Verified"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FinanceAgent"
score = 61.5
metric = "accuracy"
version = "1.1"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GeneBench"
score = 25.6
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
+215
View File
@@ -0,0 +1,215 @@
name = "GPT-5.4"
description = "Agent-ready GPT for coding and computer-use workflows at a lower cost"
family = "gpt"
release_date = "2026-03-05"
last_updated = "2026-03-05"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-08-31"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 59.1
metric = "resolve rate"
dataset = "public"
source = "https://labs.scale.com/leaderboard/swe_bench_pro_public"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 40.8
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-qna"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 36.3
metric = "score"
harness = "Mini-SWE-Agent"
source = "https://labs.scale.com/leaderboard/sweatlas-qna"
[[benchmarks]]
name = "SWE-Atlas Refactoring"
score = 44.29
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-refactoring"
[[benchmarks]]
name = "SWE-Atlas Test Writing"
score = 44.36
metric = "score"
harness = "Codex CLI"
source = "https://labs.scale.com/leaderboard/sweatlas-tw"
[[benchmarks]]
name = "SWE-Atlas Test Writing"
score = 40
metric = "score"
harness = "Mini-SWE-Agent"
source = "https://labs.scale.com/leaderboard/sweatlas-tw"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 53.6
metric = "average pass@1"
harness = "Codex"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 72.4
metric = "pass@1"
harness = "Codex"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 18.4
metric = "pass@1"
harness = "Codex"
variant = "medium"
dataset = "hard-aa"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 69.8
metric = "pass@1"
harness = "Codex"
variant = "medium"
version = "2.1"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 52.2
metric = "average pass@1"
harness = "Cursor CLI"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 72.9
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 18.9
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
dataset = "hard-aa"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 64.7
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
version = "2.1"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 75.1
metric = "success rate"
version = "2.0"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GPQA Diamond"
score = 92.8
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 39.8
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 52.1
metric = "accuracy"
variant = "with tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "OSWorld-Verified"
score = 75.0
metric = "success rate"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "BrowseComp"
score = 82.7
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GDPval"
score = 83.0
metric = "wins or ties"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "ARC-AGI-2"
score = 73.3
metric = "accuracy"
variant = "Verified"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 47.6
metric = "accuracy"
dataset = "Tier 1-3"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 27.1
metric = "accuracy"
dataset = "Tier 4"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "MMMU Pro"
score = 81.2
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
+20
View File
@@ -0,0 +1,20 @@
name = "GPT-5.5 Instant"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
release_date = "2026-05-05"
last_updated = "2026-05-28"
attachment = true
reasoning = true
temperature = true
tool_call = true
structured_output = true
knowledge = "2025-12-01"
open_weights = false
[limit]
context = 400_000
input = 400_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
+74
View File
@@ -0,0 +1,74 @@
name = "GPT-5.5 Pro"
description = "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding"
family = "gpt-pro"
release_date = "2026-04-23"
last_updated = "2026-04-23"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-12-01"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "BrowseComp"
score = 90.1
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 43.1
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 57.2
metric = "accuracy"
variant = "with tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 52.4
metric = "accuracy"
dataset = "Tier 1-3"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 39.6
metric = "accuracy"
dataset = "Tier 4"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GDPval"
score = 82.3
metric = "wins or ties"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GeneBench"
score = 33.2
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
+266
View File
@@ -0,0 +1,266 @@
name = "GPT-5.5"
description = "Default frontier GPT for coding, computer use, research, and knowledge work"
family = "gpt"
release_date = "2026-04-23"
last_updated = "2026-04-23"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2025-12-01"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 58.6
metric = "resolve rate"
source = "https://www.anthropic.com/news/claude-opus-4-8"
date = "2026-05-28"
[[benchmarks]]
name = "Terminal-Bench"
score = 78.2
metric = "success rate"
harness = "Terminus-2"
version = "2.1"
source = "https://www.anthropic.com/news/claude-opus-4-8"
date = "2026-05-28"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 45.43
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-qna"
[[benchmarks]]
name = "SWE-Atlas Refactoring"
score = 44.79
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-refactoring"
[[benchmarks]]
name = "SWE-Atlas Test Writing"
score = 42.59
metric = "score"
harness = "Codex"
source = "https://labs.scale.com/leaderboard/sweatlas-tw"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 65.3
metric = "average pass@1"
harness = "Codex"
variant = "xhigh"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 80.8
metric = "pass@1"
harness = "Codex"
variant = "xhigh"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 30.9
metric = "pass@1"
harness = "Codex"
variant = "xhigh"
dataset = "hard-aa"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 84.1
metric = "pass@1"
harness = "Codex"
variant = "xhigh"
version = "2.1"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 60.4
metric = "average pass@1"
harness = "Codex"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 79.1
metric = "pass@1"
harness = "Codex"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 26.2
metric = "pass@1"
harness = "Codex"
variant = "medium"
dataset = "hard-aa"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 75.8
metric = "pass@1"
harness = "Codex"
variant = "medium"
version = "2.1"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 57.8
metric = "average pass@1"
harness = "Cursor CLI"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Atlas Codebase QnA"
score = 75
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 24.9
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
dataset = "hard-aa"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 73.4
metric = "pass@1"
harness = "Cursor CLI"
variant = "medium"
version = "2.1"
source = "https://artificialanalysis.ai/agents/coding-agents"
[[benchmarks]]
name = "Terminal-Bench"
score = 82.7
metric = "success rate"
version = "2.0"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GPQA Diamond"
score = 93.6
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 41.4
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Humanity's Last Exam"
score = 52.2
metric = "accuracy"
variant = "with tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "OSWorld-Verified"
score = 78.7
metric = "success rate"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "BrowseComp"
score = 84.4
metric = "accuracy"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "MMMU Pro"
score = 81.2
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "ARC-AGI-2"
score = 85.0
metric = "accuracy"
variant = "Verified"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 51.7
metric = "accuracy"
dataset = "Tier 1-3"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "FrontierMath"
score = 35.4
metric = "accuracy"
dataset = "Tier 4"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "GDPval"
score = 84.9
metric = "wins or ties"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "MCP Atlas"
score = 75.3
metric = "success rate"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "Toolathlon"
score = 55.6
metric = "success rate"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
[[benchmarks]]
name = "τ²-Bench Telecom"
score = 98.0
metric = "success rate"
variant = "original prompts"
source = "https://openai.com/index/introducing-gpt-5-5/"
date = "2026-04-23"
+115
View File
@@ -0,0 +1,115 @@
name = "GPT-5.6 Luna"
description = "Cost-efficient GPT-5.6 model for fast, high-volume workloads"
family = "gpt-luna"
release_date = "2026-07-09"
last_updated = "2026-07-09"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-02-16"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 62.7
metric = "resolve rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Terminal-Bench"
score = 84.7
metric = "success rate"
version = "2.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "DeepSWE"
score = 67.2
metric = "resolve rate"
version = "1.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "GPQA Diamond"
score = 92.3
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "FrontierMath"
score = 78.6
metric = "accuracy"
dataset = "Tier 1-3"
version = "v2"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "BrowseComp"
score = 83.3
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "OSWorld"
score = 45.6
metric = "success rate"
version = "2.0"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "MMMU Pro"
score = 78.4
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Agents' Last Exam"
score = 50.3
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Toolathlon"
score = 53.4
metric = "success rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Intelligence Index"
score = 51.2
metric = "index score"
variant = "max"
version = "4.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 74.6
metric = "index score"
harness = "Codex"
variant = "max"
version = "1.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
+115
View File
@@ -0,0 +1,115 @@
name = "GPT-5.6 Sol"
description = "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows"
family = "gpt-sol"
release_date = "2026-07-09"
last_updated = "2026-07-09"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-02-16"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 64.6
metric = "resolve rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Terminal-Bench"
score = 88.8
metric = "success rate"
version = "2.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "DeepSWE"
score = 72.7
metric = "resolve rate"
version = "1.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "GPQA Diamond"
score = 94.6
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "FrontierMath"
score = 89
metric = "accuracy"
dataset = "Tier 1-3"
version = "v2"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "BrowseComp"
score = 90.4
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "OSWorld"
score = 62.6
metric = "success rate"
version = "2.0"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "MMMU Pro"
score = 83
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Agents' Last Exam"
score = 52.7
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Toolathlon"
score = 58
metric = "success rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Intelligence Index"
score = 58.9
metric = "index score"
variant = "max"
version = "4.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 80
metric = "index score"
harness = "Codex"
variant = "max"
version = "1.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
+115
View File
@@ -0,0 +1,115 @@
name = "GPT-5.6 Terra"
description = "Balanced GPT-5.6 model for capable, cost-efficient everyday work"
family = "gpt-terra"
release_date = "2026-07-09"
last_updated = "2026-07-09"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2026-02-16"
open_weights = false
[limit]
context = 1_050_000
input = 922_000
output = 128_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "SWE-Bench Pro"
score = 63.4
metric = "resolve rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Terminal-Bench"
score = 87.4
metric = "success rate"
version = "2.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "DeepSWE"
score = 69.6
metric = "resolve rate"
version = "1.1"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "GPQA Diamond"
score = 92.9
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "FrontierMath"
score = 84.9
metric = "accuracy"
dataset = "Tier 1-3"
version = "v2"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "BrowseComp"
score = 87.5
metric = "accuracy"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "OSWorld"
score = 50.2
metric = "success rate"
version = "2.0"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "MMMU Pro"
score = 80.7
metric = "accuracy"
variant = "no tools"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Agents' Last Exam"
score = 50.4
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Toolathlon"
score = 53.1
metric = "success rate"
source = "https://openai.com/index/gpt-5-6/"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Intelligence Index"
score = 55
metric = "index score"
variant = "max"
version = "4.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
[[benchmarks]]
name = "Artificial Analysis Coding Agent Index"
score = 77.4
metric = "index score"
harness = "Codex"
variant = "max"
version = "1.1"
source = "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
date = "2026-07-09"
+35
View File
@@ -0,0 +1,35 @@
name = "GPT-5"
description = "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows"
family = "gpt"
release_date = "2025-08-07"
last_updated = "2025-08-07"
attachment = true
reasoning = true
temperature = false
knowledge = "2024-09-30"
tool_call = true
structured_output = true
open_weights = false
[limit]
context = 400_000
input = 272_000
output = 128_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 88.0
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-08-23"
[[benchmarks]]
name = "SWE-Bench Pro"
score = 41.78
metric = "resolve rate"
dataset = "public"
source = "https://labs.scale.com/leaderboard/swe_bench_pro_public"
+18
View File
@@ -0,0 +1,18 @@
name = "GPT-Image-1.5"
description = "Image model for prompt-driven generation, editing, and visual design workflows"
family = "gpt-image"
release_date = "2025-11-25"
last_updated = "2025-11-25"
attachment = true
reasoning = false
temperature = false
tool_call = false
open_weights = false
[limit]
context = 0
output = 0
[modalities]
input = ["text", "image"]
output = ["text", "image"]
+18
View File
@@ -0,0 +1,18 @@
name = "GPT-Image-1"
description = "OpenAI image model for production generation, edits, and brand-safe visual workflows"
family = "gpt-image"
release_date = "2025-04-24"
last_updated = "2025-04-24"
attachment = true
reasoning = false
temperature = false
tool_call = false
open_weights = false
[limit]
context = 0
output = 0
[modalities]
input = ["text", "image"]
output = ["image"]
+18
View File
@@ -0,0 +1,18 @@
name = "GPT-Image-2"
description = "Image model for prompt-driven generation, editing, and visual design workflows"
family = "gpt-image"
release_date = "2026-04-21"
last_updated = "2026-04-21"
attachment = true
reasoning = false
temperature = false
tool_call = false
open_weights = false
[limit]
context = 0
output = 0
[modalities]
input = ["text", "image"]
output = ["image"]
+23
View File
@@ -0,0 +1,23 @@
name = "GPT OSS 120B"
description = "Open GPT reasoning model for self-hosted agents and controllable deployments"
family = "gpt-oss"
release_date = "2025-08-05"
last_updated = "2025-08-05"
attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = true
[limit]
context = 131_072
output = 32_768
[modalities]
input = ["text"]
output = ["text"]
[[weights]]
label = "Hugging Face"
url = "https://huggingface.co/openai/gpt-oss-120b"
+23
View File
@@ -0,0 +1,23 @@
name = "GPT OSS 20B"
description = "Open GPT reasoning model for self-hosted agents and controllable deployments"
family = "gpt-oss"
release_date = "2025-08-05"
last_updated = "2025-08-05"
attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = true
[limit]
context = 131_072
output = 32_768
[modalities]
input = ["text"]
output = ["text"]
[[weights]]
label = "Hugging Face"
url = "https://huggingface.co/openai/gpt-oss-20b"
+23
View File
@@ -0,0 +1,23 @@
name = "GPT OSS Safeguard 120B"
description = "Safety model for policy screening, moderation, and risk-aware routing workflows"
family = "gpt-oss"
release_date = "2025-10-29"
last_updated = "2025-10-29"
attachment = false
reasoning = true
temperature = true
tool_call = true
structured_output = true
open_weights = true
[limit]
context = 131_072
output = 32_768
[modalities]
input = ["text"]
output = ["text"]
[[weights]]
label = "Hugging Face"
url = "https://huggingface.co/openai/gpt-oss-safeguard-120b"
+21
View File
@@ -0,0 +1,21 @@
name = "GPT-Realtime-2.1"
description = "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior"
family = "gpt"
release_date = "2026-07-06"
last_updated = "2026-07-06"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = false
knowledge = "2024-09-30"
open_weights = false
[limit]
context = 128_000
input = 96_000
output = 32_000
[modalities]
input = ["text", "audio", "image"]
output = ["text", "audio"]
+20
View File
@@ -0,0 +1,20 @@
name = "o1-pro"
description = "O-series reasoning model for hard analysis, math, coding, and planning"
family = "o-pro"
release_date = "2025-03-19"
last_updated = "2025-03-19"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image"]
output = ["text"]
+27
View File
@@ -0,0 +1,27 @@
name = "o1"
description = "O-series reasoning model for hard analysis, math, coding, and planning"
family = "o"
release_date = "2024-12-05"
last_updated = "2024-12-05"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2023-09"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 61.7
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2024-12-21"
+19
View File
@@ -0,0 +1,19 @@
name = "o3-deep-research"
description = "Research model for long-horizon investigation, synthesis, and analytical reports"
family = "o"
release_date = "2024-06-26"
last_updated = "2024-06-26"
attachment = true
reasoning = true
temperature = false
tool_call = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image"]
output = ["text"]
+27
View File
@@ -0,0 +1,27 @@
name = "o3-mini"
description = "Smaller o-series reasoner for economical coding, math, and planning tasks"
family = "o-mini"
release_date = "2024-12-20"
last_updated = "2025-01-29"
attachment = false
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 60.4
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-01-31"
+27
View File
@@ -0,0 +1,27 @@
name = "o3-pro"
description = "High-effort o3 tier for difficult technical reasoning and careful answers"
family = "o-pro"
release_date = "2025-06-10"
last_updated = "2025-06-10"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 84.9
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-06-28"
+27
View File
@@ -0,0 +1,27 @@
name = "o3"
description = "Deliberate o-series reasoner for hard math, coding, and multi-step analysis"
family = "o"
release_date = "2025-04-16"
last_updated = "2025-04-16"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image", "pdf"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 81.3
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-06-25"
+19
View File
@@ -0,0 +1,19 @@
name = "o4-mini-deep-research"
description = "Research model for long-horizon investigation, synthesis, and analytical reports"
family = "o-mini"
release_date = "2024-06-26"
last_updated = "2024-06-26"
attachment = true
reasoning = true
temperature = false
tool_call = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image"]
output = ["text"]
+27
View File
@@ -0,0 +1,27 @@
name = "o4-mini"
description = "Fast o-series model for compact reasoning, coding, and tool use"
family = "o-mini"
release_date = "2025-04-16"
last_updated = "2025-04-16"
attachment = true
reasoning = true
temperature = false
tool_call = true
structured_output = true
knowledge = "2024-05"
open_weights = false
[limit]
context = 200_000
output = 100_000
[modalities]
input = ["text", "image"]
output = ["text"]
[[benchmarks]]
name = "Aider Polyglot"
score = 72.0
metric = "percent correct"
source = "https://aider.chat/docs/leaderboards/"
date = "2025-04-16"
+17
View File
@@ -0,0 +1,17 @@
name = "Whisper Large v3 Turbo"
description = "Speech transcription model for accurate audio-to-text and captioning workflows"
family = "whisper"
release_date = "2024-10-01"
last_updated = "2024-10-01"
attachment = false
reasoning = false
tool_call = false
open_weights = true
[limit]
context = 448
output = 448
[modalities]
input = ["audio"]
output = ["text"]
+17
View File
@@ -0,0 +1,17 @@
name = "Whisper 3 Large"
description = "Open Whisper checkpoint for robust multilingual transcription and captioning"
family = "whisper"
release_date = "2024-10-01"
last_updated = "2024-10-01"
attachment = false
reasoning = false
tool_call = false
open_weights = true
[limit]
context = 448
output = 4_096
[modalities]
input = ["audio"]
output = ["text"]