anomalyco--models.dev
175 行
3.8 KiB
TOML
175 行
3.8 KiB
TOML
name = "Claude Opus 4.7"
|
|
description = "Stronger Opus tier for advanced software work and high-stakes reasoning"
|
|
family = "claude-opus"
|
|
release_date = "2026-04-16"
|
|
last_updated = "2026-04-16"
|
|
attachment = true
|
|
reasoning = true
|
|
temperature = false
|
|
tool_call = true
|
|
knowledge = "2026-01-31"
|
|
open_weights = false
|
|
|
|
[limit]
|
|
context = 1_000_000
|
|
output = 128_000
|
|
|
|
[modalities]
|
|
input = ["text", "image", "pdf"]
|
|
output = ["text"]
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Bench Pro"
|
|
score = 64.3
|
|
metric = "resolve rate"
|
|
source = "https://www.anthropic.com/news/claude-opus-4-8"
|
|
date = "2026-05-28"
|
|
|
|
[[benchmarks]]
|
|
name = "Terminal-Bench"
|
|
score = 66.1
|
|
metric = "success rate"
|
|
harness = "Terminus-2"
|
|
version = "2.1"
|
|
source = "https://www.anthropic.com/news/claude-opus-4-8"
|
|
date = "2026-05-28"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Atlas Refactoring"
|
|
score = 48.57
|
|
metric = "score"
|
|
harness = "Claude Code"
|
|
source = "https://labs.scale.com/leaderboard/sweatlas-refactoring"
|
|
|
|
[[benchmarks]]
|
|
name = "Artificial Analysis Coding Agent Index"
|
|
score = 66.6
|
|
metric = "average pass@1"
|
|
harness = "Claude Code"
|
|
variant = "max"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Atlas Codebase QnA"
|
|
score = 81
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "max"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Bench Pro"
|
|
score = 44.9
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "max"
|
|
dataset = "hard-aa"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "Terminal-Bench"
|
|
score = 73.8
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "max"
|
|
version = "2.1"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "Artificial Analysis Coding Agent Index"
|
|
score = 61.2
|
|
metric = "average pass@1"
|
|
harness = "Cursor CLI"
|
|
variant = "medium"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Atlas Codebase QnA"
|
|
score = 78.4
|
|
metric = "pass@1"
|
|
harness = "Cursor CLI"
|
|
variant = "medium"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Bench Pro"
|
|
score = 34.4
|
|
metric = "pass@1"
|
|
harness = "Cursor CLI"
|
|
variant = "medium"
|
|
dataset = "hard-aa"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "Terminal-Bench"
|
|
score = 70.6
|
|
metric = "pass@1"
|
|
harness = "Cursor CLI"
|
|
variant = "medium"
|
|
version = "2.1"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "Artificial Analysis Coding Agent Index"
|
|
score = 59.9
|
|
metric = "average pass@1"
|
|
harness = "Claude Code"
|
|
variant = "medium"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Atlas Codebase QnA"
|
|
score = 71.7
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "medium"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "SWE-Bench Pro"
|
|
score = 36.4
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "medium"
|
|
dataset = "hard-aa"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "Terminal-Bench"
|
|
score = 71.4
|
|
metric = "pass@1"
|
|
harness = "Claude Code"
|
|
variant = "medium"
|
|
version = "2.1"
|
|
source = "https://artificialanalysis.ai/agents/coding-agents"
|
|
|
|
[[benchmarks]]
|
|
name = "GPQA Diamond"
|
|
score = 94.2
|
|
metric = "accuracy"
|
|
source = "https://openai.com/index/introducing-gpt-5-5/"
|
|
date = "2026-04-23"
|
|
|
|
[[benchmarks]]
|
|
name = "Humanity's Last Exam"
|
|
score = 46.9
|
|
metric = "accuracy"
|
|
variant = "no tools"
|
|
source = "https://openai.com/index/introducing-gpt-5-5/"
|
|
date = "2026-04-23"
|
|
|
|
[[benchmarks]]
|
|
name = "Humanity's Last Exam"
|
|
score = 54.7
|
|
metric = "accuracy"
|
|
variant = "with tools"
|
|
source = "https://openai.com/index/introducing-gpt-5-5/"
|
|
date = "2026-04-23"
|
|
|
|
[[benchmarks]]
|
|
name = "OSWorld-Verified"
|
|
score = 78.0
|
|
metric = "success rate"
|
|
source = "https://openai.com/index/introducing-gpt-5-5/"
|
|
date = "2026-04-23"
|