Skip to content
This repository was archived by the owner on Apr 23, 2026. It is now read-only.

Commit aba476b

Browse files
Update benchmark scores for Claude Opus 4.7 and 4.6 families, and GPT-5.4/Grok-4.20 models (#305)
Verified and updated missing or outdated benchmark values in `data/showdown.json` based on current LiveBench, LMArena, and ArtificialAnalysis data: - `claude-opus-4-7-20260401-thinking-32k`: Added LiveBench (76.91), LiveBench subscores, output_speed_tps (87), latency_ttft_ms (14120), lmarena_en_elo (1504), lmarena_coding_elo (1576), and lmarena_vision_elo (1307). - `claude-opus-4-7-20260401`: Updated lmarena_en_elo (1497), lmarena_coding_elo (1569), and lmarena_vision_elo (1300). - `claude-opus-4-6-20260205-thinking-32k`: Updated lmarena_en_elo (1502), lmarena_coding_elo (1549), and lmarena_vision_elo (1304). - `claude-opus-4-6-20260205`: Updated lmarena_en_elo (1496), lmarena_coding_elo (1544), and lmarena_vision_elo (1293). - `claude-sonnet-4-6-20260217`: Updated lmarena_coding_elo (1525) and lmarena_vision_elo (1269). - `gpt-5.4-high`: Updated lmarena_coding_elo (1457) and lmarena_en_elo (1482). - `grok-4-20-thinking`: Updated lmarena_en_elo (1480). All updates conform to strict verification requirements from two independent sources where applicable, preserving null values for insufficient or ambiguous data. Updated `meta.last_update` timestamp to reflect current dataset modification. Co-authored-by: google-labs-jules[bot] <161369871+google-labs-jules[bot]@users.noreply.github.com>
1 parent caa5f02 commit aba476b

1 file changed

Lines changed: 26 additions & 26 deletions

File tree

data/showdown.json

Lines changed: 26 additions & 26 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
{
22
"meta": {
33
"version": "2026.04.16",
4-
"last_update": "2026-04-21T17:05:56Z",
4+
"last_update": "2026-04-22T16:33:14Z",
55
"schema_version": "1.0"
66
},
77
"models": [
@@ -133,14 +133,14 @@
133133
"average_per_1m": 10.0
134134
},
135135
"performance": {
136-
"output_speed_tps": 50,
137-
"latency_ttft_ms": 0,
136+
"output_speed_tps": 87,
137+
"latency_ttft_ms": 14120,
138138
"source": "https://artificialanalysis.ai/models/claude-opus-4-7"
139139
},
140140
"editor_notes": "Thinking mode for Claude 4.7 Opus. Takes the #1 overall spot on LMArena.",
141141
"benchmark_scores": {
142-
"lmarena_en_elo": 1505,
143-
"lmarena_coding_elo": null,
142+
"lmarena_en_elo": 1504,
143+
"lmarena_coding_elo": 1576,
144144
"lmarena_hard_elo": null,
145145
"lmarena_math_elo": null,
146146
"lmarena_creative_elo": null,
@@ -149,7 +149,7 @@
149149
"swe_bench_pro": null,
150150
"gpqa_diamond": null,
151151
"humanity_last_exam": null,
152-
"livebench": null,
152+
"livebench": 76.91,
153153
"math_500": null,
154154
"aime": null,
155155
"frontiermath": 43.8,
@@ -167,14 +167,14 @@
167167
"mmmu": null,
168168
"mmmu_pro": null,
169169
"lmarena_zh_elo": null,
170-
"lmarena_vision_elo": null,
171-
"livebench_reasoning": null,
172-
"livebench_coding": null,
173-
"livebench_agentic_coding": null,
174-
"livebench_math": null,
175-
"livebench_data_analysis": null,
176-
"livebench_language": null,
177-
"livebench_if": null
170+
"lmarena_vision_elo": 1307,
171+
"livebench_reasoning": 87.69,
172+
"livebench_coding": 82.09,
173+
"livebench_agentic_coding": 60.0,
174+
"livebench_math": 93.1,
175+
"livebench_data_analysis": 78.26,
176+
"livebench_language": 77.91,
177+
"livebench_if": 59.34
178178
}
179179
},
180180
{
@@ -196,8 +196,8 @@
196196
},
197197
"editor_notes": "Anthropic's latest flagship model. Expensive and slower than average but boasts leading intelligence scores with a 1M token context window.",
198198
"benchmark_scores": {
199-
"lmarena_en_elo": 1498,
200-
"lmarena_coding_elo": null,
199+
"lmarena_en_elo": 1497,
200+
"lmarena_coding_elo": 1569,
201201
"lmarena_hard_elo": null,
202202
"lmarena_math_elo": null,
203203
"lmarena_creative_elo": null,
@@ -224,7 +224,7 @@
224224
"mmmu": null,
225225
"mmmu_pro": null,
226226
"lmarena_zh_elo": null,
227-
"lmarena_vision_elo": null,
227+
"lmarena_vision_elo": 1300,
228228
"livebench_reasoning": null,
229229
"livebench_coding": null,
230230
"livebench_agentic_coding": null,
@@ -262,13 +262,13 @@
262262
"humanity_last_exam": 40,
263263
"live_code_bench": null,
264264
"livebench": 61.81,
265-
"lmarena_coding_elo": 1547,
265+
"lmarena_coding_elo": 1544,
266266
"lmarena_creative_elo": 1468,
267-
"lmarena_en_elo": 1497,
267+
"lmarena_en_elo": 1496,
268268
"lmarena_hard_elo": 1529,
269269
"lmarena_if_elo": 1500,
270270
"lmarena_math_elo": 1501,
271-
"lmarena_vision_elo": 1289,
271+
"lmarena_vision_elo": 1293,
272272
"lmarena_zh_elo": 1557,
273273
"math_500": null,
274274
"mathvista": null,
@@ -326,13 +326,13 @@
326326
"humanity_last_exam": 53.1,
327327
"live_code_bench": null,
328328
"livebench": 76.33,
329-
"lmarena_coding_elo": 1556,
329+
"lmarena_coding_elo": 1549,
330330
"lmarena_creative_elo": 1493,
331-
"lmarena_en_elo": 1503,
331+
"lmarena_en_elo": 1502,
332332
"lmarena_hard_elo": 1536,
333333
"lmarena_if_elo": 1512,
334334
"lmarena_math_elo": 1512,
335-
"lmarena_vision_elo": 1302,
335+
"lmarena_vision_elo": 1304,
336336
"lmarena_zh_elo": 1540,
337337
"math_500": null,
338338
"mathvista": null,
@@ -513,7 +513,7 @@
513513
"humanity_last_exam": null,
514514
"live_code_bench": null,
515515
"livebench": null,
516-
"lmarena_coding_elo": 1521,
516+
"lmarena_coding_elo": 1525,
517517
"lmarena_creative_elo": 1443,
518518
"lmarena_en_elo": 1477,
519519
"lmarena_hard_elo": 1498,
@@ -1171,7 +1171,7 @@
11711171
"humanity_last_exam": 36.24,
11721172
"live_code_bench": null,
11731173
"livebench": 80.28,
1174-
"lmarena_coding_elo": 1534,
1174+
"lmarena_coding_elo": 1457,
11751175
"lmarena_creative_elo": 1461,
11761176
"lmarena_en_elo": 1482,
11771177
"lmarena_hard_elo": 1507,
@@ -2009,7 +2009,7 @@
20092009
"livebench": 67.96,
20102010
"lmarena_coding_elo": null,
20112011
"lmarena_creative_elo": null,
2012-
"lmarena_en_elo": 1493,
2012+
"lmarena_en_elo": 1480,
20132013
"lmarena_hard_elo": null,
20142014
"lmarena_if_elo": null,
20152015
"lmarena_math_elo": null,

0 commit comments

Comments
 (0)