Skip to content

Commit 656da2a

Browse files
author
AGI Clock Bot
committed
data: update benchmark scores 2026-07-17
1 parent 7979a8a commit 656da2a

1 file changed

Lines changed: 52 additions & 44 deletions

File tree

src/data/live.json

Lines changed: 52 additions & 44 deletions
Original file line numberDiff line numberDiff line change
@@ -1,14 +1,14 @@
11
{
22
"schemaVersion": 2,
33
"formula": "agi-readiness-v2",
4-
"updatedAt": "2026-07-16T05:14:47.391Z",
5-
"agiReadiness": 75.6,
6-
"agiDistance": 24.4,
4+
"updatedAt": "2026-07-17T05:17:41.505Z",
5+
"agiReadiness": 76.1,
6+
"agiDistance": 23.9,
77
"bottleneckDimension": {
88
"id": "agentsAutonomy",
99
"name": "Agents + Autonomy",
10-
"score": 59.1,
11-
"distance": 40.9
10+
"score": 59.8,
11+
"distance": 40.2
1212
},
1313
"eta": {
1414
"status": "collecting_baseline",
@@ -40,17 +40,17 @@
4040
"name": "Agents + Autonomy",
4141
"weight": 0.3,
4242
"description": "Computer use, long-horizon professional workflows, planning, and applied office/finance tasks.",
43-
"score": 59.1,
44-
"distance": 40.9,
43+
"score": 59.8,
44+
"distance": 40.2,
4545
"benchmarkCount": 6
4646
},
4747
{
4848
"id": "toolSearchReliability",
4949
"name": "Tool Use + Search Reliability",
5050
"weight": 0.15,
5151
"description": "Tool calling, web/search persistence, user-policy following, and repeated agent reliability proxies.",
52-
"score": 81.6,
53-
"distance": 18.4,
52+
"score": 83.6,
53+
"distance": 16.4,
5454
"benchmarkCount": 8
5555
}
5656
],
@@ -80,7 +80,7 @@
8080
"scoreRaw": 0.647,
8181
"modelName": "Claude Mythos Preview",
8282
"modelId": "claude-mythos-preview",
83-
"totalModels": 88,
83+
"totalModels": 89,
8484
"trend": "flat",
8585
"source": "llm-stats:humanity's-last-exam",
8686
"description": "2,500 expert-vetted multimodal questions across math, science, humanities, and vision."
@@ -95,7 +95,7 @@
9595
"scoreRaw": 0.946,
9696
"modelName": "GPT-5.6 Sol",
9797
"modelId": "gpt-5.6-sol",
98-
"totalModels": 231,
98+
"totalModels": 232,
9999
"trend": "flat",
100100
"source": "llm-stats:gpqa",
101101
"description": "PhD-level questions in biology, chemistry, and physics."
@@ -126,7 +126,7 @@
126126
"modelName": "GPT-5.6 Sol",
127127
"modelId": "gpt-5.6-sol",
128128
"totalModels": 17,
129-
"trend": "up",
129+
"trend": "flat",
130130
"source": "llm-stats:frontiermath",
131131
"description": "Advanced unpublished mathematical problems authored and reviewed by expert mathematicians."
132132
},
@@ -200,7 +200,7 @@
200200
"scoreRaw": 0.888,
201201
"modelName": "GPT-5.6 Sol",
202202
"modelId": "gpt-5.6-sol",
203-
"totalModels": 11,
203+
"totalModels": 12,
204204
"trend": "flat",
205205
"source": "llm-stats:terminal-bench-2.1",
206206
"description": "Updated terminal-agent benchmark for autonomous command-line execution."
@@ -226,12 +226,12 @@
226226
"name": "APEX-Agents",
227227
"category": "Long-Horizon Agents",
228228
"dimension": "agentsAutonomy",
229-
"score": 33.8,
230-
"scoreRaw": 0.338,
231-
"modelName": "Seed 2.1 Pro",
232-
"modelId": "seed-2.1-pro",
233-
"totalModels": 6,
234-
"trend": "flat",
229+
"score": 37.6,
230+
"scoreRaw": 0.376,
231+
"modelName": "Kimi K3",
232+
"modelId": "kimi-k3",
233+
"totalModels": 7,
234+
"trend": "up",
235235
"source": "llm-stats:apex-agents",
236236
"description": "Long-horizon professional tasks requiring sustained planning and execution."
237237
},
@@ -275,7 +275,7 @@
275275
"scoreRaw": 0.722,
276276
"modelName": "Seed 2.1 Pro",
277277
"modelId": "seed-2.1-pro",
278-
"totalModels": 6,
278+
"totalModels": 7,
279279
"trend": "flat",
280280
"source": "llm-stats:officeqa-pro",
281281
"description": "Professional knowledge-work tasks involving documents, spreadsheets, and office workflows."
@@ -346,12 +346,12 @@
346346
"name": "MCP Atlas",
347347
"category": "Tool Use",
348348
"dimension": "toolSearchReliability",
349-
"score": 83.8,
350-
"scoreRaw": 0.838,
351-
"modelName": "Seed 2.1 Pro",
352-
"modelId": "seed-2.1-pro",
353-
"totalModels": 28,
354-
"trend": "flat",
349+
"score": 84.2,
350+
"scoreRaw": 0.842,
351+
"modelName": "Kimi K3",
352+
"modelId": "kimi-k3",
353+
"totalModels": 29,
354+
"trend": "up",
355355
"source": "llm-stats:mcp-atlas",
356356
"description": "Scaled tool coordination across complex multi-step tool-use tasks."
357357
},
@@ -361,12 +361,12 @@
361361
"name": "Toolathlon",
362362
"category": "Tool Use",
363363
"dimension": "toolSearchReliability",
364-
"score": 59.9,
365-
"scoreRaw": 0.599,
366-
"modelName": "Claude Opus 4.8",
367-
"modelId": "claude-opus-4-8",
368-
"totalModels": 28,
369-
"trend": "flat",
364+
"score": 73.2,
365+
"scoreRaw": 0.732,
366+
"modelName": "Kimi K3",
367+
"modelId": "kimi-k3",
368+
"totalModels": 29,
369+
"trend": "up",
370370
"source": "llm-stats:toolathlon",
371371
"description": "Multi-tool proficiency across diverse tool-use categories."
372372
},
@@ -376,12 +376,12 @@
376376
"name": "BrowseComp",
377377
"category": "Search",
378378
"dimension": "toolSearchReliability",
379-
"score": 90.4,
380-
"scoreRaw": 0.904,
381-
"modelName": "GPT-5.6 Sol",
382-
"modelId": "gpt-5.6-sol",
383-
"totalModels": 56,
384-
"trend": "flat",
379+
"score": 91.2,
380+
"scoreRaw": 0.912,
381+
"modelName": "Kimi K3",
382+
"modelId": "kimi-k3",
383+
"totalModels": 57,
384+
"trend": "up",
385385
"source": "llm-stats:browsecomp",
386386
"description": "Persistent web browsing and hard-to-find information retrieval."
387387
},
@@ -406,12 +406,12 @@
406406
"name": "DeepSearchQA",
407407
"category": "Search",
408408
"dimension": "toolSearchReliability",
409-
"score": 93.1,
410-
"scoreRaw": 0.931,
411-
"modelName": "Claude Opus 4.8",
412-
"modelId": "claude-opus-4-8",
413-
"totalModels": 7,
414-
"trend": "flat",
409+
"score": 95,
410+
"scoreRaw": 0.95,
411+
"modelName": "Kimi K3",
412+
"modelId": "kimi-k3",
413+
"totalModels": 8,
414+
"trend": "up",
415415
"source": "llm-stats:deepsearchqa",
416416
"description": "Multi-hop deep search and answer retrieval."
417417
}
@@ -566,6 +566,14 @@
566566
"bottleneckDimension": "agentsAutonomy",
567567
"updatedAt": "2026-07-16T05:14:47.391Z",
568568
"formula": "agi-readiness-v2"
569+
},
570+
{
571+
"date": "Jul 17, 2026",
572+
"agiReadiness": 76.1,
573+
"agiDistance": 23.9,
574+
"bottleneckDimension": "agentsAutonomy",
575+
"updatedAt": "2026-07-17T05:17:41.505Z",
576+
"formula": "agi-readiness-v2"
569577
}
570578
],
571579
"legacyHistory": [

0 commit comments

Comments
 (0)