bradsansnow commited on
Commit
9268ae9
·
1 Parent(s): aec8169

Sync leaderboard: EOG + EVA, GitHub README 2026-07-23

Browse files

eog:
old: Claude Fable 5 (with fallback)=51.1, Gemini 3.5 Flash=50.1, GPT-5.5 (xhigh)=46.6, Kimi K3=45.3, Qwen3.7 Max=45
new: Claude Fable 5 (with fallback)=51.1, Gemini 3.5 Flash=50.1, Muse Spark 1.1 (xhigh)=47.2, GPT-5.5 (xhigh)=46.6, Kimi K3=45.3

.claude/skills/sync-nowai-leaderboard/github-readme/README.md CHANGED
@@ -43,7 +43,7 @@ Voice agents evaluated on task accuracy (EVA-Accuracy) and conversational experi
43
  <!-- LEADERBOARD:START -->
44
  <!-- Auto-generated from data/leaderboard.json by render-readme.mjs. Do not edit by hand. -->
45
 
46
- _Snapshot synced 2026-07-17. See the live leaderboards for the full, always-current rankings._
47
 
48
  ### EnterpriseOps-Gym — Top 5
49
 
@@ -53,9 +53,9 @@ _Task Success Rate · Oracle mode_
53
  | ---: | :--- | :--- | ---: |
54
  | 1 | Claude Fable 5 (with fallback) | Anthropic | 51.1% |
55
  | 2 | Gemini 3.5 Flash | Google | 50.1% |
56
- | 3 | GPT-5.5 (xhigh) | OpenAI | 46.6% |
57
- | 4 | Kimi K3 | Kimi | 45.3% |
58
- | 5 | Qwen3.7 Max | Alibaba | 45.0% |
59
 
60
  ### EVA-Bench — Top 3
61
 
 
43
  <!-- LEADERBOARD:START -->
44
  <!-- Auto-generated from data/leaderboard.json by render-readme.mjs. Do not edit by hand. -->
45
 
46
+ _Snapshot synced 2026-07-23. See the live leaderboards for the full, always-current rankings._
47
 
48
  ### EnterpriseOps-Gym — Top 5
49
 
 
53
  | ---: | :--- | :--- | ---: |
54
  | 1 | Claude Fable 5 (with fallback) | Anthropic | 51.1% |
55
  | 2 | Gemini 3.5 Flash | Google | 50.1% |
56
+ | 3 | Muse Spark 1.1 (xhigh) | Muse | 47.2% |
57
+ | 4 | GPT-5.5 (xhigh) | OpenAI | 46.6% |
58
+ | 5 | Kimi K3 | Kimi | 45.3% |
59
 
60
  ### EVA-Bench — Top 3
61
 
data/leaderboard.json CHANGED
@@ -1,15 +1,15 @@
1
  {
2
- "updated": "2026-07-17T21:00:09.508Z",
3
  "sources": {
4
  "eog": {
5
  "url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa/",
6
- "hash": "sha256:617ad42070acaf5d70ca7782c7f1e4960d26f56cd228667a86f89d51abc580c8",
7
- "fetchedAt": "2026-07-17T21:00:09.508Z"
8
  },
9
  "eva": {
10
  "url": "https://raw.githubusercontent.com/ServiceNow/eva/main/website/src/data/leaderboardStats.json",
11
  "hash": "sha256:e71bd9e18c294322e706ebc9ca8ac477be01c8caf5d9c25b3121566753f35b77",
12
- "fetchedAt": "2026-07-17T21:00:09.508Z"
13
  }
14
  },
15
  "eog": {
@@ -31,24 +31,24 @@
31
  },
32
  {
33
  "rank": 3,
 
 
 
 
 
 
 
34
  "model": "GPT-5.5 (xhigh)",
35
  "org": "OpenAI",
36
  "score": 46.6,
37
  "bar": 47
38
  },
39
  {
40
- "rank": 4,
41
  "model": "Kimi K3",
42
  "org": "Kimi",
43
  "score": 45.3,
44
  "bar": 45
45
- },
46
- {
47
- "rank": 5,
48
- "model": "Qwen3.7 Max",
49
- "org": "Alibaba",
50
- "score": 45,
51
- "bar": 45
52
  }
53
  ]
54
  },
 
1
  {
2
+ "updated": "2026-07-23T21:00:12.551Z",
3
  "sources": {
4
  "eog": {
5
  "url": "https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa/",
6
+ "hash": "sha256:a6b866aff1f6925b3ffca8d6428f1cc009d91a1a9bbcef7a72db7991bf800fc9",
7
+ "fetchedAt": "2026-07-23T21:00:12.551Z"
8
  },
9
  "eva": {
10
  "url": "https://raw.githubusercontent.com/ServiceNow/eva/main/website/src/data/leaderboardStats.json",
11
  "hash": "sha256:e71bd9e18c294322e706ebc9ca8ac477be01c8caf5d9c25b3121566753f35b77",
12
+ "fetchedAt": "2026-07-23T21:00:12.551Z"
13
  }
14
  },
15
  "eog": {
 
31
  },
32
  {
33
  "rank": 3,
34
+ "model": "Muse Spark 1.1 (xhigh)",
35
+ "org": "Muse",
36
+ "score": 47.2,
37
+ "bar": 47
38
+ },
39
+ {
40
+ "rank": 4,
41
  "model": "GPT-5.5 (xhigh)",
42
  "org": "OpenAI",
43
  "score": 46.6,
44
  "bar": 47
45
  },
46
  {
47
+ "rank": 5,
48
  "model": "Kimi K3",
49
  "org": "Kimi",
50
  "score": 45.3,
51
  "bar": 45
 
 
 
 
 
 
 
52
  }
53
  ]
54
  },