omnilane 0.44.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,19 +1,25 @@
1
1
  {
2
2
  "schema_version": 1,
3
3
  "snapshot": {
4
- "id": "aa-v4.2-2026-09-07-v1",
4
+ "id": "aa-v4.3.2-2026-09-22-v1",
5
5
  "benchmark": "Artificial Analysis Intelligence Index",
6
- "benchmark_version": "4.2",
7
- "as_of": "2026-09-07",
6
+ "benchmark_version": "4.3.2",
7
+ "as_of": "2026-09-22",
8
8
  "timezone": "Asia/Taipei",
9
9
  "frozen": true,
10
10
  "approval": {
11
11
  "status": "approved",
12
- "scope": "model-governance-proposal-v2",
13
- "source": "docs/model-governance-proposal.md",
12
+ "scope": "aa-v4.3.2-rebaseline",
13
+ "source": "docs/reports/aa-rebaseline-2026-09-22.md",
14
14
  "estimated_scores": "approved_provisional"
15
15
  },
16
- "automatic_refresh_grants_authority": false
16
+ "automatic_refresh_grants_authority": false,
17
+ "source": {
18
+ "extract": "docs/reports/aa-v4.3.2-extract-2026-09-22.json",
19
+ "page_url": "https://artificialanalysis.ai/models/grok-4-7",
20
+ "page_sha256": "12fe48f9352e7697fec43e5964044ca3da8d0db562de29cb341cf18ef07a03c5",
21
+ "fetched_at": "2026-09-22T01:49:21+08:00"
22
+ }
17
23
  },
18
24
  "policy": {
19
25
  "decision": "target_score <= min(caller_score, inherited_ceiling)",
@@ -40,77 +46,77 @@
40
46
  },
41
47
  "scored_configs": [
42
48
  {
43
- "id": "codex/gpt-6-astra",
44
- "vendor": "codex",
45
- "model": "gpt-6-astra",
49
+ "id": "claude/claude-fable-5-1",
50
+ "vendor": "claude",
51
+ "model": "claude-fable-5-1",
46
52
  "effort": "max",
47
- "reasoning": "reasoning",
48
- "fallback": null,
49
- "aa_slug": "gpt-6-astra",
50
- "score": 55,
53
+ "reasoning": "adaptive",
54
+ "fallback": "default",
55
+ "aa_slug": "claude-fable-5-1",
56
+ "score": 53,
51
57
  "estimated": false,
52
58
  "evidence_marker": "unmarked",
53
- "benchmark_version": "4.2",
54
- "as_of": "2026-09-07",
59
+ "benchmark_version": "4.3.2",
60
+ "as_of": "2026-09-22",
55
61
  "source_urls": [
56
- "https://artificialanalysis.ai/models/gpt-6-astra",
57
- "https://artificialanalysis.ai/models/releases/gpt-6-astra"
62
+ "https://artificialanalysis.ai/models/claude-fable-5-1"
58
63
  ],
59
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
64
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
60
65
  "transport_mapping": {
61
66
  "status": "unknown",
62
67
  "candidate_model_ids": [
63
- "gpt-6-astra"
68
+ "claude-fable-5-1"
64
69
  ],
65
70
  "runtime_verified": false,
66
71
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
67
- }
72
+ },
73
+ "score_raw": 53.35
68
74
  },
69
75
  {
70
- "id": "codex/gpt-6-astra-xhigh",
71
- "vendor": "codex",
72
- "model": "gpt-6-astra",
76
+ "id": "claude/claude-fable-5-1-xhigh",
77
+ "vendor": "claude",
78
+ "model": "claude-fable-5-1",
73
79
  "effort": "xhigh",
74
- "reasoning": "reasoning",
75
- "fallback": null,
76
- "aa_slug": "gpt-6-astra-xhigh",
77
- "score": 54,
80
+ "reasoning": "adaptive",
81
+ "fallback": "default",
82
+ "aa_slug": "claude-fable-5-1-xhigh",
83
+ "score": 53,
78
84
  "estimated": false,
79
85
  "evidence_marker": "unmarked",
80
- "benchmark_version": "4.2",
81
- "as_of": "2026-09-07",
86
+ "benchmark_version": "4.3.2",
87
+ "as_of": "2026-09-22",
82
88
  "source_urls": [
83
- "https://artificialanalysis.ai/models/gpt-6-astra-xhigh",
84
- "https://artificialanalysis.ai/models/releases/gpt-6-astra"
89
+ "https://artificialanalysis.ai/models/claude-fable-5-1-xhigh"
85
90
  ],
86
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
91
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
87
92
  "transport_mapping": {
88
93
  "status": "unknown",
89
94
  "candidate_model_ids": [
90
- "gpt-6-astra"
95
+ "claude-fable-5-1"
91
96
  ],
92
97
  "runtime_verified": false,
93
98
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
94
- }
99
+ },
100
+ "score_raw": 53.2
95
101
  },
96
102
  {
97
- "id": "codex/gpt-6-astra-high",
103
+ "id": "codex/gpt-6-astra",
98
104
  "vendor": "codex",
99
105
  "model": "gpt-6-astra",
100
- "effort": "high",
106
+ "effort": "max",
101
107
  "reasoning": "reasoning",
102
108
  "fallback": null,
103
- "aa_slug": "gpt-6-astra-high",
109
+ "aa_slug": "gpt-6-astra",
104
110
  "score": 53,
105
111
  "estimated": false,
106
112
  "evidence_marker": "unmarked",
107
- "benchmark_version": "4.2",
108
- "as_of": "2026-09-07",
113
+ "benchmark_version": "4.3.2",
114
+ "as_of": "2026-09-22",
109
115
  "source_urls": [
110
- "https://artificialanalysis.ai/models/gpt-6-astra-high",
116
+ "https://artificialanalysis.ai/models/gpt-6-astra",
111
117
  "https://artificialanalysis.ai/models/releases/gpt-6-astra"
112
118
  ],
113
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
119
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
114
120
  "transport_mapping": {
115
121
  "status": "unknown",
116
122
  "candidate_model_ids": [
@@ -118,26 +124,27 @@
118
124
  ],
119
125
  "runtime_verified": false,
120
126
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
121
- }
127
+ },
128
+ "score_raw": 52.67
122
129
  },
123
130
  {
124
- "id": "codex/gpt-6-astra-medium",
131
+ "id": "codex/gpt-6-astra-xhigh",
125
132
  "vendor": "codex",
126
133
  "model": "gpt-6-astra",
127
- "effort": "medium",
134
+ "effort": "xhigh",
128
135
  "reasoning": "reasoning",
129
136
  "fallback": null,
130
- "aa_slug": "gpt-6-astra-medium",
137
+ "aa_slug": "gpt-6-astra-xhigh",
131
138
  "score": 52,
132
139
  "estimated": false,
133
140
  "evidence_marker": "unmarked",
134
- "benchmark_version": "4.2",
135
- "as_of": "2026-09-07",
141
+ "benchmark_version": "4.3.2",
142
+ "as_of": "2026-09-22",
136
143
  "source_urls": [
137
- "https://artificialanalysis.ai/models/gpt-6-astra-medium",
144
+ "https://artificialanalysis.ai/models/gpt-6-astra-xhigh",
138
145
  "https://artificialanalysis.ai/models/releases/gpt-6-astra"
139
146
  ],
140
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
147
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
141
148
  "transport_mapping": {
142
149
  "status": "unknown",
143
150
  "candidate_model_ids": [
@@ -145,53 +152,54 @@
145
152
  ],
146
153
  "runtime_verified": false,
147
154
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
148
- }
155
+ },
156
+ "score_raw": 52.39
149
157
  },
150
158
  {
151
- "id": "codex/gpt-6-astra-low",
152
- "vendor": "codex",
153
- "model": "gpt-6-astra",
154
- "effort": "low",
155
- "reasoning": "reasoning",
156
- "fallback": null,
157
- "aa_slug": "gpt-6-astra-low",
158
- "score": 49,
159
+ "id": "claude/claude-fable-5-1-high",
160
+ "vendor": "claude",
161
+ "model": "claude-fable-5-1",
162
+ "effort": "high",
163
+ "reasoning": "adaptive",
164
+ "fallback": "default",
165
+ "aa_slug": "claude-fable-5-1-high",
166
+ "score": 51,
159
167
  "estimated": false,
160
168
  "evidence_marker": "unmarked",
161
- "benchmark_version": "4.2",
162
- "as_of": "2026-09-07",
169
+ "benchmark_version": "4.3.2",
170
+ "as_of": "2026-09-22",
163
171
  "source_urls": [
164
- "https://artificialanalysis.ai/models/gpt-6-astra-low",
165
- "https://artificialanalysis.ai/models/releases/gpt-6-astra"
172
+ "https://artificialanalysis.ai/models/claude-fable-5-1-high"
166
173
  ],
167
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
174
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
168
175
  "transport_mapping": {
169
176
  "status": "unknown",
170
177
  "candidate_model_ids": [
171
- "gpt-6-astra"
178
+ "claude-fable-5-1"
172
179
  ],
173
180
  "runtime_verified": false,
174
181
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
175
- }
182
+ },
183
+ "score_raw": 51.15
176
184
  },
177
185
  {
178
- "id": "codex/gpt-6-astra-non-reasoning",
186
+ "id": "codex/gpt-6-astra-high",
179
187
  "vendor": "codex",
180
188
  "model": "gpt-6-astra",
181
- "effort": null,
182
- "reasoning": "non-reasoning",
189
+ "effort": "high",
190
+ "reasoning": "reasoning",
183
191
  "fallback": null,
184
- "aa_slug": "gpt-6-astra-non-reasoning",
185
- "score": 48,
192
+ "aa_slug": "gpt-6-astra-high",
193
+ "score": 51,
186
194
  "estimated": false,
187
195
  "evidence_marker": "unmarked",
188
- "benchmark_version": "4.2",
189
- "as_of": "2026-09-07",
196
+ "benchmark_version": "4.3.2",
197
+ "as_of": "2026-09-22",
190
198
  "source_urls": [
191
- "https://artificialanalysis.ai/models/gpt-6-astra-non-reasoning",
199
+ "https://artificialanalysis.ai/models/gpt-6-astra-high",
192
200
  "https://artificialanalysis.ai/models/releases/gpt-6-astra"
193
201
  ],
194
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
202
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
195
203
  "transport_mapping": {
196
204
  "status": "unknown",
197
205
  "candidate_model_ids": [
@@ -199,628 +207,717 @@
199
207
  ],
200
208
  "runtime_verified": false,
201
209
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
202
- }
210
+ },
211
+ "score_raw": 50.92
203
212
  },
204
213
  {
205
- "id": "codex/gpt-5-6-sol",
206
- "vendor": "codex",
207
- "model": "gpt-5.6-sol",
214
+ "id": "claude/claude-opus-5",
215
+ "vendor": "claude",
216
+ "model": "claude-opus-5",
208
217
  "effort": "max",
209
- "reasoning": "reasoning",
218
+ "reasoning": "adaptive",
210
219
  "fallback": null,
211
- "aa_slug": "gpt-5-6-sol",
220
+ "aa_slug": "claude-opus-5",
212
221
  "score": 51,
213
222
  "estimated": false,
214
223
  "evidence_marker": "unmarked",
215
- "benchmark_version": "4.2",
216
- "as_of": "2026-09-07",
224
+ "benchmark_version": "4.3.2",
225
+ "as_of": "2026-09-22",
217
226
  "source_urls": [
218
- "https://artificialanalysis.ai/models/gpt-5-6-sol",
219
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
227
+ "https://artificialanalysis.ai/models/claude-opus-5"
220
228
  ],
221
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
229
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
222
230
  "transport_mapping": {
223
231
  "status": "unknown",
224
232
  "candidate_model_ids": [
225
- "gpt-5.6-sol"
233
+ "claude-opus-5"
226
234
  ],
227
235
  "runtime_verified": false,
228
236
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
229
- }
237
+ },
238
+ "score_raw": 50.78
230
239
  },
231
240
  {
232
- "id": "codex/gpt-5-6-sol-xhigh",
233
- "vendor": "codex",
234
- "model": "gpt-5.6-sol",
241
+ "id": "claude/claude-opus-5-xhigh",
242
+ "vendor": "claude",
243
+ "model": "claude-opus-5",
235
244
  "effort": "xhigh",
236
- "reasoning": "reasoning",
245
+ "reasoning": "adaptive",
237
246
  "fallback": null,
238
- "aa_slug": "gpt-5-6-sol-xhigh",
247
+ "aa_slug": "claude-opus-5-xhigh",
239
248
  "score": 50,
240
249
  "estimated": false,
241
250
  "evidence_marker": "unmarked",
242
- "benchmark_version": "4.2",
243
- "as_of": "2026-09-07",
251
+ "benchmark_version": "4.3.2",
252
+ "as_of": "2026-09-22",
244
253
  "source_urls": [
245
- "https://artificialanalysis.ai/models/gpt-5-6-sol-xhigh",
246
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
254
+ "https://artificialanalysis.ai/models/claude-opus-5-xhigh"
247
255
  ],
248
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
256
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
249
257
  "transport_mapping": {
250
258
  "status": "unknown",
251
259
  "candidate_model_ids": [
252
- "gpt-5.6-sol"
260
+ "claude-opus-5"
253
261
  ],
254
262
  "runtime_verified": false,
255
263
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
256
- }
264
+ },
265
+ "score_raw": 49.68
257
266
  },
258
267
  {
259
- "id": "codex/gpt-5-6-sol-high",
260
- "vendor": "codex",
261
- "model": "gpt-5.6-sol",
262
- "effort": "high",
263
- "reasoning": "reasoning",
264
- "fallback": null,
265
- "aa_slug": "gpt-5-6-sol-high",
266
- "score": 48,
268
+ "id": "claude/claude-fable-5",
269
+ "vendor": "claude",
270
+ "model": "claude-fable-5",
271
+ "effort": "max",
272
+ "reasoning": "adaptive",
273
+ "fallback": "claude-opus-4-8",
274
+ "aa_slug": "claude-fable-5",
275
+ "score": 50,
267
276
  "estimated": false,
268
277
  "evidence_marker": "unmarked",
269
- "benchmark_version": "4.2",
270
- "as_of": "2026-09-07",
278
+ "benchmark_version": "4.3.2",
279
+ "as_of": "2026-09-22",
271
280
  "source_urls": [
272
- "https://artificialanalysis.ai/models/gpt-5-6-sol-high",
273
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
281
+ "https://artificialanalysis.ai/models/claude-fable-5"
274
282
  ],
275
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
283
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
276
284
  "transport_mapping": {
277
285
  "status": "unknown",
278
286
  "candidate_model_ids": [
279
- "gpt-5.6-sol"
287
+ "claude-fable-5"
280
288
  ],
281
289
  "runtime_verified": false,
282
290
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
283
- }
291
+ },
292
+ "score_raw": 49.63
284
293
  },
285
294
  {
286
- "id": "codex/gpt-5-6-sol-medium",
295
+ "id": "codex/gpt-6-astra-medium",
287
296
  "vendor": "codex",
288
- "model": "gpt-5.6-sol",
297
+ "model": "gpt-6-astra",
289
298
  "effort": "medium",
290
299
  "reasoning": "reasoning",
291
300
  "fallback": null,
292
- "aa_slug": "gpt-5-6-sol-medium",
293
- "score": 46,
301
+ "aa_slug": "gpt-6-astra-medium",
302
+ "score": 50,
294
303
  "estimated": false,
295
304
  "evidence_marker": "unmarked",
296
- "benchmark_version": "4.2",
297
- "as_of": "2026-09-07",
305
+ "benchmark_version": "4.3.2",
306
+ "as_of": "2026-09-22",
298
307
  "source_urls": [
299
- "https://artificialanalysis.ai/models/gpt-5-6-sol-medium",
300
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
308
+ "https://artificialanalysis.ai/models/gpt-6-astra-medium",
309
+ "https://artificialanalysis.ai/models/releases/gpt-6-astra"
301
310
  ],
302
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
311
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
303
312
  "transport_mapping": {
304
313
  "status": "unknown",
305
314
  "candidate_model_ids": [
306
- "gpt-5.6-sol"
315
+ "gpt-6-astra"
307
316
  ],
308
317
  "runtime_verified": false,
309
318
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
310
- }
319
+ },
320
+ "score_raw": 49.57
311
321
  },
312
322
  {
313
- "id": "codex/gpt-5-6-sol-low",
314
- "vendor": "codex",
315
- "model": "gpt-5.6-sol",
316
- "effort": "low",
317
- "reasoning": "reasoning",
318
- "fallback": null,
319
- "aa_slug": "gpt-5-6-sol-low",
320
- "score": 41,
323
+ "id": "claude/claude-fable-5-1-medium",
324
+ "vendor": "claude",
325
+ "model": "claude-fable-5-1",
326
+ "effort": "medium",
327
+ "reasoning": "adaptive",
328
+ "fallback": "default",
329
+ "aa_slug": "claude-fable-5-1-medium",
330
+ "score": 49,
321
331
  "estimated": false,
322
332
  "evidence_marker": "unmarked",
323
- "benchmark_version": "4.2",
324
- "as_of": "2026-09-07",
333
+ "benchmark_version": "4.3.2",
334
+ "as_of": "2026-09-22",
325
335
  "source_urls": [
326
- "https://artificialanalysis.ai/models/gpt-5-6-sol-low",
327
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
336
+ "https://artificialanalysis.ai/models/claude-fable-5-1-medium"
328
337
  ],
329
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
338
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
330
339
  "transport_mapping": {
331
340
  "status": "unknown",
332
341
  "candidate_model_ids": [
333
- "gpt-5.6-sol"
342
+ "claude-fable-5-1"
334
343
  ],
335
344
  "runtime_verified": false,
336
345
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
337
- }
346
+ },
347
+ "score_raw": 48.92
338
348
  },
339
349
  {
340
- "id": "codex/gpt-5-6-sol-non-reasoning",
341
- "vendor": "codex",
342
- "model": "gpt-5.6-sol",
343
- "effort": null,
344
- "reasoning": "non-reasoning",
350
+ "id": "claude/claude-opus-5-high",
351
+ "vendor": "claude",
352
+ "model": "claude-opus-5",
353
+ "effort": "high",
354
+ "reasoning": "adaptive",
345
355
  "fallback": null,
346
- "aa_slug": "gpt-5-6-sol-non-reasoning",
347
- "score": 33,
348
- "estimated": true,
349
- "evidence_marker": "estimated",
350
- "benchmark_version": "4.2",
351
- "as_of": "2026-09-07",
356
+ "aa_slug": "claude-opus-5-high",
357
+ "score": 48,
358
+ "estimated": false,
359
+ "evidence_marker": "unmarked",
360
+ "benchmark_version": "4.3.2",
361
+ "as_of": "2026-09-22",
352
362
  "source_urls": [
353
- "https://artificialanalysis.ai/models/gpt-5-6-sol-non-reasoning",
354
- "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
363
+ "https://artificialanalysis.ai/models/claude-opus-5-high"
355
364
  ],
356
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
365
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
357
366
  "transport_mapping": {
358
367
  "status": "unknown",
359
368
  "candidate_model_ids": [
360
- "gpt-5.6-sol"
369
+ "claude-opus-5"
361
370
  ],
362
371
  "runtime_verified": false,
363
372
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
364
- }
373
+ },
374
+ "score_raw": 48.12
365
375
  },
366
376
  {
367
- "id": "codex/gpt-5-6-terra",
377
+ "id": "codex/gpt-5-6-sol",
368
378
  "vendor": "codex",
369
- "model": "gpt-5.6-terra",
379
+ "model": "gpt-5.6-sol",
370
380
  "effort": "max",
371
381
  "reasoning": "reasoning",
372
382
  "fallback": null,
373
- "aa_slug": "gpt-5-6-terra",
383
+ "aa_slug": "gpt-5-6-sol",
374
384
  "score": 47,
375
385
  "estimated": false,
376
386
  "evidence_marker": "unmarked",
377
- "benchmark_version": "4.2",
378
- "as_of": "2026-09-07",
387
+ "benchmark_version": "4.3.2",
388
+ "as_of": "2026-09-22",
379
389
  "source_urls": [
380
- "https://artificialanalysis.ai/models/gpt-5-6-terra",
381
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
390
+ "https://artificialanalysis.ai/models/gpt-5-6-sol",
391
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
382
392
  ],
383
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
393
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
384
394
  "transport_mapping": {
385
395
  "status": "unknown",
386
396
  "candidate_model_ids": [
387
- "gpt-5.6-terra"
397
+ "gpt-5.6-sol"
388
398
  ],
389
399
  "runtime_verified": false,
390
400
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
391
- }
401
+ },
402
+ "score_raw": 46.97
392
403
  },
393
404
  {
394
- "id": "codex/gpt-5-6-terra-xhigh",
395
- "vendor": "codex",
396
- "model": "gpt-5.6-terra",
397
- "effort": "xhigh",
398
- "reasoning": "reasoning",
399
- "fallback": null,
400
- "aa_slug": "gpt-5-6-terra-xhigh",
401
- "score": 44,
405
+ "id": "claude/claude-fable-5-1-low",
406
+ "vendor": "claude",
407
+ "model": "claude-fable-5-1",
408
+ "effort": "low",
409
+ "reasoning": "adaptive",
410
+ "fallback": "default",
411
+ "aa_slug": "claude-fable-5-1-low",
412
+ "score": 47,
402
413
  "estimated": false,
403
414
  "evidence_marker": "unmarked",
404
- "benchmark_version": "4.2",
405
- "as_of": "2026-09-07",
415
+ "benchmark_version": "4.3.2",
416
+ "as_of": "2026-09-22",
406
417
  "source_urls": [
407
- "https://artificialanalysis.ai/models/gpt-5-6-terra-xhigh",
408
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
418
+ "https://artificialanalysis.ai/models/claude-fable-5-1-low"
409
419
  ],
410
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
420
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
411
421
  "transport_mapping": {
412
422
  "status": "unknown",
413
423
  "candidate_model_ids": [
414
- "gpt-5.6-terra"
424
+ "claude-fable-5-1"
415
425
  ],
416
426
  "runtime_verified": false,
417
427
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
418
- }
428
+ },
429
+ "score_raw": 46.82
419
430
  },
420
431
  {
421
- "id": "codex/gpt-5-6-terra-high",
422
- "vendor": "codex",
423
- "model": "gpt-5.6-terra",
424
- "effort": "high",
432
+ "id": "grok/grok-4-7",
433
+ "vendor": "grok",
434
+ "model": "grok-4.7",
435
+ "effort": "xhigh",
425
436
  "reasoning": "reasoning",
426
437
  "fallback": null,
427
- "aa_slug": "gpt-5-6-terra-high",
428
- "score": 41,
438
+ "aa_slug": "grok-4-7",
439
+ "score": 46,
429
440
  "estimated": false,
430
441
  "evidence_marker": "unmarked",
431
- "benchmark_version": "4.2",
432
- "as_of": "2026-09-07",
442
+ "benchmark_version": "4.3.2",
443
+ "as_of": "2026-09-22",
433
444
  "source_urls": [
434
- "https://artificialanalysis.ai/models/gpt-5-6-terra-high",
435
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
445
+ "https://artificialanalysis.ai/models/grok-4-7"
436
446
  ],
437
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
447
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
438
448
  "transport_mapping": {
439
449
  "status": "unknown",
440
450
  "candidate_model_ids": [
441
- "gpt-5.6-terra"
451
+ "grok-4.7"
442
452
  ],
443
453
  "runtime_verified": false,
444
454
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
445
- }
455
+ },
456
+ "score_raw": 46.45
446
457
  },
447
458
  {
448
- "id": "codex/gpt-5-6-terra-medium",
449
- "vendor": "codex",
450
- "model": "gpt-5.6-terra",
451
- "effort": "medium",
459
+ "id": "grok/grok-4-7-high",
460
+ "vendor": "grok",
461
+ "model": "grok-4.7",
462
+ "effort": "high",
452
463
  "reasoning": "reasoning",
453
464
  "fallback": null,
454
- "aa_slug": "gpt-5-6-terra-medium",
455
- "score": 37,
456
- "estimated": true,
457
- "evidence_marker": "estimated",
458
- "benchmark_version": "4.2",
459
- "as_of": "2026-09-07",
465
+ "aa_slug": "grok-4-7-high",
466
+ "score": 46,
467
+ "estimated": false,
468
+ "evidence_marker": "unmarked",
469
+ "benchmark_version": "4.3.2",
470
+ "as_of": "2026-09-22",
460
471
  "source_urls": [
461
- "https://artificialanalysis.ai/models/gpt-5-6-terra-medium",
462
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
472
+ "https://artificialanalysis.ai/models/grok-4-7-high"
463
473
  ],
464
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
474
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
465
475
  "transport_mapping": {
466
476
  "status": "unknown",
467
477
  "candidate_model_ids": [
468
- "gpt-5.6-terra"
478
+ "grok-4.7"
469
479
  ],
470
480
  "runtime_verified": false,
471
481
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
472
- }
482
+ },
483
+ "score_raw": 46.33
473
484
  },
474
485
  {
475
- "id": "codex/gpt-5-6-terra-low",
486
+ "id": "codex/gpt-6-astra-low",
476
487
  "vendor": "codex",
477
- "model": "gpt-5.6-terra",
488
+ "model": "gpt-6-astra",
478
489
  "effort": "low",
479
490
  "reasoning": "reasoning",
480
491
  "fallback": null,
481
- "aa_slug": "gpt-5-6-terra-low",
482
- "score": 32,
483
- "estimated": true,
484
- "evidence_marker": "estimated",
485
- "benchmark_version": "4.2",
486
- "as_of": "2026-09-07",
492
+ "aa_slug": "gpt-6-astra-low",
493
+ "score": 46,
494
+ "estimated": false,
495
+ "evidence_marker": "unmarked",
496
+ "benchmark_version": "4.3.2",
497
+ "as_of": "2026-09-22",
487
498
  "source_urls": [
488
- "https://artificialanalysis.ai/models/gpt-5-6-terra-low",
489
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
499
+ "https://artificialanalysis.ai/models/gpt-6-astra-low",
500
+ "https://artificialanalysis.ai/models/releases/gpt-6-astra"
490
501
  ],
491
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
502
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
492
503
  "transport_mapping": {
493
504
  "status": "unknown",
494
505
  "candidate_model_ids": [
495
- "gpt-5.6-terra"
506
+ "gpt-6-astra"
496
507
  ],
497
508
  "runtime_verified": false,
498
509
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
499
- }
510
+ },
511
+ "score_raw": 45.78
500
512
  },
501
513
  {
502
- "id": "codex/gpt-5-6-terra-non-reasoning",
503
- "vendor": "codex",
504
- "model": "gpt-5.6-terra",
505
- "effort": null,
506
- "reasoning": "non-reasoning",
514
+ "id": "claude/claude-opus-5-medium",
515
+ "vendor": "claude",
516
+ "model": "claude-opus-5",
517
+ "effort": "medium",
518
+ "reasoning": "adaptive",
507
519
  "fallback": null,
508
- "aa_slug": "gpt-5-6-terra-non-reasoning",
509
- "score": 26,
510
- "estimated": true,
511
- "evidence_marker": "estimated",
512
- "benchmark_version": "4.2",
513
- "as_of": "2026-09-07",
520
+ "aa_slug": "claude-opus-5-medium",
521
+ "score": 45,
522
+ "estimated": false,
523
+ "evidence_marker": "unmarked",
524
+ "benchmark_version": "4.3.2",
525
+ "as_of": "2026-09-22",
514
526
  "source_urls": [
515
- "https://artificialanalysis.ai/models/gpt-5-6-terra-non-reasoning",
516
- "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
527
+ "https://artificialanalysis.ai/models/claude-opus-5-medium"
517
528
  ],
518
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
529
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
519
530
  "transport_mapping": {
520
531
  "status": "unknown",
521
532
  "candidate_model_ids": [
522
- "gpt-5.6-terra"
533
+ "claude-opus-5"
523
534
  ],
524
535
  "runtime_verified": false,
525
536
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
526
- }
537
+ },
538
+ "score_raw": 44.83
527
539
  },
528
540
  {
529
- "id": "codex/gpt-5-6-luna",
530
- "vendor": "codex",
531
- "model": "gpt-5.6-luna",
532
- "effort": "max",
541
+ "id": "grok/grok-4-6",
542
+ "vendor": "grok",
543
+ "model": "grok-4.6",
544
+ "effort": "high",
533
545
  "reasoning": "reasoning",
534
546
  "fallback": null,
535
- "aa_slug": "gpt-5-6-luna",
536
- "score": 43,
547
+ "aa_slug": "grok-4-6",
548
+ "score": 44,
537
549
  "estimated": false,
538
550
  "evidence_marker": "unmarked",
539
- "benchmark_version": "4.2",
540
- "as_of": "2026-09-07",
551
+ "benchmark_version": "4.3.2",
552
+ "as_of": "2026-09-22",
541
553
  "source_urls": [
542
- "https://artificialanalysis.ai/models/gpt-5-6-luna",
543
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
554
+ "https://artificialanalysis.ai/models/grok-4-6"
544
555
  ],
545
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
556
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
546
557
  "transport_mapping": {
547
558
  "status": "unknown",
548
559
  "candidate_model_ids": [
549
- "gpt-5.6-luna"
560
+ "grok-4.6"
550
561
  ],
551
562
  "runtime_verified": false,
552
563
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
553
- }
564
+ },
565
+ "score_raw": 44.31
554
566
  },
555
567
  {
556
- "id": "codex/gpt-5-6-luna-xhigh",
557
- "vendor": "codex",
558
- "model": "gpt-5.6-luna",
568
+ "id": "grok/grok-4-6-xhigh",
569
+ "vendor": "grok",
570
+ "model": "grok-4.6",
559
571
  "effort": "xhigh",
560
572
  "reasoning": "reasoning",
561
573
  "fallback": null,
562
- "aa_slug": "gpt-5-6-luna-xhigh",
563
- "score": 42,
574
+ "aa_slug": "grok-4-6-xhigh",
575
+ "score": 44,
564
576
  "estimated": false,
565
577
  "evidence_marker": "unmarked",
566
- "benchmark_version": "4.2",
567
- "as_of": "2026-09-07",
578
+ "benchmark_version": "4.3.2",
579
+ "as_of": "2026-09-22",
568
580
  "source_urls": [
569
- "https://artificialanalysis.ai/models/gpt-5-6-luna-xhigh",
570
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
581
+ "https://artificialanalysis.ai/models/grok-4-6-xhigh"
571
582
  ],
572
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
583
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
573
584
  "transport_mapping": {
574
585
  "status": "unknown",
575
586
  "candidate_model_ids": [
576
- "gpt-5.6-luna"
587
+ "grok-4.6"
577
588
  ],
578
589
  "runtime_verified": false,
579
590
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
580
- }
591
+ },
592
+ "score_raw": 44.2
581
593
  },
582
594
  {
583
- "id": "codex/gpt-5-6-luna-high",
595
+ "id": "codex/gpt-5-6-sol-xhigh",
584
596
  "vendor": "codex",
585
- "model": "gpt-5.6-luna",
586
- "effort": "high",
597
+ "model": "gpt-5.6-sol",
598
+ "effort": "xhigh",
587
599
  "reasoning": "reasoning",
588
600
  "fallback": null,
589
- "aa_slug": "gpt-5-6-luna-high",
590
- "score": 37,
591
- "estimated": true,
592
- "evidence_marker": "estimated",
593
- "benchmark_version": "4.2",
594
- "as_of": "2026-09-07",
601
+ "aa_slug": "gpt-5-6-sol-xhigh",
602
+ "score": 44,
603
+ "estimated": false,
604
+ "evidence_marker": "unmarked",
605
+ "benchmark_version": "4.3.2",
606
+ "as_of": "2026-09-22",
595
607
  "source_urls": [
596
- "https://artificialanalysis.ai/models/gpt-5-6-luna-high",
597
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
608
+ "https://artificialanalysis.ai/models/gpt-5-6-sol-xhigh",
609
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
598
610
  ],
599
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
611
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
600
612
  "transport_mapping": {
601
613
  "status": "unknown",
602
614
  "candidate_model_ids": [
603
- "gpt-5.6-luna"
615
+ "gpt-5.6-sol"
604
616
  ],
605
617
  "runtime_verified": false,
606
618
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
607
- }
619
+ },
620
+ "score_raw": 44.01
608
621
  },
609
622
  {
610
- "id": "codex/gpt-5-6-luna-medium",
611
- "vendor": "codex",
612
- "model": "gpt-5.6-luna",
623
+ "id": "grok/grok-4-6-medium",
624
+ "vendor": "grok",
625
+ "model": "grok-4.6",
613
626
  "effort": "medium",
614
627
  "reasoning": "reasoning",
615
628
  "fallback": null,
616
- "aa_slug": "gpt-5-6-luna-medium",
617
- "score": 30,
618
- "estimated": true,
619
- "evidence_marker": "estimated",
620
- "benchmark_version": "4.2",
621
- "as_of": "2026-09-07",
629
+ "aa_slug": "grok-4-6-medium",
630
+ "score": 43,
631
+ "estimated": false,
632
+ "evidence_marker": "unmarked",
633
+ "benchmark_version": "4.3.2",
634
+ "as_of": "2026-09-22",
622
635
  "source_urls": [
623
- "https://artificialanalysis.ai/models/gpt-5-6-luna-medium",
624
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
636
+ "https://artificialanalysis.ai/models/grok-4-6-medium"
625
637
  ],
626
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
638
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
627
639
  "transport_mapping": {
628
640
  "status": "unknown",
629
641
  "candidate_model_ids": [
630
- "gpt-5.6-luna"
642
+ "grok-4.6"
631
643
  ],
632
644
  "runtime_verified": false,
633
645
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
634
- }
646
+ },
647
+ "score_raw": 42.84
635
648
  },
636
649
  {
637
- "id": "codex/gpt-5-6-luna-low",
650
+ "id": "codex/gpt-5-6-sol-high",
638
651
  "vendor": "codex",
639
- "model": "gpt-5.6-luna",
640
- "effort": "low",
652
+ "model": "gpt-5.6-sol",
653
+ "effort": "high",
641
654
  "reasoning": "reasoning",
642
655
  "fallback": null,
643
- "aa_slug": "gpt-5-6-luna-low",
644
- "score": 26,
645
- "estimated": true,
646
- "evidence_marker": "estimated",
647
- "benchmark_version": "4.2",
648
- "as_of": "2026-09-07",
656
+ "aa_slug": "gpt-5-6-sol-high",
657
+ "score": 42,
658
+ "estimated": false,
659
+ "evidence_marker": "unmarked",
660
+ "benchmark_version": "4.3.2",
661
+ "as_of": "2026-09-22",
649
662
  "source_urls": [
650
- "https://artificialanalysis.ai/models/gpt-5-6-luna-low",
651
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
663
+ "https://artificialanalysis.ai/models/gpt-5-6-sol-high",
664
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
652
665
  ],
653
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
666
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
654
667
  "transport_mapping": {
655
668
  "status": "unknown",
656
669
  "candidate_model_ids": [
657
- "gpt-5.6-luna"
670
+ "gpt-5.6-sol"
658
671
  ],
659
672
  "runtime_verified": false,
660
673
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
661
- }
674
+ },
675
+ "score_raw": 42.35
662
676
  },
663
677
  {
664
- "id": "codex/gpt-5-6-luna-non-reasoning",
678
+ "id": "codex/gpt-5-6-terra",
665
679
  "vendor": "codex",
666
- "model": "gpt-5.6-luna",
667
- "effort": null,
668
- "reasoning": "non-reasoning",
680
+ "model": "gpt-5.6-terra",
681
+ "effort": "max",
682
+ "reasoning": "reasoning",
669
683
  "fallback": null,
670
- "aa_slug": "gpt-5-6-luna-non-reasoning",
671
- "score": 19,
672
- "estimated": true,
673
- "evidence_marker": "estimated",
674
- "benchmark_version": "4.2",
675
- "as_of": "2026-09-07",
684
+ "aa_slug": "gpt-5-6-terra",
685
+ "score": 42,
686
+ "estimated": false,
687
+ "evidence_marker": "unmarked",
688
+ "benchmark_version": "4.3.2",
689
+ "as_of": "2026-09-22",
676
690
  "source_urls": [
677
- "https://artificialanalysis.ai/models/gpt-5-6-luna-non-reasoning",
678
- "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
691
+ "https://artificialanalysis.ai/models/gpt-5-6-terra",
692
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
679
693
  ],
680
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
694
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
681
695
  "transport_mapping": {
682
696
  "status": "unknown",
683
697
  "candidate_model_ids": [
684
- "gpt-5.6-luna"
698
+ "gpt-5.6-terra"
685
699
  ],
686
700
  "runtime_verified": false,
687
701
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
688
- }
702
+ },
703
+ "score_raw": 42.08
689
704
  },
690
705
  {
691
- "id": "codex/gpt-5-5",
692
- "vendor": "codex",
693
- "model": "gpt-5.5",
694
- "effort": "xhigh",
695
- "reasoning": "reasoning",
706
+ "id": "claude/claude-opus-4-8",
707
+ "vendor": "claude",
708
+ "model": "claude-opus-4-8",
709
+ "effort": "max",
710
+ "reasoning": "adaptive",
696
711
  "fallback": null,
697
- "aa_slug": "gpt-5-5",
698
- "score": 46,
699
- "estimated": true,
700
- "evidence_marker": "estimated",
701
- "benchmark_version": "4.2",
702
- "as_of": "2026-09-07",
712
+ "aa_slug": "claude-opus-4-8",
713
+ "score": 42,
714
+ "estimated": false,
715
+ "evidence_marker": "unmarked",
716
+ "benchmark_version": "4.3.2",
717
+ "as_of": "2026-09-22",
703
718
  "source_urls": [
704
- "https://artificialanalysis.ai/models/gpt-5-5",
705
- "https://artificialanalysis.ai/models/releases/gpt-5-5"
719
+ "https://artificialanalysis.ai/models/claude-opus-4-8"
706
720
  ],
707
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
721
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
708
722
  "transport_mapping": {
709
723
  "status": "unknown",
710
724
  "candidate_model_ids": [
711
- "gpt-5.5"
725
+ "claude-opus-4-8"
712
726
  ],
713
727
  "runtime_verified": false,
714
728
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
715
- }
729
+ },
730
+ "score_raw": 41.79
716
731
  },
717
732
  {
718
- "id": "codex/gpt-5-5-high",
719
- "vendor": "codex",
720
- "model": "gpt-5.5",
733
+ "id": "gemini/gemini-3-8-flash",
734
+ "vendor": "gemini",
735
+ "model": "gemini-3.8-flash",
721
736
  "effort": "high",
722
- "reasoning": "reasoning",
737
+ "reasoning": "unspecified",
723
738
  "fallback": null,
724
- "aa_slug": "gpt-5-5-high",
725
- "score": 44,
726
- "estimated": true,
727
- "evidence_marker": "estimated",
728
- "benchmark_version": "4.2",
729
- "as_of": "2026-09-07",
739
+ "aa_slug": "gemini-3-8-flash",
740
+ "score": 41,
741
+ "estimated": false,
742
+ "evidence_marker": "unmarked",
743
+ "benchmark_version": "4.3.2",
744
+ "as_of": "2026-09-22",
730
745
  "source_urls": [
731
- "https://artificialanalysis.ai/models/gpt-5-5-high",
732
- "https://artificialanalysis.ai/models/releases/gpt-5-5"
746
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-medium"
733
747
  ],
734
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
748
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
735
749
  "transport_mapping": {
736
750
  "status": "unknown",
737
751
  "candidate_model_ids": [
738
- "gpt-5.5"
752
+ "gemini-3.8-flash-high"
739
753
  ],
740
754
  "runtime_verified": false,
741
- "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
742
- }
755
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
756
+ },
757
+ "score_raw": 40.93
743
758
  },
744
759
  {
745
- "id": "codex/gpt-5-5-medium",
746
- "vendor": "codex",
747
- "model": "gpt-5.5",
748
- "effort": "medium",
749
- "reasoning": "reasoning",
750
- "fallback": null,
751
- "aa_slug": "gpt-5-5-medium",
752
- "score": 42,
760
+ "id": "claude/claude-opus-4-7",
761
+ "vendor": "claude",
762
+ "model": "claude-opus-4-7",
763
+ "effort": "max",
764
+ "reasoning": "adaptive",
765
+ "fallback": null,
766
+ "aa_slug": "claude-opus-4-7",
767
+ "score": 41,
753
768
  "estimated": true,
754
769
  "evidence_marker": "estimated",
755
- "benchmark_version": "4.2",
756
- "as_of": "2026-09-07",
770
+ "benchmark_version": "4.3.2",
771
+ "as_of": "2026-09-22",
757
772
  "source_urls": [
758
- "https://artificialanalysis.ai/models/gpt-5-5-medium",
759
- "https://artificialanalysis.ai/models/releases/gpt-5-5"
773
+ "https://artificialanalysis.ai/models/claude-opus-4-7"
760
774
  ],
761
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
775
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
762
776
  "transport_mapping": {
763
777
  "status": "unknown",
764
778
  "candidate_model_ids": [
765
- "gpt-5.5"
779
+ "claude-opus-4-7"
766
780
  ],
767
781
  "runtime_verified": false,
768
782
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
769
- }
783
+ },
784
+ "score_raw": 40.69
770
785
  },
771
786
  {
772
- "id": "codex/gpt-5-5-low",
773
- "vendor": "codex",
774
- "model": "gpt-5.5",
775
- "effort": "low",
776
- "reasoning": "reasoning",
787
+ "id": "gemini/gemini-3-8-flash-medium",
788
+ "vendor": "gemini",
789
+ "model": "gemini-3.8-flash",
790
+ "effort": "medium",
791
+ "reasoning": "unspecified",
777
792
  "fallback": null,
778
- "aa_slug": "gpt-5-5-low",
779
- "score": 35,
793
+ "aa_slug": "gemini-3-8-flash-medium",
794
+ "score": 40,
795
+ "estimated": false,
796
+ "evidence_marker": "unmarked",
797
+ "benchmark_version": "4.3.2",
798
+ "as_of": "2026-09-22",
799
+ "source_urls": [
800
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-medium"
801
+ ],
802
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
803
+ "transport_mapping": {
804
+ "status": "unknown",
805
+ "candidate_model_ids": [
806
+ "gemini-3.8-flash-medium"
807
+ ],
808
+ "runtime_verified": false,
809
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
810
+ },
811
+ "score_raw": 39.77
812
+ },
813
+ {
814
+ "id": "gemini/gemini-3-7-flash-medium",
815
+ "vendor": "gemini",
816
+ "model": "gemini-3.7-flash",
817
+ "effort": "medium",
818
+ "reasoning": "unspecified",
819
+ "fallback": null,
820
+ "aa_slug": "gemini-3-7-flash-medium",
821
+ "score": 40,
780
822
  "estimated": true,
781
823
  "evidence_marker": "estimated",
782
- "benchmark_version": "4.2",
783
- "as_of": "2026-09-07",
824
+ "benchmark_version": "4.3.2",
825
+ "as_of": "2026-09-22",
784
826
  "source_urls": [
785
- "https://artificialanalysis.ai/models/gpt-5-5-low",
786
- "https://artificialanalysis.ai/models/releases/gpt-5-5"
827
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash-medium"
787
828
  ],
788
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
829
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
789
830
  "transport_mapping": {
790
831
  "status": "unknown",
791
832
  "candidate_model_ids": [
792
- "gpt-5.5"
833
+ "gemini-3.7-flash-medium"
834
+ ],
835
+ "runtime_verified": false,
836
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
837
+ },
838
+ "score_raw": 39.62
839
+ },
840
+ {
841
+ "id": "claude/claude-opus-5-low",
842
+ "vendor": "claude",
843
+ "model": "claude-opus-5",
844
+ "effort": "low",
845
+ "reasoning": "adaptive",
846
+ "fallback": null,
847
+ "aa_slug": "claude-opus-5-low",
848
+ "score": 39,
849
+ "estimated": false,
850
+ "evidence_marker": "unmarked",
851
+ "benchmark_version": "4.3.2",
852
+ "as_of": "2026-09-22",
853
+ "source_urls": [
854
+ "https://artificialanalysis.ai/models/claude-opus-5-low"
855
+ ],
856
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
857
+ "transport_mapping": {
858
+ "status": "unknown",
859
+ "candidate_model_ids": [
860
+ "claude-opus-5"
793
861
  ],
794
862
  "runtime_verified": false,
795
863
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
796
- }
864
+ },
865
+ "score_raw": 39.35
797
866
  },
798
867
  {
799
- "id": "codex/gpt-5-5-non-reasoning",
868
+ "id": "codex/gpt-5-6-sol-medium",
800
869
  "vendor": "codex",
801
- "model": "gpt-5.5",
802
- "effort": null,
803
- "reasoning": "non-reasoning",
870
+ "model": "gpt-5.6-sol",
871
+ "effort": "medium",
872
+ "reasoning": "reasoning",
804
873
  "fallback": null,
805
- "aa_slug": "gpt-5-5-non-reasoning",
806
- "score": 27,
807
- "estimated": true,
808
- "evidence_marker": "estimated",
809
- "benchmark_version": "4.2",
810
- "as_of": "2026-09-07",
874
+ "aa_slug": "gpt-5-6-sol-medium",
875
+ "score": 39,
876
+ "estimated": false,
877
+ "evidence_marker": "unmarked",
878
+ "benchmark_version": "4.3.2",
879
+ "as_of": "2026-09-22",
811
880
  "source_urls": [
812
- "https://artificialanalysis.ai/models/gpt-5-5-non-reasoning",
813
- "https://artificialanalysis.ai/models/releases/gpt-5-5"
881
+ "https://artificialanalysis.ai/models/gpt-5-6-sol-medium",
882
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
814
883
  ],
815
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
884
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
816
885
  "transport_mapping": {
817
886
  "status": "unknown",
818
887
  "candidate_model_ids": [
819
- "gpt-5.5"
888
+ "gpt-5.6-sol"
820
889
  ],
821
890
  "runtime_verified": false,
822
891
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
823
- }
892
+ },
893
+ "score_raw": 39.24
894
+ },
895
+ {
896
+ "id": "gemini/gemini-3-7-flash",
897
+ "vendor": "gemini",
898
+ "model": "gemini-3.7-flash",
899
+ "effort": "high",
900
+ "reasoning": "unspecified",
901
+ "fallback": null,
902
+ "aa_slug": "gemini-3-7-flash",
903
+ "score": 39,
904
+ "estimated": false,
905
+ "evidence_marker": "unmarked",
906
+ "benchmark_version": "4.3.2",
907
+ "as_of": "2026-09-22",
908
+ "source_urls": [
909
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash"
910
+ ],
911
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
912
+ "transport_mapping": {
913
+ "status": "unknown",
914
+ "candidate_model_ids": [
915
+ "gemini-3.7-flash-high"
916
+ ],
917
+ "runtime_verified": false,
918
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
919
+ },
920
+ "score_raw": 39.06
824
921
  },
825
922
  {
826
923
  "id": "codex/gpt-5-4",
@@ -830,16 +927,16 @@
830
927
  "reasoning": "reasoning",
831
928
  "fallback": null,
832
929
  "aa_slug": "gpt-5-4",
833
- "score": 43,
930
+ "score": 39,
834
931
  "estimated": true,
835
932
  "evidence_marker": "estimated",
836
- "benchmark_version": "4.2",
837
- "as_of": "2026-09-07",
933
+ "benchmark_version": "4.3.2",
934
+ "as_of": "2026-09-22",
838
935
  "source_urls": [
839
936
  "https://artificialanalysis.ai/models/gpt-5-4",
840
937
  "https://artificialanalysis.ai/models/releases/gpt-5-4"
841
938
  ],
842
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
939
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
843
940
  "transport_mapping": {
844
941
  "status": "unknown",
845
942
  "candidate_model_ids": [
@@ -847,446 +944,494 @@
847
944
  ],
848
945
  "runtime_verified": false,
849
946
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
850
- }
947
+ },
948
+ "score_raw": 38.98
851
949
  },
852
950
  {
853
- "id": "codex/gpt-5-4-low",
854
- "vendor": "codex",
855
- "model": "gpt-5.4",
856
- "effort": "low",
951
+ "id": "grok/grok-4-5",
952
+ "vendor": "grok",
953
+ "model": "grok-4.5",
954
+ "effort": "high",
857
955
  "reasoning": "reasoning",
858
956
  "fallback": null,
859
- "aa_slug": "gpt-5-4-low",
860
- "score": 32,
861
- "estimated": true,
862
- "evidence_marker": "estimated",
863
- "benchmark_version": "4.2",
864
- "as_of": "2026-09-07",
957
+ "aa_slug": "grok-4-5",
958
+ "score": 39,
959
+ "estimated": false,
960
+ "evidence_marker": "unmarked",
961
+ "benchmark_version": "4.3.2",
962
+ "as_of": "2026-09-22",
865
963
  "source_urls": [
866
- "https://artificialanalysis.ai/models/gpt-5-4-low",
867
- "https://artificialanalysis.ai/models/releases/gpt-5-4"
964
+ "https://artificialanalysis.ai/models/grok-4-5"
868
965
  ],
869
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
966
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
870
967
  "transport_mapping": {
871
968
  "status": "unknown",
872
969
  "candidate_model_ids": [
873
- "gpt-5.4"
970
+ "grok-4.5"
874
971
  ],
875
972
  "runtime_verified": false,
876
973
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
877
- }
974
+ },
975
+ "score_raw": 38.81
878
976
  },
879
977
  {
880
- "id": "codex/gpt-5-4-non-reasoning",
978
+ "id": "codex/gpt-5-5",
881
979
  "vendor": "codex",
882
- "model": "gpt-5.4",
883
- "effort": null,
884
- "reasoning": "non-reasoning",
980
+ "model": "gpt-5.5",
981
+ "effort": "xhigh",
982
+ "reasoning": "reasoning",
885
983
  "fallback": null,
886
- "aa_slug": "gpt-5-4-non-reasoning",
887
- "score": 21,
888
- "estimated": true,
889
- "evidence_marker": "estimated",
890
- "benchmark_version": "4.2",
891
- "as_of": "2026-09-07",
984
+ "aa_slug": "gpt-5-5",
985
+ "score": 38,
986
+ "estimated": false,
987
+ "evidence_marker": "unmarked",
988
+ "benchmark_version": "4.3.2",
989
+ "as_of": "2026-09-22",
892
990
  "source_urls": [
893
- "https://artificialanalysis.ai/models/gpt-5-4-non-reasoning",
894
- "https://artificialanalysis.ai/models/releases/gpt-5-4"
991
+ "https://artificialanalysis.ai/models/gpt-5-5",
992
+ "https://artificialanalysis.ai/models/releases/gpt-5-5"
895
993
  ],
896
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
994
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
897
995
  "transport_mapping": {
898
996
  "status": "unknown",
899
997
  "candidate_model_ids": [
900
- "gpt-5.4"
998
+ "gpt-5.5"
901
999
  ],
902
1000
  "runtime_verified": false,
903
1001
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
904
- }
1002
+ },
1003
+ "score_raw": 38.36
905
1004
  },
906
1005
  {
907
- "id": "codex/gpt-5-4-mini",
908
- "vendor": "codex",
909
- "model": "gpt-5.4-mini",
910
- "effort": "xhigh",
911
- "reasoning": "reasoning",
1006
+ "id": "claude/claude-sonnet-5",
1007
+ "vendor": "claude",
1008
+ "model": "claude-sonnet-5",
1009
+ "effort": "max",
1010
+ "reasoning": "adaptive",
912
1011
  "fallback": null,
913
- "aa_slug": "gpt-5-4-mini",
914
- "score": 32,
915
- "estimated": true,
916
- "evidence_marker": "estimated",
917
- "benchmark_version": "4.2",
918
- "as_of": "2026-09-07",
1012
+ "aa_slug": "claude-sonnet-5",
1013
+ "score": 38,
1014
+ "estimated": false,
1015
+ "evidence_marker": "unmarked",
1016
+ "benchmark_version": "4.3.2",
1017
+ "as_of": "2026-09-22",
919
1018
  "source_urls": [
920
- "https://artificialanalysis.ai/models/gpt-5-4-mini",
921
- "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
1019
+ "https://artificialanalysis.ai/models/claude-sonnet-5"
922
1020
  ],
923
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
1021
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
924
1022
  "transport_mapping": {
925
1023
  "status": "unknown",
926
1024
  "candidate_model_ids": [
927
- "gpt-5.4-mini"
1025
+ "claude-sonnet-5"
928
1026
  ],
929
1027
  "runtime_verified": false,
930
1028
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
931
- }
1029
+ },
1030
+ "score_raw": 38.16
932
1031
  },
933
1032
  {
934
- "id": "codex/gpt-5-4-mini-medium",
1033
+ "id": "codex/gpt-5-6-terra-xhigh",
935
1034
  "vendor": "codex",
936
- "model": "gpt-5.4-mini",
937
- "effort": "medium",
1035
+ "model": "gpt-5.6-terra",
1036
+ "effort": "xhigh",
938
1037
  "reasoning": "reasoning",
939
1038
  "fallback": null,
940
- "aa_slug": "gpt-5-4-mini-medium",
941
- "score": 23,
942
- "estimated": true,
943
- "evidence_marker": "estimated",
944
- "benchmark_version": "4.2",
945
- "as_of": "2026-09-07",
1039
+ "aa_slug": "gpt-5-6-terra-xhigh",
1040
+ "score": 38,
1041
+ "estimated": false,
1042
+ "evidence_marker": "unmarked",
1043
+ "benchmark_version": "4.3.2",
1044
+ "as_of": "2026-09-22",
946
1045
  "source_urls": [
947
- "https://artificialanalysis.ai/models/gpt-5-4-mini-medium",
948
- "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
1046
+ "https://artificialanalysis.ai/models/gpt-5-6-terra-xhigh",
1047
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
949
1048
  ],
950
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
1049
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
951
1050
  "transport_mapping": {
952
1051
  "status": "unknown",
953
1052
  "candidate_model_ids": [
954
- "gpt-5.4-mini"
1053
+ "gpt-5.6-terra"
955
1054
  ],
956
1055
  "runtime_verified": false,
957
1056
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
958
- }
1057
+ },
1058
+ "score_raw": 37.95
959
1059
  },
960
1060
  {
961
- "id": "codex/gpt-5-4-mini-non-reasoning",
1061
+ "id": "codex/gpt-5-6-luna",
962
1062
  "vendor": "codex",
963
- "model": "gpt-5.4-mini",
964
- "effort": null,
965
- "reasoning": "non-reasoning",
1063
+ "model": "gpt-5.6-luna",
1064
+ "effort": "max",
1065
+ "reasoning": "reasoning",
966
1066
  "fallback": null,
967
- "aa_slug": "gpt-5-4-mini-non-reasoning",
968
- "score": 11,
969
- "estimated": true,
970
- "evidence_marker": "estimated",
971
- "benchmark_version": "4.2",
972
- "as_of": "2026-09-07",
1067
+ "aa_slug": "gpt-5-6-luna",
1068
+ "score": 37,
1069
+ "estimated": false,
1070
+ "evidence_marker": "unmarked",
1071
+ "benchmark_version": "4.3.2",
1072
+ "as_of": "2026-09-22",
973
1073
  "source_urls": [
974
- "https://artificialanalysis.ai/models/gpt-5-4-mini-non-reasoning",
975
- "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
1074
+ "https://artificialanalysis.ai/models/gpt-5-6-luna",
1075
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
976
1076
  ],
977
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
1077
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
978
1078
  "transport_mapping": {
979
1079
  "status": "unknown",
980
1080
  "candidate_model_ids": [
981
- "gpt-5.4-mini"
1081
+ "gpt-5.6-luna"
982
1082
  ],
983
1083
  "runtime_verified": false,
984
1084
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
985
- }
1085
+ },
1086
+ "score_raw": 37.32
986
1087
  },
987
1088
  {
988
- "id": "claude/claude-fable-5-1",
989
- "vendor": "claude",
990
- "model": "claude-fable-5-1",
991
- "effort": "max",
992
- "reasoning": "adaptive",
993
- "fallback": "default",
994
- "aa_slug": "claude-fable-5-1",
995
- "score": 57,
1089
+ "id": "codex/gpt-5-5-high",
1090
+ "vendor": "codex",
1091
+ "model": "gpt-5.5",
1092
+ "effort": "high",
1093
+ "reasoning": "reasoning",
1094
+ "fallback": null,
1095
+ "aa_slug": "gpt-5-5-high",
1096
+ "score": 37,
996
1097
  "estimated": false,
997
1098
  "evidence_marker": "unmarked",
998
- "benchmark_version": "4.2",
999
- "as_of": "2026-09-07",
1099
+ "benchmark_version": "4.3.2",
1100
+ "as_of": "2026-09-22",
1000
1101
  "source_urls": [
1001
- "https://artificialanalysis.ai/models/claude-fable-5-1"
1102
+ "https://artificialanalysis.ai/models/gpt-5-5-high",
1103
+ "https://artificialanalysis.ai/models/releases/gpt-5-5"
1002
1104
  ],
1003
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1105
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1004
1106
  "transport_mapping": {
1005
1107
  "status": "unknown",
1006
1108
  "candidate_model_ids": [
1007
- "claude-fable-5-1"
1109
+ "gpt-5.5"
1008
1110
  ],
1009
1111
  "runtime_verified": false,
1010
1112
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1011
- }
1113
+ },
1114
+ "score_raw": 36.98
1012
1115
  },
1013
1116
  {
1014
- "id": "claude/claude-fable-5-1-xhigh",
1015
- "vendor": "claude",
1016
- "model": "claude-fable-5-1",
1017
- "effort": "xhigh",
1018
- "reasoning": "adaptive",
1019
- "fallback": "default",
1020
- "aa_slug": "claude-fable-5-1-xhigh",
1021
- "score": 54,
1117
+ "id": "gemini/gemini-3-7-flash-low",
1118
+ "vendor": "gemini",
1119
+ "model": "gemini-3.7-flash",
1120
+ "effort": "low",
1121
+ "reasoning": "unspecified",
1122
+ "fallback": null,
1123
+ "aa_slug": "gemini-3-7-flash-low",
1124
+ "score": 37,
1022
1125
  "estimated": true,
1023
1126
  "evidence_marker": "estimated",
1024
- "benchmark_version": "4.2",
1025
- "as_of": "2026-09-07",
1127
+ "benchmark_version": "4.3.2",
1128
+ "as_of": "2026-09-22",
1026
1129
  "source_urls": [
1027
- "https://artificialanalysis.ai/models/claude-fable-5-1-xhigh"
1130
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash-low"
1028
1131
  ],
1029
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1132
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
1030
1133
  "transport_mapping": {
1031
1134
  "status": "unknown",
1032
1135
  "candidate_model_ids": [
1033
- "claude-fable-5-1"
1136
+ "gemini-3.7-flash-low"
1034
1137
  ],
1035
1138
  "runtime_verified": false,
1036
- "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1037
- }
1139
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1140
+ },
1141
+ "score_raw": 36.95
1038
1142
  },
1039
1143
  {
1040
- "id": "claude/claude-fable-5-1-high",
1041
- "vendor": "claude",
1042
- "model": "claude-fable-5-1",
1043
- "effort": "high",
1044
- "reasoning": "adaptive",
1045
- "fallback": "default",
1046
- "aa_slug": "claude-fable-5-1-high",
1047
- "score": 52,
1048
- "estimated": true,
1049
- "evidence_marker": "estimated",
1050
- "benchmark_version": "4.2",
1051
- "as_of": "2026-09-07",
1144
+ "id": "grok/grok-4-6-low",
1145
+ "vendor": "grok",
1146
+ "model": "grok-4.6",
1147
+ "effort": "low",
1148
+ "reasoning": "reasoning",
1149
+ "fallback": null,
1150
+ "aa_slug": "grok-4-6-low",
1151
+ "score": 35,
1152
+ "estimated": false,
1153
+ "evidence_marker": "unmarked",
1154
+ "benchmark_version": "4.3.2",
1155
+ "as_of": "2026-09-22",
1052
1156
  "source_urls": [
1053
- "https://artificialanalysis.ai/models/claude-fable-5-1-high"
1157
+ "https://artificialanalysis.ai/models/grok-4-6-low"
1054
1158
  ],
1055
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1159
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
1056
1160
  "transport_mapping": {
1057
1161
  "status": "unknown",
1058
1162
  "candidate_model_ids": [
1059
- "claude-fable-5-1"
1163
+ "grok-4.6"
1060
1164
  ],
1061
1165
  "runtime_verified": false,
1062
1166
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1063
- }
1167
+ },
1168
+ "score_raw": 35.12
1064
1169
  },
1065
1170
  {
1066
- "id": "claude/claude-fable-5-1-medium",
1067
- "vendor": "claude",
1068
- "model": "claude-fable-5-1",
1069
- "effort": "medium",
1070
- "reasoning": "adaptive",
1071
- "fallback": "default",
1072
- "aa_slug": "claude-fable-5-1-medium",
1073
- "score": 50,
1074
- "estimated": true,
1075
- "evidence_marker": "estimated",
1076
- "benchmark_version": "4.2",
1077
- "as_of": "2026-09-07",
1171
+ "id": "codex/gpt-5-6-luna-xhigh",
1172
+ "vendor": "codex",
1173
+ "model": "gpt-5.6-luna",
1174
+ "effort": "xhigh",
1175
+ "reasoning": "reasoning",
1176
+ "fallback": null,
1177
+ "aa_slug": "gpt-5-6-luna-xhigh",
1178
+ "score": 35,
1179
+ "estimated": false,
1180
+ "evidence_marker": "unmarked",
1181
+ "benchmark_version": "4.3.2",
1182
+ "as_of": "2026-09-22",
1078
1183
  "source_urls": [
1079
- "https://artificialanalysis.ai/models/claude-fable-5-1-medium"
1184
+ "https://artificialanalysis.ai/models/gpt-5-6-luna-xhigh",
1185
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
1080
1186
  ],
1081
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1187
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1082
1188
  "transport_mapping": {
1083
1189
  "status": "unknown",
1084
1190
  "candidate_model_ids": [
1085
- "claude-fable-5-1"
1191
+ "gpt-5.6-luna"
1086
1192
  ],
1087
1193
  "runtime_verified": false,
1088
1194
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1089
- }
1195
+ },
1196
+ "score_raw": 34.56
1090
1197
  },
1091
1198
  {
1092
- "id": "claude/claude-fable-5-1-low",
1199
+ "id": "claude/claude-sonnet-5-xhigh",
1093
1200
  "vendor": "claude",
1094
- "model": "claude-fable-5-1",
1095
- "effort": "low",
1201
+ "model": "claude-sonnet-5",
1202
+ "effort": "xhigh",
1096
1203
  "reasoning": "adaptive",
1097
- "fallback": "default",
1098
- "aa_slug": "claude-fable-5-1-low",
1099
- "score": 48,
1100
- "estimated": true,
1101
- "evidence_marker": "estimated",
1102
- "benchmark_version": "4.2",
1103
- "as_of": "2026-09-07",
1204
+ "fallback": null,
1205
+ "aa_slug": "claude-sonnet-5-xhigh",
1206
+ "score": 34,
1207
+ "estimated": false,
1208
+ "evidence_marker": "unmarked",
1209
+ "benchmark_version": "4.3.2",
1210
+ "as_of": "2026-09-22",
1104
1211
  "source_urls": [
1105
- "https://artificialanalysis.ai/models/claude-fable-5-1-low"
1212
+ "https://artificialanalysis.ai/models/claude-sonnet-5-xhigh"
1106
1213
  ],
1107
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1214
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1108
1215
  "transport_mapping": {
1109
1216
  "status": "unknown",
1110
1217
  "candidate_model_ids": [
1111
- "claude-fable-5-1"
1218
+ "claude-sonnet-5"
1112
1219
  ],
1113
1220
  "runtime_verified": false,
1114
1221
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1115
- }
1222
+ },
1223
+ "score_raw": 34.38
1116
1224
  },
1117
1225
  {
1118
- "id": "claude/claude-fable-5",
1119
- "vendor": "claude",
1120
- "model": "claude-fable-5",
1121
- "effort": "max",
1122
- "reasoning": "adaptive",
1123
- "fallback": "claude-opus-4-8",
1124
- "aa_slug": "claude-fable-5",
1125
- "score": 53,
1226
+ "id": "codex/gpt-5-6-terra-high",
1227
+ "vendor": "codex",
1228
+ "model": "gpt-5.6-terra",
1229
+ "effort": "high",
1230
+ "reasoning": "reasoning",
1231
+ "fallback": null,
1232
+ "aa_slug": "gpt-5-6-terra-high",
1233
+ "score": 34,
1126
1234
  "estimated": false,
1127
1235
  "evidence_marker": "unmarked",
1128
- "benchmark_version": "4.2",
1129
- "as_of": "2026-09-07",
1236
+ "benchmark_version": "4.3.2",
1237
+ "as_of": "2026-09-22",
1130
1238
  "source_urls": [
1131
- "https://artificialanalysis.ai/models/claude-fable-5"
1239
+ "https://artificialanalysis.ai/models/gpt-5-6-terra-high",
1240
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
1132
1241
  ],
1133
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1242
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1134
1243
  "transport_mapping": {
1135
1244
  "status": "unknown",
1136
1245
  "candidate_model_ids": [
1137
- "claude-fable-5"
1246
+ "gpt-5.6-terra"
1138
1247
  ],
1139
1248
  "runtime_verified": false,
1140
1249
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1141
- }
1250
+ },
1251
+ "score_raw": 34.24
1142
1252
  },
1143
1253
  {
1144
- "id": "claude/claude-opus-5",
1145
- "vendor": "claude",
1146
- "model": "claude-opus-5",
1147
- "effort": "max",
1148
- "reasoning": "adaptive",
1254
+ "id": "gemini/gemini-3-6-flash",
1255
+ "vendor": "gemini",
1256
+ "model": "gemini-3.6-flash",
1257
+ "effort": "high",
1258
+ "reasoning": "unspecified",
1149
1259
  "fallback": null,
1150
- "aa_slug": "claude-opus-5",
1151
- "score": 54,
1260
+ "aa_slug": "gemini-3-6-flash",
1261
+ "score": 34,
1152
1262
  "estimated": false,
1153
1263
  "evidence_marker": "unmarked",
1154
- "benchmark_version": "4.2",
1155
- "as_of": "2026-09-07",
1264
+ "benchmark_version": "4.3.2",
1265
+ "as_of": "2026-09-22",
1156
1266
  "source_urls": [
1157
- "https://artificialanalysis.ai/models/claude-opus-5"
1267
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-6-flash"
1158
1268
  ],
1159
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1269
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
1160
1270
  "transport_mapping": {
1161
1271
  "status": "unknown",
1162
1272
  "candidate_model_ids": [
1163
- "claude-opus-5"
1273
+ "gemini-3.6-flash-high"
1164
1274
  ],
1165
1275
  "runtime_verified": false,
1166
- "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1167
- }
1276
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1277
+ },
1278
+ "score_raw": 33.98
1168
1279
  },
1169
1280
  {
1170
- "id": "claude/claude-opus-5-xhigh",
1171
- "vendor": "claude",
1172
- "model": "claude-opus-5",
1173
- "effort": "xhigh",
1174
- "reasoning": "adaptive",
1281
+ "id": "codex/gpt-5-5-medium",
1282
+ "vendor": "codex",
1283
+ "model": "gpt-5.5",
1284
+ "effort": "medium",
1285
+ "reasoning": "reasoning",
1175
1286
  "fallback": null,
1176
- "aa_slug": "claude-opus-5-xhigh",
1177
- "score": 53,
1287
+ "aa_slug": "gpt-5-5-medium",
1288
+ "score": 34,
1178
1289
  "estimated": false,
1179
1290
  "evidence_marker": "unmarked",
1180
- "benchmark_version": "4.2",
1181
- "as_of": "2026-09-07",
1291
+ "benchmark_version": "4.3.2",
1292
+ "as_of": "2026-09-22",
1182
1293
  "source_urls": [
1183
- "https://artificialanalysis.ai/models/claude-opus-5-xhigh"
1294
+ "https://artificialanalysis.ai/models/gpt-5-5-medium",
1295
+ "https://artificialanalysis.ai/models/releases/gpt-5-5"
1184
1296
  ],
1185
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1297
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1186
1298
  "transport_mapping": {
1187
1299
  "status": "unknown",
1188
1300
  "candidate_model_ids": [
1189
- "claude-opus-5"
1301
+ "gpt-5.5"
1190
1302
  ],
1191
1303
  "runtime_verified": false,
1192
1304
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1193
- }
1305
+ },
1306
+ "score_raw": 33.8
1194
1307
  },
1195
1308
  {
1196
- "id": "claude/claude-opus-5-high",
1197
- "vendor": "claude",
1198
- "model": "claude-opus-5",
1199
- "effort": "high",
1200
- "reasoning": "adaptive",
1309
+ "id": "codex/gpt-5-6-sol-low",
1310
+ "vendor": "codex",
1311
+ "model": "gpt-5.6-sol",
1312
+ "effort": "low",
1313
+ "reasoning": "reasoning",
1201
1314
  "fallback": null,
1202
- "aa_slug": "claude-opus-5-high",
1203
- "score": 52,
1315
+ "aa_slug": "gpt-5-6-sol-low",
1316
+ "score": 33,
1204
1317
  "estimated": false,
1205
1318
  "evidence_marker": "unmarked",
1206
- "benchmark_version": "4.2",
1207
- "as_of": "2026-09-07",
1319
+ "benchmark_version": "4.3.2",
1320
+ "as_of": "2026-09-22",
1208
1321
  "source_urls": [
1209
- "https://artificialanalysis.ai/models/claude-opus-5-high"
1322
+ "https://artificialanalysis.ai/models/gpt-5-6-sol-low",
1323
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
1210
1324
  ],
1211
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1325
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1212
1326
  "transport_mapping": {
1213
1327
  "status": "unknown",
1214
1328
  "candidate_model_ids": [
1215
- "claude-opus-5"
1329
+ "gpt-5.6-sol"
1216
1330
  ],
1217
1331
  "runtime_verified": false,
1218
1332
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1219
- }
1333
+ },
1334
+ "score_raw": 33.47
1220
1335
  },
1221
1336
  {
1222
- "id": "claude/claude-opus-5-medium",
1223
- "vendor": "claude",
1224
- "model": "claude-opus-5",
1225
- "effort": "medium",
1226
- "reasoning": "adaptive",
1337
+ "id": "gemini/gemini-3-8-flash-low",
1338
+ "vendor": "gemini",
1339
+ "model": "gemini-3.8-flash",
1340
+ "effort": "low",
1341
+ "reasoning": "unspecified",
1227
1342
  "fallback": null,
1228
- "aa_slug": "claude-opus-5-medium",
1229
- "score": 50,
1343
+ "aa_slug": "gemini-3-8-flash-low",
1344
+ "score": 33,
1230
1345
  "estimated": false,
1231
1346
  "evidence_marker": "unmarked",
1232
- "benchmark_version": "4.2",
1233
- "as_of": "2026-09-07",
1347
+ "benchmark_version": "4.3.2",
1348
+ "as_of": "2026-09-22",
1234
1349
  "source_urls": [
1235
- "https://artificialanalysis.ai/models/claude-opus-5-medium"
1350
+ "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-low"
1236
1351
  ],
1237
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1352
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
1238
1353
  "transport_mapping": {
1239
1354
  "status": "unknown",
1240
1355
  "candidate_model_ids": [
1241
- "claude-opus-5"
1356
+ "gemini-3.8-flash-low"
1357
+ ],
1358
+ "runtime_verified": false,
1359
+ "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1360
+ },
1361
+ "score_raw": 33.45
1362
+ },
1363
+ {
1364
+ "id": "codex/gpt-5-6-luna-high",
1365
+ "vendor": "codex",
1366
+ "model": "gpt-5.6-luna",
1367
+ "effort": "high",
1368
+ "reasoning": "reasoning",
1369
+ "fallback": null,
1370
+ "aa_slug": "gpt-5-6-luna-high",
1371
+ "score": 32,
1372
+ "estimated": false,
1373
+ "evidence_marker": "unmarked",
1374
+ "benchmark_version": "4.3.2",
1375
+ "as_of": "2026-09-22",
1376
+ "source_urls": [
1377
+ "https://artificialanalysis.ai/models/gpt-5-6-luna-high",
1378
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
1379
+ ],
1380
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1381
+ "transport_mapping": {
1382
+ "status": "unknown",
1383
+ "candidate_model_ids": [
1384
+ "gpt-5.6-luna"
1242
1385
  ],
1243
1386
  "runtime_verified": false,
1244
1387
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1245
- }
1388
+ },
1389
+ "score_raw": 32.12
1246
1390
  },
1247
1391
  {
1248
- "id": "claude/claude-opus-5-low",
1249
- "vendor": "claude",
1250
- "model": "claude-opus-5",
1251
- "effort": "low",
1392
+ "id": "claude/claude-opus-4-6-adaptive",
1393
+ "vendor": "claude",
1394
+ "model": "claude-opus-4-6",
1395
+ "effort": "max",
1252
1396
  "reasoning": "adaptive",
1253
1397
  "fallback": null,
1254
- "aa_slug": "claude-opus-5-low",
1255
- "score": 44,
1256
- "estimated": false,
1257
- "evidence_marker": "unmarked",
1258
- "benchmark_version": "4.2",
1259
- "as_of": "2026-09-07",
1398
+ "aa_slug": "claude-opus-4-6-adaptive",
1399
+ "score": 32,
1400
+ "estimated": true,
1401
+ "evidence_marker": "estimated",
1402
+ "benchmark_version": "4.3.2",
1403
+ "as_of": "2026-09-22",
1260
1404
  "source_urls": [
1261
- "https://artificialanalysis.ai/models/claude-opus-5-low"
1405
+ "https://artificialanalysis.ai/models/claude-opus-4-6-adaptive"
1262
1406
  ],
1263
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1407
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1264
1408
  "transport_mapping": {
1265
1409
  "status": "unknown",
1266
1410
  "candidate_model_ids": [
1267
- "claude-opus-5"
1411
+ "claude-opus-4-6"
1268
1412
  ],
1269
1413
  "runtime_verified": false,
1270
1414
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1271
- }
1415
+ },
1416
+ "score_raw": 31.95
1272
1417
  },
1273
1418
  {
1274
- "id": "claude/claude-sonnet-5",
1419
+ "id": "claude/claude-sonnet-5-high",
1275
1420
  "vendor": "claude",
1276
1421
  "model": "claude-sonnet-5",
1277
- "effort": "max",
1422
+ "effort": "high",
1278
1423
  "reasoning": "adaptive",
1279
1424
  "fallback": null,
1280
- "aa_slug": "claude-sonnet-5",
1281
- "score": 45,
1425
+ "aa_slug": "claude-sonnet-5-high",
1426
+ "score": 32,
1282
1427
  "estimated": false,
1283
1428
  "evidence_marker": "unmarked",
1284
- "benchmark_version": "4.2",
1285
- "as_of": "2026-09-07",
1429
+ "benchmark_version": "4.3.2",
1430
+ "as_of": "2026-09-22",
1286
1431
  "source_urls": [
1287
- "https://artificialanalysis.ai/models/claude-sonnet-5"
1432
+ "https://artificialanalysis.ai/models/claude-sonnet-5-high"
1288
1433
  ],
1289
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1434
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1290
1435
  "transport_mapping": {
1291
1436
  "status": "unknown",
1292
1437
  "candidate_model_ids": [
@@ -1294,779 +1439,849 @@
1294
1439
  ],
1295
1440
  "runtime_verified": false,
1296
1441
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1297
- }
1442
+ },
1443
+ "score_raw": 31.66
1298
1444
  },
1299
1445
  {
1300
- "id": "claude/claude-sonnet-5-non-reasoning",
1446
+ "id": "claude/claude-opus-4-7-non-reasoning",
1301
1447
  "vendor": "claude",
1302
- "model": "claude-sonnet-5",
1448
+ "model": "claude-opus-4-7",
1303
1449
  "effort": "high",
1304
1450
  "reasoning": "non-reasoning",
1305
1451
  "fallback": null,
1306
- "aa_slug": "claude-sonnet-5-non-reasoning",
1307
- "score": 33,
1452
+ "aa_slug": "claude-opus-4-7-non-reasoning",
1453
+ "score": 31,
1308
1454
  "estimated": true,
1309
1455
  "evidence_marker": "estimated",
1310
- "benchmark_version": "4.2",
1311
- "as_of": "2026-09-07",
1456
+ "benchmark_version": "4.3.2",
1457
+ "as_of": "2026-09-22",
1312
1458
  "source_urls": [
1313
- "https://artificialanalysis.ai/models/claude-sonnet-5-non-reasoning"
1459
+ "https://artificialanalysis.ai/models/claude-opus-4-7-non-reasoning"
1314
1460
  ],
1315
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1461
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1316
1462
  "transport_mapping": {
1317
1463
  "status": "unknown",
1318
1464
  "candidate_model_ids": [
1319
- "claude-sonnet-5"
1465
+ "claude-opus-4-7"
1320
1466
  ],
1321
1467
  "runtime_verified": false,
1322
1468
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1323
- }
1469
+ },
1470
+ "score_raw": 30.93
1324
1471
  },
1325
1472
  {
1326
- "id": "claude/claude-opus-4-8",
1327
- "vendor": "claude",
1328
- "model": "claude-opus-4-8",
1329
- "effort": "max",
1330
- "reasoning": "adaptive",
1473
+ "id": "codex/gpt-5-5-low",
1474
+ "vendor": "codex",
1475
+ "model": "gpt-5.5",
1476
+ "effort": "low",
1477
+ "reasoning": "reasoning",
1331
1478
  "fallback": null,
1332
- "aa_slug": "claude-opus-4-8",
1333
- "score": 46,
1479
+ "aa_slug": "gpt-5-5-low",
1480
+ "score": 31,
1334
1481
  "estimated": true,
1335
1482
  "evidence_marker": "estimated",
1336
- "benchmark_version": "4.2",
1337
- "as_of": "2026-09-07",
1483
+ "benchmark_version": "4.3.2",
1484
+ "as_of": "2026-09-22",
1338
1485
  "source_urls": [
1339
- "https://artificialanalysis.ai/models/claude-opus-4-8"
1486
+ "https://artificialanalysis.ai/models/gpt-5-5-low",
1487
+ "https://artificialanalysis.ai/models/releases/gpt-5-5"
1340
1488
  ],
1341
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1489
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1342
1490
  "transport_mapping": {
1343
1491
  "status": "unknown",
1344
1492
  "candidate_model_ids": [
1345
- "claude-opus-4-8"
1493
+ "gpt-5.5"
1346
1494
  ],
1347
1495
  "runtime_verified": false,
1348
1496
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1349
- }
1497
+ },
1498
+ "score_raw": 30.71
1350
1499
  },
1351
1500
  {
1352
- "id": "claude/claude-opus-4-7",
1353
- "vendor": "claude",
1354
- "model": "claude-opus-4-7",
1355
- "effort": "max",
1356
- "reasoning": "adaptive",
1501
+ "id": "codex/gpt-5-6-terra-medium",
1502
+ "vendor": "codex",
1503
+ "model": "gpt-5.6-terra",
1504
+ "effort": "medium",
1505
+ "reasoning": "reasoning",
1357
1506
  "fallback": null,
1358
- "aa_slug": "claude-opus-4-7",
1359
- "score": 44,
1360
- "estimated": true,
1361
- "evidence_marker": "estimated",
1362
- "benchmark_version": "4.2",
1363
- "as_of": "2026-09-07",
1507
+ "aa_slug": "gpt-5-6-terra-medium",
1508
+ "score": 30,
1509
+ "estimated": false,
1510
+ "evidence_marker": "unmarked",
1511
+ "benchmark_version": "4.3.2",
1512
+ "as_of": "2026-09-22",
1364
1513
  "source_urls": [
1365
- "https://artificialanalysis.ai/models/claude-opus-4-7"
1514
+ "https://artificialanalysis.ai/models/gpt-5-6-terra-medium",
1515
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
1366
1516
  ],
1367
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1517
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1368
1518
  "transport_mapping": {
1369
1519
  "status": "unknown",
1370
1520
  "candidate_model_ids": [
1371
- "claude-opus-4-7"
1521
+ "gpt-5.6-terra"
1372
1522
  ],
1373
1523
  "runtime_verified": false,
1374
1524
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1375
- }
1525
+ },
1526
+ "score_raw": 30.09
1376
1527
  },
1377
1528
  {
1378
- "id": "claude/claude-opus-4-7-non-reasoning",
1529
+ "id": "claude/claude-sonnet-4-6-adaptive",
1379
1530
  "vendor": "claude",
1380
- "model": "claude-opus-4-7",
1381
- "effort": "high",
1382
- "reasoning": "non-reasoning",
1531
+ "model": "claude-sonnet-4-6",
1532
+ "effort": "max",
1533
+ "reasoning": "adaptive",
1383
1534
  "fallback": null,
1384
- "aa_slug": "claude-opus-4-7-non-reasoning",
1385
- "score": 35,
1386
- "estimated": true,
1387
- "evidence_marker": "estimated",
1388
- "benchmark_version": "4.2",
1389
- "as_of": "2026-09-07",
1535
+ "aa_slug": "claude-sonnet-4-6-adaptive",
1536
+ "score": 30,
1537
+ "estimated": false,
1538
+ "evidence_marker": "unmarked",
1539
+ "benchmark_version": "4.3.2",
1540
+ "as_of": "2026-09-22",
1390
1541
  "source_urls": [
1391
- "https://artificialanalysis.ai/models/claude-opus-4-7-non-reasoning"
1542
+ "https://artificialanalysis.ai/models/claude-sonnet-4-6-adaptive"
1392
1543
  ],
1393
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1544
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1394
1545
  "transport_mapping": {
1395
1546
  "status": "unknown",
1396
1547
  "candidate_model_ids": [
1397
- "claude-opus-4-7"
1548
+ "claude-sonnet-4-6"
1398
1549
  ],
1399
1550
  "runtime_verified": false,
1400
1551
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1401
- }
1552
+ },
1553
+ "score_raw": 30.06
1402
1554
  },
1403
1555
  {
1404
- "id": "claude/claude-opus-4-6-adaptive",
1556
+ "id": "claude/claude-opus-4-5-thinking",
1405
1557
  "vendor": "claude",
1406
- "model": "claude-opus-4-6",
1407
- "effort": "max",
1408
- "reasoning": "adaptive",
1558
+ "model": "claude-opus-4-5",
1559
+ "effort": null,
1560
+ "reasoning": "reasoning",
1409
1561
  "fallback": null,
1410
- "aa_slug": "claude-opus-4-6-adaptive",
1411
- "score": 36,
1562
+ "aa_slug": "claude-opus-4-5-thinking",
1563
+ "score": 29,
1412
1564
  "estimated": true,
1413
1565
  "evidence_marker": "estimated",
1414
- "benchmark_version": "4.2",
1415
- "as_of": "2026-09-07",
1566
+ "benchmark_version": "4.3.2",
1567
+ "as_of": "2026-09-22",
1416
1568
  "source_urls": [
1417
- "https://artificialanalysis.ai/models/claude-opus-4-6-adaptive"
1569
+ "https://artificialanalysis.ai/models/claude-opus-4-5-thinking"
1418
1570
  ],
1419
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1571
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1420
1572
  "transport_mapping": {
1421
1573
  "status": "unknown",
1422
1574
  "candidate_model_ids": [
1423
- "claude-opus-4-6"
1575
+ "claude-opus-4-5"
1424
1576
  ],
1425
1577
  "runtime_verified": false,
1426
1578
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1427
- }
1579
+ },
1580
+ "score_raw": 29.1
1428
1581
  },
1429
1582
  {
1430
- "id": "claude/claude-opus-4-6",
1431
- "vendor": "claude",
1432
- "model": "claude-opus-4-6",
1433
- "effort": "high",
1583
+ "id": "codex/gpt-5-6-sol-non-reasoning",
1584
+ "vendor": "codex",
1585
+ "model": "gpt-5.6-sol",
1586
+ "effort": null,
1434
1587
  "reasoning": "non-reasoning",
1435
1588
  "fallback": null,
1436
- "aa_slug": "claude-opus-4-6",
1437
- "score": 31,
1589
+ "aa_slug": "gpt-5-6-sol-non-reasoning",
1590
+ "score": 28,
1438
1591
  "estimated": true,
1439
1592
  "evidence_marker": "estimated",
1440
- "benchmark_version": "4.2",
1441
- "as_of": "2026-09-07",
1593
+ "benchmark_version": "4.3.2",
1594
+ "as_of": "2026-09-22",
1442
1595
  "source_urls": [
1443
- "https://artificialanalysis.ai/models/claude-opus-4-6"
1596
+ "https://artificialanalysis.ai/models/gpt-5-6-sol-non-reasoning",
1597
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-sol"
1444
1598
  ],
1445
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1599
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1446
1600
  "transport_mapping": {
1447
1601
  "status": "unknown",
1448
1602
  "candidate_model_ids": [
1449
- "claude-opus-4-6"
1603
+ "gpt-5.6-sol"
1450
1604
  ],
1451
1605
  "runtime_verified": false,
1452
1606
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1453
- }
1607
+ },
1608
+ "score_raw": 28.33
1454
1609
  },
1455
1610
  {
1456
- "id": "claude/claude-opus-4-5-thinking",
1611
+ "id": "claude/claude-sonnet-5-medium",
1457
1612
  "vendor": "claude",
1458
- "model": "claude-opus-4-5",
1459
- "effort": null,
1460
- "reasoning": "reasoning",
1613
+ "model": "claude-sonnet-5",
1614
+ "effort": "medium",
1615
+ "reasoning": "adaptive",
1461
1616
  "fallback": null,
1462
- "aa_slug": "claude-opus-4-5-thinking",
1463
- "score": 34,
1464
- "estimated": true,
1465
- "evidence_marker": "estimated",
1466
- "benchmark_version": "4.2",
1467
- "as_of": "2026-09-07",
1617
+ "aa_slug": "claude-sonnet-5-medium",
1618
+ "score": 28,
1619
+ "estimated": false,
1620
+ "evidence_marker": "unmarked",
1621
+ "benchmark_version": "4.3.2",
1622
+ "as_of": "2026-09-22",
1468
1623
  "source_urls": [
1469
- "https://artificialanalysis.ai/models/claude-opus-4-5-thinking"
1624
+ "https://artificialanalysis.ai/models/claude-sonnet-5-medium"
1470
1625
  ],
1471
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1626
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1472
1627
  "transport_mapping": {
1473
1628
  "status": "unknown",
1474
1629
  "candidate_model_ids": [
1475
- "claude-opus-4-5"
1630
+ "claude-sonnet-5"
1476
1631
  ],
1477
1632
  "runtime_verified": false,
1478
1633
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1479
- }
1634
+ },
1635
+ "score_raw": 28.05
1480
1636
  },
1481
1637
  {
1482
- "id": "claude/claude-opus-4-5",
1483
- "vendor": "claude",
1484
- "model": "claude-opus-4-5",
1485
- "effort": null,
1486
- "reasoning": "non-reasoning",
1638
+ "id": "codex/gpt-5-4-low",
1639
+ "vendor": "codex",
1640
+ "model": "gpt-5.4",
1641
+ "effort": "low",
1642
+ "reasoning": "reasoning",
1487
1643
  "fallback": null,
1488
- "aa_slug": "claude-opus-4-5",
1644
+ "aa_slug": "gpt-5-4-low",
1489
1645
  "score": 28,
1490
1646
  "estimated": true,
1491
1647
  "evidence_marker": "estimated",
1492
- "benchmark_version": "4.2",
1493
- "as_of": "2026-09-07",
1648
+ "benchmark_version": "4.3.2",
1649
+ "as_of": "2026-09-22",
1494
1650
  "source_urls": [
1495
- "https://artificialanalysis.ai/models/claude-opus-4-5"
1651
+ "https://artificialanalysis.ai/models/gpt-5-4-low",
1652
+ "https://artificialanalysis.ai/models/releases/gpt-5-4"
1496
1653
  ],
1497
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1654
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1498
1655
  "transport_mapping": {
1499
1656
  "status": "unknown",
1500
1657
  "candidate_model_ids": [
1501
- "claude-opus-4-5"
1658
+ "gpt-5.4"
1502
1659
  ],
1503
1660
  "runtime_verified": false,
1504
1661
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1505
- }
1662
+ },
1663
+ "score_raw": 27.58
1506
1664
  },
1507
1665
  {
1508
- "id": "claude/claude-sonnet-4-6-adaptive",
1509
- "vendor": "claude",
1510
- "model": "claude-sonnet-4-6",
1511
- "effort": "max",
1512
- "reasoning": "adaptive",
1666
+ "id": "codex/gpt-5-6-terra-low",
1667
+ "vendor": "codex",
1668
+ "model": "gpt-5.6-terra",
1669
+ "effort": "low",
1670
+ "reasoning": "reasoning",
1513
1671
  "fallback": null,
1514
- "aa_slug": "claude-sonnet-4-6-adaptive",
1515
- "score": 38,
1516
- "estimated": true,
1517
- "evidence_marker": "estimated",
1518
- "benchmark_version": "4.2",
1519
- "as_of": "2026-09-07",
1672
+ "aa_slug": "gpt-5-6-terra-low",
1673
+ "score": 27,
1674
+ "estimated": false,
1675
+ "evidence_marker": "unmarked",
1676
+ "benchmark_version": "4.3.2",
1677
+ "as_of": "2026-09-22",
1520
1678
  "source_urls": [
1521
- "https://artificialanalysis.ai/models/claude-sonnet-4-6-adaptive"
1679
+ "https://artificialanalysis.ai/models/gpt-5-6-terra-low",
1680
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
1522
1681
  ],
1523
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1682
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1524
1683
  "transport_mapping": {
1525
1684
  "status": "unknown",
1526
1685
  "candidate_model_ids": [
1527
- "claude-sonnet-4-6"
1686
+ "gpt-5.6-terra"
1528
1687
  ],
1529
1688
  "runtime_verified": false,
1530
1689
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1531
- }
1690
+ },
1691
+ "score_raw": 27.5
1532
1692
  },
1533
1693
  {
1534
- "id": "claude/claude-sonnet-4-6",
1694
+ "id": "claude/claude-opus-4-6",
1535
1695
  "vendor": "claude",
1536
- "model": "claude-sonnet-4-6",
1696
+ "model": "claude-opus-4-6",
1537
1697
  "effort": "high",
1538
1698
  "reasoning": "non-reasoning",
1539
1699
  "fallback": null,
1540
- "aa_slug": "claude-sonnet-4-6",
1541
- "score": 29,
1700
+ "aa_slug": "claude-opus-4-6",
1701
+ "score": 26,
1542
1702
  "estimated": true,
1543
1703
  "evidence_marker": "estimated",
1544
- "benchmark_version": "4.2",
1545
- "as_of": "2026-09-07",
1704
+ "benchmark_version": "4.3.2",
1705
+ "as_of": "2026-09-22",
1546
1706
  "source_urls": [
1547
- "https://artificialanalysis.ai/models/claude-sonnet-4-6"
1707
+ "https://artificialanalysis.ai/models/claude-opus-4-6"
1548
1708
  ],
1549
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1709
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1550
1710
  "transport_mapping": {
1551
1711
  "status": "unknown",
1552
1712
  "candidate_model_ids": [
1553
- "claude-sonnet-4-6"
1713
+ "claude-opus-4-6"
1554
1714
  ],
1555
1715
  "runtime_verified": false,
1556
1716
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1557
- }
1717
+ },
1718
+ "score_raw": 26.35
1558
1719
  },
1559
1720
  {
1560
- "id": "claude/claude-sonnet-4-6-non-reasoning-low-effort",
1561
- "vendor": "claude",
1562
- "model": "claude-sonnet-4-6",
1563
- "effort": "low",
1564
- "reasoning": "non-reasoning",
1721
+ "id": "codex/gpt-5-6-luna-medium",
1722
+ "vendor": "codex",
1723
+ "model": "gpt-5.6-luna",
1724
+ "effort": "medium",
1725
+ "reasoning": "reasoning",
1565
1726
  "fallback": null,
1566
- "aa_slug": "claude-sonnet-4-6-non-reasoning-low-effort",
1567
- "score": 27,
1568
- "estimated": true,
1569
- "evidence_marker": "estimated",
1570
- "benchmark_version": "4.2",
1571
- "as_of": "2026-09-07",
1727
+ "aa_slug": "gpt-5-6-luna-medium",
1728
+ "score": 25,
1729
+ "estimated": false,
1730
+ "evidence_marker": "unmarked",
1731
+ "benchmark_version": "4.3.2",
1732
+ "as_of": "2026-09-22",
1733
+ "source_urls": [
1734
+ "https://artificialanalysis.ai/models/gpt-5-6-luna-medium",
1735
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
1736
+ ],
1737
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1738
+ "transport_mapping": {
1739
+ "status": "unknown",
1740
+ "candidate_model_ids": [
1741
+ "gpt-5.6-luna"
1742
+ ],
1743
+ "runtime_verified": false,
1744
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1745
+ },
1746
+ "score_raw": 25.04
1747
+ },
1748
+ {
1749
+ "id": "grok/grok-4-3",
1750
+ "vendor": "grok",
1751
+ "model": "grok-4.3",
1752
+ "effort": "high",
1753
+ "reasoning": "reasoning",
1754
+ "fallback": null,
1755
+ "aa_slug": "grok-4-3",
1756
+ "score": 25,
1757
+ "estimated": false,
1758
+ "evidence_marker": "unmarked",
1759
+ "benchmark_version": "4.3.2",
1760
+ "as_of": "2026-09-22",
1572
1761
  "source_urls": [
1573
- "https://artificialanalysis.ai/models/claude-sonnet-4-6-non-reasoning-low-effort"
1762
+ "https://artificialanalysis.ai/models/grok-4-3"
1574
1763
  ],
1575
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1764
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
1576
1765
  "transport_mapping": {
1577
1766
  "status": "unknown",
1578
1767
  "candidate_model_ids": [
1579
- "claude-sonnet-4-6"
1768
+ "grok-4.3"
1580
1769
  ],
1581
1770
  "runtime_verified": false,
1582
1771
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1583
- }
1772
+ },
1773
+ "score_raw": 24.88
1584
1774
  },
1585
1775
  {
1586
- "id": "claude/claude-4-5-sonnet-thinking",
1587
- "vendor": "claude",
1588
- "model": "claude-sonnet-4-5",
1589
- "effort": null,
1776
+ "id": "grok/grok-4-3-medium",
1777
+ "vendor": "grok",
1778
+ "model": "grok-4.3",
1779
+ "effort": "medium",
1590
1780
  "reasoning": "reasoning",
1591
1781
  "fallback": null,
1592
- "aa_slug": "claude-4-5-sonnet-thinking",
1593
- "score": 29,
1782
+ "aa_slug": "grok-4-3-medium",
1783
+ "score": 25,
1594
1784
  "estimated": true,
1595
1785
  "evidence_marker": "estimated",
1596
- "benchmark_version": "4.2",
1597
- "as_of": "2026-09-07",
1786
+ "benchmark_version": "4.3.2",
1787
+ "as_of": "2026-09-22",
1598
1788
  "source_urls": [
1599
- "https://artificialanalysis.ai/models/claude-4-5-sonnet-thinking"
1789
+ "https://artificialanalysis.ai/models/grok-4-3-medium"
1600
1790
  ],
1601
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1791
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
1602
1792
  "transport_mapping": {
1603
1793
  "status": "unknown",
1604
1794
  "candidate_model_ids": [
1605
- "claude-sonnet-4-5"
1795
+ "grok-4.3"
1606
1796
  ],
1607
1797
  "runtime_verified": false,
1608
1798
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1609
- }
1799
+ },
1800
+ "score_raw": 24.78
1610
1801
  },
1611
1802
  {
1612
- "id": "claude/claude-4-5-sonnet",
1803
+ "id": "claude/claude-sonnet-4-6",
1613
1804
  "vendor": "claude",
1614
- "model": "claude-sonnet-4-5",
1615
- "effort": null,
1805
+ "model": "claude-sonnet-4-6",
1806
+ "effort": "high",
1616
1807
  "reasoning": "non-reasoning",
1617
1808
  "fallback": null,
1618
- "aa_slug": "claude-4-5-sonnet",
1619
- "score": 23,
1809
+ "aa_slug": "claude-sonnet-4-6",
1810
+ "score": 25,
1620
1811
  "estimated": true,
1621
1812
  "evidence_marker": "estimated",
1622
- "benchmark_version": "4.2",
1623
- "as_of": "2026-09-07",
1813
+ "benchmark_version": "4.3.2",
1814
+ "as_of": "2026-09-22",
1624
1815
  "source_urls": [
1625
- "https://artificialanalysis.ai/models/claude-4-5-sonnet"
1816
+ "https://artificialanalysis.ai/models/claude-sonnet-4-6"
1626
1817
  ],
1627
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1818
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1628
1819
  "transport_mapping": {
1629
1820
  "status": "unknown",
1630
1821
  "candidate_model_ids": [
1631
- "claude-sonnet-4-5"
1822
+ "claude-sonnet-4-6"
1632
1823
  ],
1633
1824
  "runtime_verified": false,
1634
1825
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1635
- }
1826
+ },
1827
+ "score_raw": 24.69
1636
1828
  },
1637
1829
  {
1638
- "id": "claude/claude-4-5-haiku-reasoning",
1639
- "vendor": "claude",
1640
- "model": "claude-haiku-4-5",
1641
- "effort": null,
1830
+ "id": "grok/grok-4-3-low",
1831
+ "vendor": "grok",
1832
+ "model": "grok-4.3",
1833
+ "effort": "low",
1642
1834
  "reasoning": "reasoning",
1643
1835
  "fallback": null,
1644
- "aa_slug": "claude-4-5-haiku-reasoning",
1645
- "score": 22,
1646
- "estimated": false,
1647
- "evidence_marker": "unmarked",
1648
- "benchmark_version": "4.2",
1649
- "as_of": "2026-09-07",
1836
+ "aa_slug": "grok-4-3-low",
1837
+ "score": 24,
1838
+ "estimated": true,
1839
+ "evidence_marker": "estimated",
1840
+ "benchmark_version": "4.3.2",
1841
+ "as_of": "2026-09-22",
1650
1842
  "source_urls": [
1651
- "https://artificialanalysis.ai/models/claude-4-5-haiku-reasoning"
1843
+ "https://artificialanalysis.ai/models/grok-4-3-low"
1652
1844
  ],
1653
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1845
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
1654
1846
  "transport_mapping": {
1655
1847
  "status": "unknown",
1656
1848
  "candidate_model_ids": [
1657
- "claude-haiku-4-5"
1849
+ "grok-4.3"
1658
1850
  ],
1659
1851
  "runtime_verified": false,
1660
1852
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1661
- }
1853
+ },
1854
+ "score_raw": 24.3
1662
1855
  },
1663
1856
  {
1664
- "id": "claude/claude-4-5-haiku",
1857
+ "id": "claude/claude-sonnet-5-low",
1665
1858
  "vendor": "claude",
1666
- "model": "claude-haiku-4-5",
1667
- "effort": null,
1668
- "reasoning": "non-reasoning",
1859
+ "model": "claude-sonnet-5",
1860
+ "effort": "low",
1861
+ "reasoning": "adaptive",
1669
1862
  "fallback": null,
1670
- "aa_slug": "claude-4-5-haiku",
1671
- "score": 17,
1672
- "estimated": true,
1673
- "evidence_marker": "estimated",
1674
- "benchmark_version": "4.2",
1675
- "as_of": "2026-09-07",
1863
+ "aa_slug": "claude-sonnet-5-low",
1864
+ "score": 24,
1865
+ "estimated": false,
1866
+ "evidence_marker": "unmarked",
1867
+ "benchmark_version": "4.3.2",
1868
+ "as_of": "2026-09-22",
1676
1869
  "source_urls": [
1677
- "https://artificialanalysis.ai/models/claude-4-5-haiku"
1870
+ "https://artificialanalysis.ai/models/claude-sonnet-5-low"
1678
1871
  ],
1679
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
1872
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1680
1873
  "transport_mapping": {
1681
1874
  "status": "unknown",
1682
1875
  "candidate_model_ids": [
1683
- "claude-haiku-4-5"
1876
+ "claude-sonnet-5"
1684
1877
  ],
1685
1878
  "runtime_verified": false,
1686
1879
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1687
- }
1880
+ },
1881
+ "score_raw": 24.26
1688
1882
  },
1689
1883
  {
1690
- "id": "gemini/gemini-3-8-flash",
1691
- "vendor": "gemini",
1692
- "model": "gemini-3.8-flash",
1693
- "effort": "high",
1694
- "reasoning": "unspecified",
1884
+ "id": "codex/gpt-5-4-mini",
1885
+ "vendor": "codex",
1886
+ "model": "gpt-5.4-mini",
1887
+ "effort": "xhigh",
1888
+ "reasoning": "reasoning",
1695
1889
  "fallback": null,
1696
- "aa_slug": "gemini-3-8-flash",
1697
- "score": 47,
1890
+ "aa_slug": "gpt-5-4-mini",
1891
+ "score": 24,
1698
1892
  "estimated": false,
1699
1893
  "evidence_marker": "unmarked",
1700
- "benchmark_version": "4.2",
1701
- "as_of": "2026-09-07",
1894
+ "benchmark_version": "4.3.2",
1895
+ "as_of": "2026-09-22",
1702
1896
  "source_urls": [
1703
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-medium"
1897
+ "https://artificialanalysis.ai/models/gpt-5-4-mini",
1898
+ "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
1704
1899
  ],
1705
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
1900
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1706
1901
  "transport_mapping": {
1707
1902
  "status": "unknown",
1708
1903
  "candidate_model_ids": [
1709
- "gemini-3.8-flash-high"
1904
+ "gpt-5.4-mini"
1710
1905
  ],
1711
1906
  "runtime_verified": false,
1712
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1713
- }
1907
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1908
+ },
1909
+ "score_raw": 24.07
1714
1910
  },
1715
1911
  {
1716
- "id": "gemini/gemini-3-8-flash-medium",
1717
- "vendor": "gemini",
1718
- "model": "gemini-3.8-flash",
1719
- "effort": "medium",
1720
- "reasoning": "unspecified",
1912
+ "id": "claude/claude-opus-4-5",
1913
+ "vendor": "claude",
1914
+ "model": "claude-opus-4-5",
1915
+ "effort": null,
1916
+ "reasoning": "non-reasoning",
1721
1917
  "fallback": null,
1722
- "aa_slug": "gemini-3-8-flash-medium",
1723
- "score": 47,
1918
+ "aa_slug": "claude-opus-4-5",
1919
+ "score": 24,
1724
1920
  "estimated": true,
1725
1921
  "evidence_marker": "estimated",
1726
- "benchmark_version": "4.2",
1727
- "as_of": "2026-09-07",
1922
+ "benchmark_version": "4.3.2",
1923
+ "as_of": "2026-09-22",
1728
1924
  "source_urls": [
1729
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-medium"
1925
+ "https://artificialanalysis.ai/models/claude-opus-4-5"
1730
1926
  ],
1731
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
1927
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1732
1928
  "transport_mapping": {
1733
1929
  "status": "unknown",
1734
1930
  "candidate_model_ids": [
1735
- "gemini-3.8-flash-medium"
1931
+ "claude-opus-4-5"
1736
1932
  ],
1737
1933
  "runtime_verified": false,
1738
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1739
- }
1934
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1935
+ },
1936
+ "score_raw": 23.68
1740
1937
  },
1741
1938
  {
1742
- "id": "gemini/gemini-3-8-flash-low",
1743
- "vendor": "gemini",
1744
- "model": "gemini-3.8-flash",
1939
+ "id": "claude/claude-sonnet-4-6-non-reasoning-low-effort",
1940
+ "vendor": "claude",
1941
+ "model": "claude-sonnet-4-6",
1745
1942
  "effort": "low",
1746
- "reasoning": "unspecified",
1943
+ "reasoning": "non-reasoning",
1747
1944
  "fallback": null,
1748
- "aa_slug": "gemini-3-8-flash-low",
1749
- "score": 42,
1945
+ "aa_slug": "claude-sonnet-4-6-non-reasoning-low-effort",
1946
+ "score": 23,
1750
1947
  "estimated": true,
1751
1948
  "evidence_marker": "estimated",
1752
- "benchmark_version": "4.2",
1753
- "as_of": "2026-09-07",
1949
+ "benchmark_version": "4.3.2",
1950
+ "as_of": "2026-09-22",
1754
1951
  "source_urls": [
1755
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-8-flash-low"
1952
+ "https://artificialanalysis.ai/models/claude-sonnet-4-6-non-reasoning-low-effort"
1756
1953
  ],
1757
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
1954
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1758
1955
  "transport_mapping": {
1759
1956
  "status": "unknown",
1760
1957
  "candidate_model_ids": [
1761
- "gemini-3.8-flash-low"
1958
+ "claude-sonnet-4-6"
1762
1959
  ],
1763
1960
  "runtime_verified": false,
1764
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1765
- }
1961
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1962
+ },
1963
+ "score_raw": 23.3
1766
1964
  },
1767
1965
  {
1768
- "id": "gemini/gemini-3-7-flash",
1769
- "vendor": "gemini",
1770
- "model": "gemini-3.7-flash",
1966
+ "id": "claude/claude-sonnet-5-non-reasoning",
1967
+ "vendor": "claude",
1968
+ "model": "claude-sonnet-5",
1771
1969
  "effort": "high",
1772
- "reasoning": "unspecified",
1970
+ "reasoning": "non-reasoning",
1773
1971
  "fallback": null,
1774
- "aa_slug": "gemini-3-7-flash",
1775
- "score": 45,
1776
- "estimated": false,
1777
- "evidence_marker": "unmarked",
1778
- "benchmark_version": "4.2",
1779
- "as_of": "2026-09-07",
1972
+ "aa_slug": "claude-sonnet-5-non-reasoning",
1973
+ "score": 23,
1974
+ "estimated": true,
1975
+ "evidence_marker": "estimated",
1976
+ "benchmark_version": "4.3.2",
1977
+ "as_of": "2026-09-22",
1780
1978
  "source_urls": [
1781
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash"
1979
+ "https://artificialanalysis.ai/models/claude-sonnet-5-non-reasoning"
1782
1980
  ],
1783
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
1981
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1784
1982
  "transport_mapping": {
1785
1983
  "status": "unknown",
1786
1984
  "candidate_model_ids": [
1787
- "gemini-3.7-flash-high"
1985
+ "claude-sonnet-5"
1788
1986
  ],
1789
1987
  "runtime_verified": false,
1790
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1791
- }
1988
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1989
+ },
1990
+ "score_raw": 23.2
1792
1991
  },
1793
1992
  {
1794
- "id": "gemini/gemini-3-7-flash-medium",
1795
- "vendor": "gemini",
1796
- "model": "gemini-3.7-flash",
1797
- "effort": "medium",
1798
- "reasoning": "unspecified",
1993
+ "id": "codex/gpt-5-5-non-reasoning",
1994
+ "vendor": "codex",
1995
+ "model": "gpt-5.5",
1996
+ "effort": null,
1997
+ "reasoning": "non-reasoning",
1799
1998
  "fallback": null,
1800
- "aa_slug": "gemini-3-7-flash-medium",
1801
- "score": 43,
1999
+ "aa_slug": "gpt-5-5-non-reasoning",
2000
+ "score": 23,
1802
2001
  "estimated": true,
1803
2002
  "evidence_marker": "estimated",
1804
- "benchmark_version": "4.2",
1805
- "as_of": "2026-09-07",
2003
+ "benchmark_version": "4.3.2",
2004
+ "as_of": "2026-09-22",
1806
2005
  "source_urls": [
1807
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash-medium"
2006
+ "https://artificialanalysis.ai/models/gpt-5-5-non-reasoning",
2007
+ "https://artificialanalysis.ai/models/releases/gpt-5-5"
1808
2008
  ],
1809
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2009
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1810
2010
  "transport_mapping": {
1811
2011
  "status": "unknown",
1812
2012
  "candidate_model_ids": [
1813
- "gemini-3.7-flash-medium"
2013
+ "gpt-5.5"
1814
2014
  ],
1815
2015
  "runtime_verified": false,
1816
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1817
- }
2016
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2017
+ },
2018
+ "score_raw": 23.17
1818
2019
  },
1819
2020
  {
1820
- "id": "gemini/gemini-3-7-flash-low",
1821
- "vendor": "gemini",
1822
- "model": "gemini-3.7-flash",
2021
+ "id": "codex/gpt-5-6-luna-low",
2022
+ "vendor": "codex",
2023
+ "model": "gpt-5.6-luna",
1823
2024
  "effort": "low",
1824
- "reasoning": "unspecified",
2025
+ "reasoning": "reasoning",
1825
2026
  "fallback": null,
1826
- "aa_slug": "gemini-3-7-flash-low",
1827
- "score": 41,
1828
- "estimated": true,
1829
- "evidence_marker": "estimated",
1830
- "benchmark_version": "4.2",
1831
- "as_of": "2026-09-07",
2027
+ "aa_slug": "gpt-5-6-luna-low",
2028
+ "score": 21,
2029
+ "estimated": false,
2030
+ "evidence_marker": "unmarked",
2031
+ "benchmark_version": "4.3.2",
2032
+ "as_of": "2026-09-22",
1832
2033
  "source_urls": [
1833
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-7-flash-low"
2034
+ "https://artificialanalysis.ai/models/gpt-5-6-luna-low",
2035
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
1834
2036
  ],
1835
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2037
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1836
2038
  "transport_mapping": {
1837
2039
  "status": "unknown",
1838
2040
  "candidate_model_ids": [
1839
- "gemini-3.7-flash-low"
2041
+ "gpt-5.6-luna"
1840
2042
  ],
1841
2043
  "runtime_verified": false,
1842
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1843
- }
2044
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2045
+ },
2046
+ "score_raw": 21.01
1844
2047
  },
1845
2048
  {
1846
- "id": "gemini/gemini-3-6-flash",
1847
- "vendor": "gemini",
1848
- "model": "gemini-3.6-flash",
1849
- "effort": "high",
1850
- "reasoning": "unspecified",
2049
+ "id": "codex/gpt-5-6-terra-non-reasoning",
2050
+ "vendor": "codex",
2051
+ "model": "gpt-5.6-terra",
2052
+ "effort": null,
2053
+ "reasoning": "non-reasoning",
1851
2054
  "fallback": null,
1852
- "aa_slug": "gemini-3-6-flash",
1853
- "score": 40,
2055
+ "aa_slug": "gpt-5-6-terra-non-reasoning",
2056
+ "score": 21,
1854
2057
  "estimated": false,
1855
2058
  "evidence_marker": "unmarked",
1856
- "benchmark_version": "4.2",
1857
- "as_of": "2026-09-07",
2059
+ "benchmark_version": "4.3.2",
2060
+ "as_of": "2026-09-22",
1858
2061
  "source_urls": [
1859
- "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-6-flash"
2062
+ "https://artificialanalysis.ai/models/gpt-5-6-terra-non-reasoning",
2063
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-terra"
1860
2064
  ],
1861
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2065
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1862
2066
  "transport_mapping": {
1863
2067
  "status": "unknown",
1864
2068
  "candidate_model_ids": [
1865
- "gemini-3.6-flash-high"
2069
+ "gpt-5.6-terra"
1866
2070
  ],
1867
2071
  "runtime_verified": false,
1868
- "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
1869
- }
2072
+ "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2073
+ },
2074
+ "score_raw": 20.78
1870
2075
  },
1871
2076
  {
1872
- "id": "grok/grok-4-6",
1873
- "vendor": "grok",
1874
- "model": "grok-4.6",
1875
- "effort": "high",
2077
+ "id": "claude/claude-4-5-sonnet-thinking",
2078
+ "vendor": "claude",
2079
+ "model": "claude-sonnet-4-5",
2080
+ "effort": null,
1876
2081
  "reasoning": "reasoning",
1877
2082
  "fallback": null,
1878
- "aa_slug": "grok-4-6",
1879
- "score": 51,
2083
+ "aa_slug": "claude-4-5-sonnet-thinking",
2084
+ "score": 21,
1880
2085
  "estimated": false,
1881
2086
  "evidence_marker": "unmarked",
1882
- "benchmark_version": "4.2",
1883
- "as_of": "2026-09-07",
2087
+ "benchmark_version": "4.3.2",
2088
+ "as_of": "2026-09-22",
1884
2089
  "source_urls": [
1885
- "https://artificialanalysis.ai/models/grok-4-6"
2090
+ "https://artificialanalysis.ai/models/claude-4-5-sonnet-thinking"
1886
2091
  ],
1887
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2092
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1888
2093
  "transport_mapping": {
1889
2094
  "status": "unknown",
1890
2095
  "candidate_model_ids": [
1891
- "grok-4.6"
2096
+ "claude-sonnet-4-5"
1892
2097
  ],
1893
2098
  "runtime_verified": false,
1894
2099
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1895
- }
2100
+ },
2101
+ "score_raw": 20.67
1896
2102
  },
1897
2103
  {
1898
- "id": "grok/grok-4-6-xhigh",
1899
- "vendor": "grok",
1900
- "model": "grok-4.6",
1901
- "effort": "xhigh",
2104
+ "id": "codex/gpt-5-4-mini-medium",
2105
+ "vendor": "codex",
2106
+ "model": "gpt-5.4-mini",
2107
+ "effort": "medium",
1902
2108
  "reasoning": "reasoning",
1903
2109
  "fallback": null,
1904
- "aa_slug": "grok-4-6-xhigh",
1905
- "score": 49,
2110
+ "aa_slug": "gpt-5-4-mini-medium",
2111
+ "score": 20,
1906
2112
  "estimated": true,
1907
2113
  "evidence_marker": "estimated",
1908
- "benchmark_version": "4.2",
1909
- "as_of": "2026-09-07",
2114
+ "benchmark_version": "4.3.2",
2115
+ "as_of": "2026-09-22",
1910
2116
  "source_urls": [
1911
- "https://artificialanalysis.ai/models/grok-4-6-xhigh"
2117
+ "https://artificialanalysis.ai/models/gpt-5-4-mini-medium",
2118
+ "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
1912
2119
  ],
1913
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2120
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1914
2121
  "transport_mapping": {
1915
2122
  "status": "unknown",
1916
2123
  "candidate_model_ids": [
1917
- "grok-4.6"
2124
+ "gpt-5.4-mini"
1918
2125
  ],
1919
2126
  "runtime_verified": false,
1920
2127
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1921
- }
2128
+ },
2129
+ "score_raw": 19.75
1922
2130
  },
1923
2131
  {
1924
- "id": "grok/grok-4-6-medium",
1925
- "vendor": "grok",
1926
- "model": "grok-4.6",
1927
- "effort": "medium",
1928
- "reasoning": "reasoning",
2132
+ "id": "claude/claude-4-5-sonnet",
2133
+ "vendor": "claude",
2134
+ "model": "claude-sonnet-4-5",
2135
+ "effort": null,
2136
+ "reasoning": "non-reasoning",
1929
2137
  "fallback": null,
1930
- "aa_slug": "grok-4-6-medium",
1931
- "score": 48,
2138
+ "aa_slug": "claude-4-5-sonnet",
2139
+ "score": 19,
1932
2140
  "estimated": true,
1933
2141
  "evidence_marker": "estimated",
1934
- "benchmark_version": "4.2",
1935
- "as_of": "2026-09-07",
2142
+ "benchmark_version": "4.3.2",
2143
+ "as_of": "2026-09-22",
1936
2144
  "source_urls": [
1937
- "https://artificialanalysis.ai/models/grok-4-6-medium"
2145
+ "https://artificialanalysis.ai/models/claude-4-5-sonnet"
1938
2146
  ],
1939
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2147
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1940
2148
  "transport_mapping": {
1941
2149
  "status": "unknown",
1942
2150
  "candidate_model_ids": [
1943
- "grok-4.6"
2151
+ "claude-sonnet-4-5"
1944
2152
  ],
1945
2153
  "runtime_verified": false,
1946
2154
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1947
- }
2155
+ },
2156
+ "score_raw": 19.34
1948
2157
  },
1949
2158
  {
1950
- "id": "grok/grok-4-6-low",
1951
- "vendor": "grok",
1952
- "model": "grok-4.6",
1953
- "effort": "low",
1954
- "reasoning": "reasoning",
2159
+ "id": "codex/gpt-5-4-non-reasoning",
2160
+ "vendor": "codex",
2161
+ "model": "gpt-5.4",
2162
+ "effort": null,
2163
+ "reasoning": "non-reasoning",
1955
2164
  "fallback": null,
1956
- "aa_slug": "grok-4-6-low",
1957
- "score": 42,
2165
+ "aa_slug": "gpt-5-4-non-reasoning",
2166
+ "score": 18,
1958
2167
  "estimated": true,
1959
2168
  "evidence_marker": "estimated",
1960
- "benchmark_version": "4.2",
1961
- "as_of": "2026-09-07",
2169
+ "benchmark_version": "4.3.2",
2170
+ "as_of": "2026-09-22",
1962
2171
  "source_urls": [
1963
- "https://artificialanalysis.ai/models/grok-4-6-low"
2172
+ "https://artificialanalysis.ai/models/gpt-5-4-non-reasoning",
2173
+ "https://artificialanalysis.ai/models/releases/gpt-5-4"
1964
2174
  ],
1965
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2175
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
1966
2176
  "transport_mapping": {
1967
2177
  "status": "unknown",
1968
2178
  "candidate_model_ids": [
1969
- "grok-4.6"
2179
+ "gpt-5.4"
1970
2180
  ],
1971
2181
  "runtime_verified": false,
1972
2182
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1973
- }
2183
+ },
2184
+ "score_raw": 18.16
1974
2185
  },
1975
2186
  {
1976
- "id": "grok/grok-4-5",
1977
- "vendor": "grok",
1978
- "model": "grok-4.5",
1979
- "effort": "high",
2187
+ "id": "claude/claude-4-5-haiku-reasoning",
2188
+ "vendor": "claude",
2189
+ "model": "claude-haiku-4-5",
2190
+ "effort": null,
1980
2191
  "reasoning": "reasoning",
1981
2192
  "fallback": null,
1982
- "aa_slug": "grok-4-5",
1983
- "score": 45,
2193
+ "aa_slug": "claude-4-5-haiku-reasoning",
2194
+ "score": 17,
1984
2195
  "estimated": false,
1985
2196
  "evidence_marker": "unmarked",
1986
- "benchmark_version": "4.2",
1987
- "as_of": "2026-09-07",
2197
+ "benchmark_version": "4.3.2",
2198
+ "as_of": "2026-09-22",
1988
2199
  "source_urls": [
1989
- "https://artificialanalysis.ai/models/grok-4-5"
2200
+ "https://artificialanalysis.ai/models/claude-4-5-haiku-reasoning"
1990
2201
  ],
1991
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2202
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
1992
2203
  "transport_mapping": {
1993
2204
  "status": "unknown",
1994
2205
  "candidate_model_ids": [
1995
- "grok-4.5"
2206
+ "claude-haiku-4-5"
1996
2207
  ],
1997
2208
  "runtime_verified": false,
1998
2209
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
1999
- }
2210
+ },
2211
+ "score_raw": 16.88
2000
2212
  },
2001
2213
  {
2002
- "id": "grok/grok-4-3",
2003
- "vendor": "grok",
2004
- "model": "grok-4.3",
2005
- "effort": "high",
2006
- "reasoning": "reasoning",
2214
+ "id": "codex/gpt-5-6-luna-non-reasoning",
2215
+ "vendor": "codex",
2216
+ "model": "gpt-5.6-luna",
2217
+ "effort": null,
2218
+ "reasoning": "non-reasoning",
2007
2219
  "fallback": null,
2008
- "aa_slug": "grok-4-3",
2009
- "score": 29,
2010
- "estimated": true,
2011
- "evidence_marker": "estimated",
2012
- "benchmark_version": "4.2",
2013
- "as_of": "2026-09-07",
2220
+ "aa_slug": "gpt-5-6-luna-non-reasoning",
2221
+ "score": 16,
2222
+ "estimated": false,
2223
+ "evidence_marker": "unmarked",
2224
+ "benchmark_version": "4.3.2",
2225
+ "as_of": "2026-09-22",
2014
2226
  "source_urls": [
2015
- "https://artificialanalysis.ai/models/grok-4-3"
2227
+ "https://artificialanalysis.ai/models/gpt-5-6-luna-non-reasoning",
2228
+ "https://artificialanalysis.ai/models/releases/gpt-5-6-luna"
2016
2229
  ],
2017
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2230
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
2018
2231
  "transport_mapping": {
2019
2232
  "status": "unknown",
2020
2233
  "candidate_model_ids": [
2021
- "grok-4.3"
2234
+ "gpt-5.6-luna"
2022
2235
  ],
2023
2236
  "runtime_verified": false,
2024
2237
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2025
- }
2238
+ },
2239
+ "score_raw": 15.53
2026
2240
  },
2027
2241
  {
2028
- "id": "grok/grok-4-3-medium",
2029
- "vendor": "grok",
2030
- "model": "grok-4.3",
2031
- "effort": "medium",
2032
- "reasoning": "reasoning",
2242
+ "id": "claude/claude-4-5-haiku",
2243
+ "vendor": "claude",
2244
+ "model": "claude-haiku-4-5",
2245
+ "effort": null,
2246
+ "reasoning": "non-reasoning",
2033
2247
  "fallback": null,
2034
- "aa_slug": "grok-4-3-medium",
2035
- "score": 29,
2248
+ "aa_slug": "claude-4-5-haiku",
2249
+ "score": 15,
2036
2250
  "estimated": true,
2037
2251
  "evidence_marker": "estimated",
2038
- "benchmark_version": "4.2",
2039
- "as_of": "2026-09-07",
2252
+ "benchmark_version": "4.3.2",
2253
+ "as_of": "2026-09-22",
2040
2254
  "source_urls": [
2041
- "https://artificialanalysis.ai/models/grok-4-3-medium"
2255
+ "https://artificialanalysis.ai/models/claude-4-5-haiku"
2042
2256
  ],
2043
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2257
+ "evidence_report": "docs/reports/aa-claude-evidence-2026-09-22.md",
2044
2258
  "transport_mapping": {
2045
2259
  "status": "unknown",
2046
2260
  "candidate_model_ids": [
2047
- "grok-4.3"
2261
+ "claude-haiku-4-5"
2048
2262
  ],
2049
2263
  "runtime_verified": false,
2050
2264
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2051
- }
2265
+ },
2266
+ "score_raw": 15.41
2052
2267
  },
2053
2268
  {
2054
- "id": "grok/grok-4-3-low",
2269
+ "id": "grok/grok-4-3-non-reasoning",
2055
2270
  "vendor": "grok",
2056
2271
  "model": "grok-4.3",
2057
- "effort": "low",
2058
- "reasoning": "reasoning",
2272
+ "effort": null,
2273
+ "reasoning": "non-reasoning",
2059
2274
  "fallback": null,
2060
- "aa_slug": "grok-4-3-low",
2061
- "score": 29,
2062
- "estimated": true,
2063
- "evidence_marker": "estimated",
2064
- "benchmark_version": "4.2",
2065
- "as_of": "2026-09-07",
2275
+ "aa_slug": "grok-4-3-non-reasoning",
2276
+ "score": 14,
2277
+ "estimated": false,
2278
+ "evidence_marker": "unmarked",
2279
+ "benchmark_version": "4.3.2",
2280
+ "as_of": "2026-09-22",
2066
2281
  "source_urls": [
2067
- "https://artificialanalysis.ai/models/grok-4-3-low"
2282
+ "https://artificialanalysis.ai/models/grok-4-3-non-reasoning"
2068
2283
  ],
2069
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2284
+ "evidence_report": "docs/reports/aa-grok-evidence-2026-09-22.md",
2070
2285
  "transport_mapping": {
2071
2286
  "status": "unknown",
2072
2287
  "candidate_model_ids": [
@@ -2074,33 +2289,36 @@
2074
2289
  ],
2075
2290
  "runtime_verified": false,
2076
2291
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2077
- }
2292
+ },
2293
+ "score_raw": 13.99
2078
2294
  },
2079
2295
  {
2080
- "id": "grok/grok-4-3-non-reasoning",
2081
- "vendor": "grok",
2082
- "model": "grok-4.3",
2296
+ "id": "codex/gpt-5-4-mini-non-reasoning",
2297
+ "vendor": "codex",
2298
+ "model": "gpt-5.4-mini",
2083
2299
  "effort": null,
2084
2300
  "reasoning": "non-reasoning",
2085
2301
  "fallback": null,
2086
- "aa_slug": "grok-4-3-non-reasoning",
2087
- "score": 17,
2302
+ "aa_slug": "gpt-5-4-mini-non-reasoning",
2303
+ "score": 11,
2088
2304
  "estimated": true,
2089
2305
  "evidence_marker": "estimated",
2090
- "benchmark_version": "4.2",
2091
- "as_of": "2026-09-07",
2306
+ "benchmark_version": "4.3.2",
2307
+ "as_of": "2026-09-22",
2092
2308
  "source_urls": [
2093
- "https://artificialanalysis.ai/models/grok-4-3-non-reasoning"
2309
+ "https://artificialanalysis.ai/models/gpt-5-4-mini-non-reasoning",
2310
+ "https://artificialanalysis.ai/models/releases/gpt-5-4-mini"
2094
2311
  ],
2095
- "evidence_report": "docs/reports/aa-grok-evidence-2026-09-07.md",
2312
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
2096
2313
  "transport_mapping": {
2097
2314
  "status": "unknown",
2098
2315
  "candidate_model_ids": [
2099
- "grok-4.3"
2316
+ "gpt-5.4-mini"
2100
2317
  ],
2101
2318
  "runtime_verified": false,
2102
2319
  "reason": "AA evaluation identity is known; runtime transport identity has not been verified in this evidence snapshot"
2103
- }
2320
+ },
2321
+ "score_raw": 11.14
2104
2322
  }
2105
2323
  ],
2106
2324
  "unknown_configs": [
@@ -2113,13 +2331,13 @@
2113
2331
  "fallback": null,
2114
2332
  "score": null,
2115
2333
  "estimated": null,
2116
- "benchmark_version": "4.2",
2117
- "as_of": "2026-09-07",
2334
+ "benchmark_version": "4.3.2",
2335
+ "as_of": "2026-09-22",
2118
2336
  "status": "unknown",
2119
2337
  "authority_eligible": false,
2120
2338
  "reason": "Generic model alias is not verified as Sol",
2121
2339
  "source_urls": [],
2122
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
2340
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
2123
2341
  "mapping_status": "unknown",
2124
2342
  "transport_mapping": {
2125
2343
  "status": "unknown",
@@ -2136,59 +2354,13 @@
2136
2354
  "fallback": null,
2137
2355
  "score": null,
2138
2356
  "estimated": null,
2139
- "benchmark_version": "4.2",
2140
- "as_of": "2026-09-07",
2357
+ "benchmark_version": "4.3.2",
2358
+ "as_of": "2026-09-22",
2141
2359
  "status": "unknown",
2142
2360
  "authority_eligible": false,
2143
2361
  "reason": "No exact AA configuration found; GPT-5.3 Codex score is not transferable",
2144
2362
  "source_urls": [],
2145
- "evidence_report": "docs/reports/aa-codex-evidence-2026-09-07.md",
2146
- "mapping_status": "unknown",
2147
- "transport_mapping": {
2148
- "status": "unknown",
2149
- "runtime_verified": false,
2150
- "resolved_config_id": null
2151
- }
2152
- },
2153
- {
2154
- "id": "claude/claude-sonnet-5/xhigh/adaptive",
2155
- "vendor": "claude",
2156
- "model": "claude-sonnet-5",
2157
- "effort": "xhigh",
2158
- "reasoning": "adaptive",
2159
- "fallback": null,
2160
- "score": null,
2161
- "estimated": null,
2162
- "benchmark_version": "4.2",
2163
- "as_of": "2026-09-07",
2164
- "status": "unknown",
2165
- "authority_eligible": false,
2166
- "reason": "AA variant exists but Intelligence score unavailable in approved evidence",
2167
- "source_urls": [],
2168
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
2169
- "mapping_status": "unknown",
2170
- "transport_mapping": {
2171
- "status": "unknown",
2172
- "runtime_verified": false,
2173
- "resolved_config_id": null
2174
- }
2175
- },
2176
- {
2177
- "id": "claude/claude-sonnet-5/high/adaptive",
2178
- "vendor": "claude",
2179
- "model": "claude-sonnet-5",
2180
- "effort": "high",
2181
- "reasoning": "adaptive",
2182
- "fallback": null,
2183
- "score": null,
2184
- "estimated": null,
2185
- "benchmark_version": "4.2",
2186
- "as_of": "2026-09-07",
2187
- "status": "unknown",
2188
- "authority_eligible": false,
2189
- "reason": "AA variant exists but Intelligence score unavailable in approved evidence",
2190
- "source_urls": [],
2191
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
2363
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
2192
2364
  "mapping_status": "unknown",
2193
2365
  "transport_mapping": {
2194
2366
  "status": "unknown",
@@ -2197,44 +2369,21 @@
2197
2369
  }
2198
2370
  },
2199
2371
  {
2200
- "id": "claude/claude-sonnet-5/medium/adaptive",
2201
- "vendor": "claude",
2202
- "model": "claude-sonnet-5",
2372
+ "id": "gemini/gemini-3.6-flash/medium/unspecified",
2373
+ "vendor": "gemini",
2374
+ "model": "gemini-3.6-flash",
2203
2375
  "effort": "medium",
2204
- "reasoning": "adaptive",
2205
- "fallback": null,
2206
- "score": null,
2207
- "estimated": null,
2208
- "benchmark_version": "4.2",
2209
- "as_of": "2026-09-07",
2210
- "status": "unknown",
2211
- "authority_eligible": false,
2212
- "reason": "AA variant exists but Intelligence score unavailable in approved evidence",
2213
- "source_urls": [],
2214
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
2215
- "mapping_status": "unknown",
2216
- "transport_mapping": {
2217
- "status": "unknown",
2218
- "runtime_verified": false,
2219
- "resolved_config_id": null
2220
- }
2221
- },
2222
- {
2223
- "id": "claude/claude-sonnet-5/low/adaptive",
2224
- "vendor": "claude",
2225
- "model": "claude-sonnet-5",
2226
- "effort": "low",
2227
- "reasoning": "adaptive",
2376
+ "reasoning": "unspecified",
2228
2377
  "fallback": null,
2229
2378
  "score": null,
2230
2379
  "estimated": null,
2231
- "benchmark_version": "4.2",
2232
- "as_of": "2026-09-07",
2380
+ "benchmark_version": "4.3.2",
2381
+ "as_of": "2026-09-22",
2233
2382
  "status": "unknown",
2234
2383
  "authority_eligible": false,
2235
- "reason": "AA variant exists but Intelligence score unavailable in approved evidence",
2384
+ "reason": "No separately verified AA score; do not reuse high score",
2236
2385
  "source_urls": [],
2237
- "evidence_report": "docs/reports/aa-claude-evidence-2026-09-07.md",
2386
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
2238
2387
  "mapping_status": "unknown",
2239
2388
  "transport_mapping": {
2240
2389
  "status": "unknown",
@@ -2243,21 +2392,21 @@
2243
2392
  }
2244
2393
  },
2245
2394
  {
2246
- "id": "gemini/gemini-3.6-flash/medium/unspecified",
2395
+ "id": "gemini/gemini-3.6-flash/low/unspecified",
2247
2396
  "vendor": "gemini",
2248
2397
  "model": "gemini-3.6-flash",
2249
- "effort": "medium",
2398
+ "effort": "low",
2250
2399
  "reasoning": "unspecified",
2251
2400
  "fallback": null,
2252
2401
  "score": null,
2253
2402
  "estimated": null,
2254
- "benchmark_version": "4.2",
2255
- "as_of": "2026-09-07",
2403
+ "benchmark_version": "4.3.2",
2404
+ "as_of": "2026-09-22",
2256
2405
  "status": "unknown",
2257
2406
  "authority_eligible": false,
2258
2407
  "reason": "No separately verified AA score; do not reuse high score",
2259
2408
  "source_urls": [],
2260
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2409
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
2261
2410
  "mapping_status": "unknown",
2262
2411
  "transport_mapping": {
2263
2412
  "status": "unknown",
@@ -2266,21 +2415,21 @@
2266
2415
  }
2267
2416
  },
2268
2417
  {
2269
- "id": "gemini/gemini-3.6-flash/low/unspecified",
2418
+ "id": "gemini/gemini-3.1-pro/high/unspecified",
2270
2419
  "vendor": "gemini",
2271
- "model": "gemini-3.6-flash",
2272
- "effort": "low",
2420
+ "model": "gemini-3.1-pro",
2421
+ "effort": "high",
2273
2422
  "reasoning": "unspecified",
2274
2423
  "fallback": null,
2275
2424
  "score": null,
2276
2425
  "estimated": null,
2277
- "benchmark_version": "4.2",
2278
- "as_of": "2026-09-07",
2426
+ "benchmark_version": "4.3.2",
2427
+ "as_of": "2026-09-22",
2279
2428
  "status": "unknown",
2280
2429
  "authority_eligible": false,
2281
- "reason": "No separately verified AA score; do not reuse high score",
2430
+ "reason": "Reference-only Preview is not exact local high/low configuration",
2282
2431
  "source_urls": [],
2283
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2432
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
2284
2433
  "mapping_status": "unknown",
2285
2434
  "transport_mapping": {
2286
2435
  "status": "unknown",
@@ -2289,21 +2438,21 @@
2289
2438
  }
2290
2439
  },
2291
2440
  {
2292
- "id": "gemini/gemini-3.1-pro/high/unspecified",
2441
+ "id": "gemini/gemini-3.1-pro/low/unspecified",
2293
2442
  "vendor": "gemini",
2294
2443
  "model": "gemini-3.1-pro",
2295
- "effort": "high",
2444
+ "effort": "low",
2296
2445
  "reasoning": "unspecified",
2297
2446
  "fallback": null,
2298
2447
  "score": null,
2299
2448
  "estimated": null,
2300
- "benchmark_version": "4.2",
2301
- "as_of": "2026-09-07",
2449
+ "benchmark_version": "4.3.2",
2450
+ "as_of": "2026-09-22",
2302
2451
  "status": "unknown",
2303
2452
  "authority_eligible": false,
2304
2453
  "reason": "Reference-only Preview is not exact local high/low configuration",
2305
2454
  "source_urls": [],
2306
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2455
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
2307
2456
  "mapping_status": "unknown",
2308
2457
  "transport_mapping": {
2309
2458
  "status": "unknown",
@@ -2312,21 +2461,21 @@
2312
2461
  }
2313
2462
  },
2314
2463
  {
2315
- "id": "gemini/gemini-3.1-pro/low/unspecified",
2316
- "vendor": "gemini",
2317
- "model": "gemini-3.1-pro",
2318
- "effort": "low",
2319
- "reasoning": "unspecified",
2464
+ "id": "codex/gpt-6-astra/none/non-reasoning",
2465
+ "vendor": "codex",
2466
+ "model": "gpt-6-astra",
2467
+ "effort": null,
2468
+ "reasoning": "non-reasoning",
2320
2469
  "fallback": null,
2321
2470
  "score": null,
2322
2471
  "estimated": null,
2323
- "benchmark_version": "4.2",
2324
- "as_of": "2026-09-07",
2472
+ "benchmark_version": "4.3.2",
2473
+ "as_of": "2026-09-22",
2325
2474
  "status": "unknown",
2326
2475
  "authority_eligible": false,
2327
- "reason": "Reference-only Preview is not exact local high/low configuration",
2476
+ "reason": "AA v4.3.2 no longer lists gpt-6-astra-non-reasoning; the earlier score is not carried over",
2328
2477
  "source_urls": [],
2329
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2478
+ "evidence_report": "docs/reports/aa-codex-evidence-2026-09-22.md",
2330
2479
  "mapping_status": "unknown",
2331
2480
  "transport_mapping": {
2332
2481
  "status": "unknown",
@@ -2344,15 +2493,15 @@
2344
2493
  "reasoning": "unspecified",
2345
2494
  "fallback": null,
2346
2495
  "aa_slug": "gemini-3-1-pro-preview",
2347
- "score": 37,
2496
+ "score": 30,
2348
2497
  "estimated": false,
2349
2498
  "evidence_marker": "unmarked",
2350
- "benchmark_version": "4.2",
2351
- "as_of": "2026-09-07",
2499
+ "benchmark_version": "4.3.2",
2500
+ "as_of": "2026-09-22",
2352
2501
  "source_urls": [
2353
2502
  "https://artificialanalysis.ai/models/comparisons/gemini-3-8-flash-vs-gemini-3-1-pro-preview"
2354
2503
  ],
2355
- "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-07.md",
2504
+ "evidence_report": "docs/reports/aa-gemini-evidence-2026-09-22.md",
2356
2505
  "transport_mapping": {
2357
2506
  "status": "unknown",
2358
2507
  "candidate_model_ids": [],
@@ -2360,7 +2509,8 @@
2360
2509
  "reason": "Report establishes AA model/effort but not exact runtime transport or reasoning mode; unspecified is literal, not a wildcard"
2361
2510
  },
2362
2511
  "authority_eligible": false,
2363
- "reference_reason": "Proposal v2 retains this only as comparison; no local high/low mapping approved"
2512
+ "reference_reason": "Proposal v2 retains this only as comparison; no local high/low mapping approved",
2513
+ "score_raw": 29.72
2364
2514
  }
2365
2515
  ],
2366
2516
  "aliases": [
@@ -2374,8 +2524,7 @@
2374
2524
  "codex/gpt-6-astra-xhigh",
2375
2525
  "codex/gpt-6-astra-high",
2376
2526
  "codex/gpt-6-astra-medium",
2377
- "codex/gpt-6-astra-low",
2378
- "codex/gpt-6-astra-non-reasoning"
2527
+ "codex/gpt-6-astra-low"
2379
2528
  ],
2380
2529
  "implicit_default_effort": null,
2381
2530
  "runtime_verified": false,
@@ -2679,7 +2828,11 @@
2679
2828
  "resolved_config_id": null,
2680
2829
  "candidate_config_ids": [
2681
2830
  "claude/claude-sonnet-5",
2682
- "claude/claude-sonnet-5-non-reasoning"
2831
+ "claude/claude-sonnet-5-non-reasoning",
2832
+ "claude/claude-sonnet-5-xhigh",
2833
+ "claude/claude-sonnet-5-high",
2834
+ "claude/claude-sonnet-5-medium",
2835
+ "claude/claude-sonnet-5-low"
2683
2836
  ],
2684
2837
  "implicit_default_effort": null,
2685
2838
  "runtime_verified": false,
@@ -2980,6 +3133,21 @@
2980
3133
  "catalog_source": "scripts/configure.sh",
2981
3134
  "source_kind": "local_catalog_not_runtime_identity"
2982
3135
  },
3136
+ {
3137
+ "catalog_vendor": "grok",
3138
+ "catalog_model": "grok-4.7",
3139
+ "status": "unknown",
3140
+ "resolved_config_id": null,
3141
+ "candidate_config_ids": [
3142
+ "grok/grok-4-7",
3143
+ "grok/grok-4-7-high"
3144
+ ],
3145
+ "implicit_default_effort": null,
3146
+ "runtime_verified": false,
3147
+ "reason": "Exact effort/reasoning/fallback and runtime transport verification required; candidates are not authorization",
3148
+ "catalog_source": "scripts/configure.sh",
3149
+ "source_kind": "local_catalog_not_runtime_identity"
3150
+ },
2983
3151
  {
2984
3152
  "catalog_vendor": "grok",
2985
3153
  "catalog_model": "grok-4.6",
@@ -3023,15 +3191,15 @@
3023
3191
  }
3024
3192
  ],
3025
3193
  "coverage": {
3026
- "scored_configs": 78,
3194
+ "scored_configs": 83,
3027
3195
  "reference_configs": 1,
3028
- "unknown_configs": 10,
3029
- "aliases": 47,
3196
+ "unknown_configs": 7,
3197
+ "aliases": 48,
3030
3198
  "by_vendor": {
3031
- "codex": 35,
3032
- "claude": 27,
3199
+ "codex": 34,
3200
+ "claude": 31,
3033
3201
  "gemini": 7,
3034
- "grok": 9
3202
+ "grok": 11
3035
3203
  }
3036
3204
  },
3037
3205
  "schema_notes": {
@@ -3041,6 +3209,7 @@
3041
3209
  "effort_null": "No AA effort label established for this reasoning configuration; do not fill with model maximum",
3042
3210
  "candidate_config_ids": "Research cross-reference only; never authorize from an unresolved candidate",
3043
3211
  "transport_mapping": "Live executor must verify exact vendor/model/effort/reasoning/fallback before evaluating score ceiling; all snapshot transports remain unverified",
3044
- "reference_configs": "Excluded from authority lookup"
3212
+ "reference_configs": "Excluded from authority lookup",
3213
+ "score_rounding": "score is score_raw rounded half-up to an integer; score_raw is the AA index to two decimals"
3045
3214
  }
3046
3215
  }