omnilane 0.44.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,409 @@
1
+ #!/usr/bin/env python3
2
+ """Re-baseline config/aa-model-policy.json onto a newer AA Intelligence Index.
3
+
4
+ fetch download one AA model page and save the per-model records as an extract
5
+ build regenerate the registry from a saved extract (never from the network)
6
+ report write per-vendor evidence tables and the old-vs-new score diff
7
+ matrix show, per controller, the first reachable target of every lane
8
+ lanes print, per lane, each candidate with the measurements the lane is ordered on
9
+
10
+ The registry is an approval artifact: build only rewrites the file. Pinning its
11
+ sha256 in scripts/lib/aa_policy.py and re-signing the host overlay stay manual.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import argparse
16
+ import copy
17
+ import hashlib
18
+ import json
19
+ import re
20
+ import subprocess
21
+ import sys
22
+ from datetime import datetime
23
+ from decimal import ROUND_HALF_UP, Decimal
24
+ from pathlib import Path
25
+
26
+ REPO = Path(__file__).resolve().parents[1]
27
+ REGISTRY = REPO / "config/aa-model-policy.json"
28
+ PAGE = "https://artificialanalysis.ai/models/{slug}"
29
+ UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
30
+ "(KHTML, like Gecko) Chrome/140.0 Safari/537.36")
31
+ VENDOR_SLUG = re.compile(r"^(gpt|claude|gemini|grok)-")
32
+ KEEP = (
33
+ "name", "intelligenceIndex", "intelligenceIndexIsEstimated", "releaseDate", "deprecated",
34
+ "contextWindowTokens", "terminalBench21", "terminalbenchHard", "scicode", "hle", "gpqa",
35
+ "omniscience", "tau2", "lcr", "ifbench", "critpt", "gdpvalNormalized", "itBenchSre",
36
+ "apexAgents", "price1mInputTokens", "price1mOutputTokens", "price1mBlended7To2To1",
37
+ "intelligenceIndexOutputTokensPerTask", "timeToFirstAnswerToken",
38
+ # what routing.yaml orders its lanes on
39
+ "terminalBench40", "automationBenchPartialScore", "tauBanking", "mlcrOverall", "mmmuPro",
40
+ "omniscienceBreakdown", "briefcaseBreakdown", "intelligenceIndexCost", "intelligenceIndexTimePerTask",
41
+ )
42
+
43
+ # Rows this snapshot adds. identity mirrors the sibling rows of the same model.
44
+ NEW_ROWS = [
45
+ # (id, vendor, model, effort, reasoning, aa_slug, copy transport/shape from)
46
+ ("grok/grok-4-7", "grok", "grok-4.7", "xhigh", "reasoning", "grok-4-7", "grok/grok-4-6-xhigh"),
47
+ ("grok/grok-4-7-high", "grok", "grok-4.7", "high", "reasoning", "grok-4-7-high", "grok/grok-4-6"),
48
+ ("claude/claude-sonnet-5-xhigh", "claude", "claude-sonnet-5", "xhigh", "adaptive",
49
+ "claude-sonnet-5-xhigh", "claude/claude-sonnet-5"),
50
+ ("claude/claude-sonnet-5-high", "claude", "claude-sonnet-5", "high", "adaptive",
51
+ "claude-sonnet-5-high", "claude/claude-sonnet-5"),
52
+ ("claude/claude-sonnet-5-medium", "claude", "claude-sonnet-5", "medium", "adaptive",
53
+ "claude-sonnet-5-medium", "claude/claude-sonnet-5"),
54
+ ("claude/claude-sonnet-5-low", "claude", "claude-sonnet-5", "low", "adaptive",
55
+ "claude-sonnet-5-low", "claude/claude-sonnet-5"),
56
+ ]
57
+ NEW_ALIASES = {"grok-4.7": "grok-4.6"} # new catalog model -> alias entry to clone
58
+
59
+
60
+ def half_up(value: float) -> int:
61
+ # round() is banker's rounding; a registry score must not depend on parity.
62
+ return int(Decimal(str(value)).quantize(Decimal("1"), rounding=ROUND_HALF_UP))
63
+
64
+
65
+ def dump(value: dict) -> bytes:
66
+ return (json.dumps(value, indent=2, ensure_ascii=False) + "\n").encode()
67
+
68
+
69
+ def parse_records(body: str) -> dict[str, dict]:
70
+ """The page embeds every model as escaped JSON; take each object that carries a score."""
71
+ text = body.replace('\\"', '"').replace("\\\\", "\\")
72
+ records: dict[str, dict] = {}
73
+ for match in re.finditer(r'"intelligenceIndex":', text):
74
+ start, depth = match.start(), 0
75
+ while start > 0:
76
+ char = text[start]
77
+ if char == "}":
78
+ depth += 1
79
+ elif char == "{":
80
+ if depth == 0:
81
+ break
82
+ depth -= 1
83
+ start -= 1
84
+ end, depth = match.start(), 0
85
+ while end < len(text):
86
+ char = text[end]
87
+ if char == "{":
88
+ depth += 1
89
+ elif char == "}":
90
+ if depth == 0:
91
+ break
92
+ depth -= 1
93
+ end += 1
94
+ try:
95
+ obj = json.loads(text[start:end + 1])
96
+ except ValueError:
97
+ continue
98
+ slug = obj.get("slug")
99
+ if not isinstance(slug, str):
100
+ continue
101
+ # A model appears several times, each copy carrying a different subset of
102
+ # fields; which copy comes first varies between fetches.
103
+ merged = records.setdefault(slug, {})
104
+ for key, value in obj.items():
105
+ if merged.get(key) is None:
106
+ merged[key] = value
107
+ return records
108
+
109
+
110
+ def cmd_fetch(args) -> int:
111
+ url = PAGE.format(slug=args.slug)
112
+ body = subprocess.run(["curl", "-sSL", "--fail", "-A", UA, "--max-time", "120", url],
113
+ check=True, capture_output=True, text=True).stdout
114
+ version = re.search(r"Intelligence Index v([0-9.]+)", body)
115
+ if not version:
116
+ sys.exit("fetch: the page names no Intelligence Index version")
117
+ records = parse_records(body)
118
+ kept = {}
119
+ for slug, obj in sorted(records.items()):
120
+ if not VENDOR_SLUG.match(slug) or obj.get("intelligenceIndex") is None:
121
+ continue
122
+ row = {key: obj.get(key) for key in KEEP}
123
+ effort = obj.get("effort")
124
+ row["effort_label"] = effort.get("slug") if isinstance(effort, dict) else effort
125
+ kept[slug] = row
126
+ extract = {
127
+ "source_url": url,
128
+ "fetched_at": datetime.now().astimezone().isoformat(timespec="seconds"),
129
+ "page_sha256": hashlib.sha256(body.encode()).hexdigest(),
130
+ "benchmark_version": version.group(1),
131
+ "records_on_page": len(records),
132
+ "records": kept,
133
+ }
134
+ Path(args.out).write_bytes(dump(extract))
135
+ print(f"fetch: AA v{extract['benchmark_version']}, {len(records)} records on page, "
136
+ f"{len(kept)} vendor records kept -> {args.out}")
137
+ return 0
138
+
139
+
140
+ def rescore(row: dict, record: dict, version: str, as_of: str, report: str) -> None:
141
+ row["score"] = half_up(record["intelligenceIndex"])
142
+ row["score_raw"] = round(record["intelligenceIndex"], 2)
143
+ row["estimated"] = bool(record["intelligenceIndexIsEstimated"])
144
+ row["evidence_marker"] = "estimated" if row["estimated"] else "unmarked"
145
+ row["benchmark_version"], row["as_of"], row["evidence_report"] = version, as_of, report
146
+
147
+
148
+ def cmd_build(args) -> int:
149
+ extract = json.loads(Path(args.extract).read_text())
150
+ records, version, as_of = extract["records"], extract["benchmark_version"], args.as_of
151
+ old = json.loads(Path(args.base).read_text())
152
+ new = copy.deepcopy(old)
153
+ report = lambda vendor: f"docs/reports/aa-{vendor}-evidence-{as_of}.md" # noqa: E731
154
+
155
+ scored, dropped = [], []
156
+ for row in new["scored_configs"]:
157
+ record = records.get(row["aa_slug"])
158
+ if record is None:
159
+ dropped.append(row)
160
+ continue
161
+ rescore(row, record, version, as_of, report(row["vendor"]))
162
+ scored.append(row)
163
+ by_id = {row["id"]: row for row in old["scored_configs"]}
164
+ for cid, vendor, model, effort, reasoning, slug, shape in NEW_ROWS:
165
+ if cid in {row["id"] for row in scored}:
166
+ continue
167
+ if slug not in records:
168
+ sys.exit(f"build: {slug} is not in the extract")
169
+ row = copy.deepcopy(by_id[shape])
170
+ row.update(id=cid, vendor=vendor, model=model, effort=effort, reasoning=reasoning,
171
+ fallback=None, aa_slug=slug, source_urls=[PAGE.format(slug=slug)])
172
+ row["transport_mapping"]["candidate_model_ids"] = [model]
173
+ rescore(row, records[slug], version, as_of, report(vendor))
174
+ scored.append(row)
175
+ scored.sort(key=lambda row: (-row["score"], -row["score_raw"], row["id"]))
176
+ new["scored_configs"] = scored
177
+
178
+ added = {(v, m, e, r) for _, v, m, e, r, _, _ in NEW_ROWS}
179
+ unknown = [row for row in new["unknown_configs"]
180
+ if (row["vendor"], row["model"], row["effort"], row["reasoning"]) not in added]
181
+ for row in unknown:
182
+ row["benchmark_version"], row["as_of"] = version, as_of
183
+ row["evidence_report"] = report(row["vendor"])
184
+ for row in dropped:
185
+ unknown.append({
186
+ "id": "/".join(str(row[key]) for key in ("vendor", "model", "effort", "reasoning")).replace("None", "none"),
187
+ **{key: row[key] for key in ("vendor", "model", "effort", "reasoning", "fallback")},
188
+ "score": None, "estimated": None, "benchmark_version": version, "as_of": as_of,
189
+ "status": "unknown", "authority_eligible": False,
190
+ "reason": f"AA v{version} no longer lists {row['aa_slug']}; the earlier score is not carried over",
191
+ "source_urls": [], "evidence_report": report(row["vendor"]), "mapping_status": "unknown",
192
+ "transport_mapping": {"status": "unknown", "runtime_verified": False, "resolved_config_id": None},
193
+ })
194
+ new["unknown_configs"] = unknown
195
+
196
+ for row in new["reference_configs"]:
197
+ if row["aa_slug"] in records:
198
+ rescore(row, records[row["aa_slug"]], version, as_of, report(row["vendor"]))
199
+
200
+ live_ids = {row["id"] for row in scored}
201
+ for alias in new["aliases"]:
202
+ alias["candidate_config_ids"] = [cid for cid in alias["candidate_config_ids"] if cid in live_ids]
203
+ for cid, vendor, model, *_ in NEW_ROWS:
204
+ if (alias["catalog_vendor"], alias["catalog_model"]) == (vendor, model) \
205
+ and cid not in alias["candidate_config_ids"]:
206
+ alias["candidate_config_ids"].append(cid)
207
+ # aliases mirror scripts/configure.sh's catalog, so a new alias needs the model there too.
208
+ have = {(alias["catalog_vendor"], alias["catalog_model"]) for alias in new["aliases"]}
209
+ for model, source in NEW_ALIASES.items():
210
+ template = next(a for a in new["aliases"] if a["catalog_model"] == source)
211
+ if (template["catalog_vendor"], model) not in have:
212
+ clone = copy.deepcopy(template)
213
+ clone["catalog_model"] = model
214
+ clone["candidate_config_ids"] = [cid for cid, _, m, *_ in NEW_ROWS if m == model]
215
+ new["aliases"].insert(new["aliases"].index(template), clone)
216
+
217
+ vendors: dict[str, int] = {}
218
+ for row in scored:
219
+ vendors[row["vendor"]] = vendors.get(row["vendor"], 0) + 1
220
+ new["coverage"] = {"scored_configs": len(scored), "reference_configs": len(new["reference_configs"]),
221
+ "unknown_configs": len(unknown), "aliases": len(new["aliases"]),
222
+ "by_vendor": {v: vendors[v] for v in old["coverage"]["by_vendor"]}}
223
+ new["snapshot"].update(
224
+ id=f"aa-v{version}-{as_of}-v1", benchmark_version=version, as_of=as_of,
225
+ source={"extract": str(Path(args.extract).resolve().relative_to(REPO)),
226
+ "page_url": extract["source_url"], "page_sha256": extract["page_sha256"],
227
+ "fetched_at": extract["fetched_at"]},
228
+ approval={"status": args.approval, "scope": f"aa-v{version}-rebaseline",
229
+ "source": f"docs/reports/aa-rebaseline-{as_of}.md",
230
+ "estimated_scores": "approved_provisional" if args.approval == "approved"
231
+ else "provisional_pending_review"})
232
+ new.setdefault("schema_notes", {})["score_rounding"] = (
233
+ "score is score_raw rounded half-up to an integer; score_raw is the AA index to two decimals")
234
+ REGISTRY.write_bytes(dump(new))
235
+ print(f"build: {len(scored)} scored ({sum(r['estimated'] for r in scored)} estimated), "
236
+ f"{len(unknown)} unknown, dropped {[r['id'] for r in dropped]}")
237
+ print(f"build: sha256 {hashlib.sha256(REGISTRY.read_bytes()).hexdigest()}")
238
+ return 0
239
+
240
+
241
+ def cmd_report(args) -> int:
242
+ old = {row["id"]: row for row in json.loads(Path(args.old).read_text())["scored_configs"]}
243
+ new = json.loads(REGISTRY.read_text())
244
+ as_of, version = new["snapshot"]["as_of"], new["snapshot"]["benchmark_version"]
245
+ out = REPO / "docs/reports"
246
+ out.mkdir(parents=True, exist_ok=True)
247
+ for vendor in new["coverage"]["by_vendor"]:
248
+ lines = [f"# AA v{version} evidence: {vendor} ({as_of})", "",
249
+ f"Generated by `scripts/aa_rebaseline.py report` from `{new['snapshot']['source']['extract']}` "
250
+ f"(page sha256 `{new['snapshot']['source']['page_sha256'][:16]}…`).", "",
251
+ "| config | effort | raw | score | estimated | previous | source |", "|---|---|---|---|---|---|---|"]
252
+ for row in new["scored_configs"]:
253
+ if row["vendor"] != vendor:
254
+ continue
255
+ before = old.get(row["id"])
256
+ lines.append(f"| {row['id']} | {row['effort']} | {row['score_raw']} | {row['score']} | "
257
+ f"{'yes' if row['estimated'] else 'no'} | "
258
+ f"{before['score'] if before else 'new'} | {row['source_urls'][0]} |")
259
+ gone = [cid for cid, row in old.items() if row["vendor"] == vendor
260
+ and cid not in {r["id"] for r in new["scored_configs"]}]
261
+ if gone:
262
+ lines += ["", "No longer listed by AA, moved to unknown_configs: " + ", ".join(gone)]
263
+ (out / f"aa-{vendor}-evidence-{as_of}.md").write_text("\n".join(lines) + "\n")
264
+ print(f"report: wrote {len(new['coverage']['by_vendor'])} vendor reports under {out}")
265
+ return 0
266
+
267
+
268
+ def lane_table(path: Path) -> dict[str, list[tuple[str, str, str | None]]]:
269
+ table = {}
270
+ for line in path.read_text().splitlines():
271
+ match = re.match(r"^([a-z][a-z-]*):\s*(.+)$", line.split("#")[0].rstrip())
272
+ if not match or match.group(2).split()[0] in ("off", "vote"):
273
+ continue
274
+ chain = []
275
+ for segment in match.group(2).split("|"):
276
+ parts = segment.split()
277
+ if len(parts) >= 2:
278
+ chain.append((parts[0], parts[1], parts[2] if len(parts) > 2 and parts[2] != "-" else None))
279
+ table[match.group(1)] = chain
280
+ return table
281
+
282
+
283
+ def target_row(rows: list[dict], vendor: str, model: str, effort: str | None) -> dict | None:
284
+ hits = [row for row in rows if row["vendor"] == vendor and row["reasoning"] != "non-reasoning"
285
+ and (f"{row['model']}-{row['effort']}" == model if vendor == "gemini"
286
+ else row["model"] == model and row["effort"] == effort)]
287
+ return hits[0] if len(hits) == 1 else None
288
+
289
+
290
+ def cmd_matrix(args) -> int:
291
+ """Score-only view: it ignores transport verification, which is per host."""
292
+ rows = json.loads(Path(args.registry).read_text())["scored_configs"]
293
+ by_id = {row["id"]: row for row in rows}
294
+ table = lane_table(Path(args.routing))
295
+ print("| controller (ceiling) | " + " | ".join(table) + " |")
296
+ print("|---|" + "---|" * len(table))
297
+ for cid in args.controller:
298
+ ceiling = by_id[cid]["score"]
299
+ cells = []
300
+ for chain in table.values():
301
+ cell = "none"
302
+ for index, (vendor, model, effort) in enumerate(chain):
303
+ row = target_row(rows, vendor, model, effort)
304
+ if row and row["score"] <= ceiling:
305
+ cell = f"{row['id'].split('/')[1]} {row['score']}" + (f" (#{index + 1})" if index else "")
306
+ break
307
+ cells.append(cell)
308
+ print(f"| {cid.split('/')[1]} ({ceiling}) | " + " | ".join(cells) + " |")
309
+ return 0
310
+
311
+
312
+ def _nested(*path):
313
+ def read(record):
314
+ for key in path:
315
+ record = (record or {}).get(key)
316
+ return record
317
+ return read
318
+
319
+
320
+ MEASURES = { # column title -> (reader, decimals, scale)
321
+ "index": (_nested("intelligenceIndex"), 1, 1),
322
+ "Terminal-Bench 4.0": (_nested("terminalBench40"), 3, 1),
323
+ "Terminal-Bench 2.1": (_nested("terminalBench21"), 3, 1),
324
+ "SciCode": (_nested("scicode"), 3, 1),
325
+ "hallucination rate": (_nested("omniscienceBreakdown", "hallucinationRate"), 3, 1),
326
+ "knowledge (omniscience)": (_nested("omniscience"), 1, 1),
327
+ "HLE": (_nested("hle"), 3, 1),
328
+ "GPQA": (_nested("gpqa"), 3, 1),
329
+ "CritPt": (_nested("critpt"), 3, 1),
330
+ "Briefcase analytical Elo": (_nested("briefcaseBreakdown", "analyticalQuality", "elo"), 0, 1),
331
+ "Briefcase overall Elo": (_nested("briefcaseBreakdown", "overall", "elo"), 0, 1),
332
+ "Briefcase presentation Elo": (_nested("briefcaseBreakdown", "presentation", "elo"), 0, 1),
333
+ "GDPval": (_nested("gdpvalNormalized"), 3, 1),
334
+ "AutomationBench": (_nested("automationBenchPartialScore"), 3, 1),
335
+ "MMMU-Pro": (_nested("mmmuPro"), 3, 1),
336
+ "mlcrOverall": (_nested("mlcrOverall"), 3, 1),
337
+ "AA-LCR": (_nested("lcr"), 3, 1),
338
+ "minutes / task": (_nested("intelligenceIndexTimePerTask"), 1, 1 / 60),
339
+ "first answer token (s)": (_nested("timeToFirstAnswerToken", "total"), 0, 1),
340
+ "index run cost ($)": (_nested("intelligenceIndexCost", "total"), 0, 1),
341
+ }
342
+ LANE_MEASURES = {
343
+ "hardest-coding": ("Terminal-Bench 4.0", "Terminal-Bench 2.1", "SciCode", "hallucination rate"),
344
+ "bulk-mechanical": ("Terminal-Bench 4.0", "Terminal-Bench 2.1", "hallucination rate", "minutes / task",
345
+ "index run cost ($)"),
346
+ "triage": ("index", "index run cost ($)", "minutes / task"),
347
+ "hard-judgment": ("HLE", "GPQA", "CritPt", "Briefcase analytical Elo", "hallucination rate"),
348
+ "taste-final": ("Briefcase overall Elo", "Briefcase presentation Elo", "GDPval"),
349
+ "consult": ("index",),
350
+ "ui-draft": ("MMMU-Pro", "Terminal-Bench 4.0"),
351
+ "long-context": ("mlcrOverall", "AA-LCR", "index run cost ($)"),
352
+ "fast-agentic": ("AutomationBench", "minutes / task", "first answer token (s)"),
353
+ "live-search": ("hallucination rate", "knowledge (omniscience)"),
354
+ "coding-overflow": ("Terminal-Bench 4.0", "SciCode"),
355
+ }
356
+
357
+
358
+ def cmd_lanes(args) -> int:
359
+ """Per lane, every candidate with the measurements that lane is ordered on."""
360
+ records = json.loads(Path(args.extract).read_text())["records"]
361
+ rows = json.loads(Path(args.registry).read_text())["scored_configs"]
362
+ for lane, chain in lane_table(Path(args.routing)).items():
363
+ titles = LANE_MEASURES.get(lane, ("index",))
364
+ print(f"\n**{lane}**\n\n| # | candidate | score | " + " | ".join(titles) + " |")
365
+ print("|---|---|---|" + "---|" * len(titles))
366
+ for index, (vendor, model, effort) in enumerate(chain, 1):
367
+ row = target_row(rows, vendor, model, effort)
368
+ name = f"{vendor} {model}" + (f" {effort}" if effort else "")
369
+ if row is None:
370
+ print(f"| {index} | {name} | not scored | " + " | ".join("—" for _ in titles) + " |")
371
+ continue
372
+ cells = []
373
+ for title in titles:
374
+ reader, places, scale = MEASURES[title]
375
+ value = reader(records.get(row["aa_slug"]))
376
+ cells.append("not published" if value is None else f"{value * scale:.{places}f}")
377
+ print(f"| {index} | {name} | {row['score']} | " + " | ".join(cells) + " |")
378
+ return 0
379
+
380
+
381
+ def main() -> int:
382
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
383
+ sub = parser.add_subparsers(dest="command", required=True)
384
+ fetch = sub.add_parser("fetch")
385
+ fetch.add_argument("--slug", default="grok-4-7")
386
+ fetch.add_argument("--out", required=True)
387
+ build = sub.add_parser("build")
388
+ build.add_argument("--extract", required=True)
389
+ build.add_argument("--as-of", required=True)
390
+ build.add_argument("--base", default=str(REGISTRY),
391
+ help="the registry to re-score; pass the previous snapshot to rebuild from scratch")
392
+ build.add_argument("--approval", default="proposed", choices=("proposed", "approved"))
393
+ report = sub.add_parser("report")
394
+ report.add_argument("--old", required=True, help="the previous registry file")
395
+ matrix = sub.add_parser("matrix")
396
+ matrix.add_argument("--registry", default=str(REGISTRY))
397
+ matrix.add_argument("--routing", default=str(REPO / "routing.yaml"))
398
+ matrix.add_argument("--controller", nargs="+", required=True, help="registry config ids")
399
+ lanes = sub.add_parser("lanes")
400
+ lanes.add_argument("--extract", required=True)
401
+ lanes.add_argument("--registry", default=str(REGISTRY))
402
+ lanes.add_argument("--routing", default=str(REPO / "routing.yaml"))
403
+ args = parser.parse_args()
404
+ return {"fetch": cmd_fetch, "build": cmd_build, "report": cmd_report, "matrix": cmd_matrix,
405
+ "lanes": cmd_lanes}[args.command](args)
406
+
407
+
408
+ if __name__ == "__main__":
409
+ sys.exit(main())
@@ -154,7 +154,7 @@ CODEX_EFFORTS=("xhigh" "max" "ultra" "high" "medium" "low" "minimal" "none")
154
154
  CLAUDE_MODELS=("default" "best" "fable" "opus" "sonnet" "haiku" "opus[1m]" "sonnet[1m]" "opusplan" "claude-fable-5" "claude-fable-5-1" "claude-opus-5" "claude-sonnet-5" "claude-opus-4-8" "claude-opus-4-7" "claude-opus-4-6" "claude-opus-4-5-20251101" "claude-sonnet-4-6" "claude-sonnet-4-5-20250929" "claude-haiku-4-5" "claude-haiku-4-5-20251001")
155
155
  CLAUDE_EFFORTS=("max" "xhigh" "high" "medium" "low" "-")
156
156
  GEMINI_MODELS=("gemini-3.8-flash-high" "gemini-3.8-flash-medium" "gemini-3.8-flash-low" "gemini-3.7-flash-high" "gemini-3.7-flash-medium" "gemini-3.7-flash-low" "gemini-3.6-flash-high" "gemini-3.6-flash-medium" "gemini-3.6-flash-low" "gemini-3.1-pro-high" "gemini-3.1-pro-low" "claude-sonnet-4-6" "claude-opus-4-6-thinking" "gpt-oss-120b-medium")
157
- GROK_MODELS=("grok-4.6" "headroom-grok-build" "grok-4.3-official")
157
+ GROK_MODELS=("grok-4.7" "grok-4.6" "headroom-grok-build" "grok-4.3-official")
158
158
  KIMI_MODELS=("kimi-k3" "kimi-k2.7-code" "kimi-k2.5")
159
159
  QWEN_MODELS=("qwen3.7-max" "qwen3.7-plus" "qwen3.6-plus" "qwen3.5-plus" "qwen3-max-2026-01-23" "qwen3-coder-next" "qwen3-coder-plus" "qwen3-coder-flash")
160
160
  # OpenCode models use provider/model form; OpenRouter models use catalog slugs.
@@ -23,9 +23,12 @@ from typing import Any
23
23
 
24
24
 
25
25
  MAX_BYTES = 1_048_576
26
- # Approval anchor for the frozen AA v4.2 2026-09-07 source bytes. Updating this
27
- # constant is a governance change, never a caller argument/environment override.
28
- APPROVED_REGISTRY_SHA256 = "0782c87de123c02738c3ff60e4bc3c1cc10d110113e872b8f8627212861cdaab"
26
+ # Approval anchor for the frozen AA source bytes. Updating these constants is a
27
+ # governance change, never a caller argument/environment override.
28
+ # scripts/aa_rebaseline.py regenerates the registry; it never touches this pin.
29
+ APPROVED_BENCHMARK_VERSION = "4.3.2"
30
+ APPROVED_AS_OF = "2026-09-22"
31
+ APPROVED_REGISTRY_SHA256 = "a1109913b9928d943bc440787d26caaf7818e40abdaae32899750d65e5eb5fdd"
29
32
 
30
33
  IDENTITY_FIELDS = ("vendor", "model", "effort", "reasoning", "fallback")
31
34
  TRANSPORT_EVIDENCE_VENDORS = frozenset(("codex", "claude", "grok", "gemini"))
@@ -109,10 +112,10 @@ def _validate_registry(value: dict[str, Any]) -> dict[str, Any]:
109
112
  _check(isinstance(snapshot, dict), "invalid AA registry snapshot")
110
113
  for key in ("id", "benchmark_version", "as_of", "frozen"):
111
114
  _check(key in snapshot, "incomplete AA registry snapshot")
112
- _check(snapshot["benchmark_version"] == "4.2"
113
- and snapshot["as_of"] == "2026-09-07"
115
+ _check(snapshot["benchmark_version"] == APPROVED_BENCHMARK_VERSION
116
+ and snapshot["as_of"] == APPROVED_AS_OF
114
117
  and snapshot["frozen"] is True,
115
- "AA registry is not frozen v4.2 dated 2026-09-07")
118
+ f"AA registry is not frozen v{APPROVED_BENCHMARK_VERSION} dated {APPROVED_AS_OF}")
116
119
  policy = value["policy"]
117
120
  _check(isinstance(policy, dict)
118
121
  and policy.get("decision") == "target_score <= min(caller_score, inherited_ceiling)"
@@ -135,7 +138,8 @@ def _validate_registry(value: dict[str, Any]) -> dict[str, Any]:
135
138
  _check(type(row["score"]) is int and 0 <= row["score"] <= 100,
136
139
  "invalid AA score")
137
140
  _check(type(row["estimated"]) is bool, "invalid AA estimated flag")
138
- _check(row["benchmark_version"] == "4.2" and row["as_of"] == "2026-09-07",
141
+ _check(row["benchmark_version"] == APPROVED_BENCHMARK_VERSION
142
+ and row["as_of"] == APPROVED_AS_OF,
139
143
  "mixed AA registry snapshot")
140
144
  _check(isinstance(row["transport_mapping"], dict), "invalid transport mapping")
141
145
  return value
@@ -49,6 +49,9 @@ PROVEN["grok/grok-4-6"] = ("cli_reasoning_effort", "grok-4.6", "gk-grok-4_6-high
49
49
  for effort in ["xhigh", "medium", "low"]:
50
50
  PROVEN[f"grok/grok-4-6-{effort}"] = ("cli_reasoning_effort", "grok-4.6", f"gk-grok-4_6-{effort}")
51
51
  PROVEN["grok/grok-4-5"] = ("cli_reasoning_effort", "grok-4.5", "gk-grok-4_5-high")
52
+ # AA scores grok-4.7 at xhigh and high only; its base row is the xhigh one.
53
+ PROVEN["grok/grok-4-7"] = ("cli_reasoning_effort", "grok-4.7", "gk-grok-4_7-xhigh")
54
+ PROVEN["grok/grok-4-7-high"] = ("cli_reasoning_effort", "grok-4.7", "gk-grok-4_7-high")
52
55
 
53
56
  for cid, rid, ev in [
54
57
  ("gemini/gemini-3-8-flash", "gemini-3.8-flash-high", "agy-gemini-3_8-flash-high"),
@@ -81,6 +84,12 @@ for cid, model in [("claude/claude-sonnet-5", "claude-sonnet-5"),
81
84
  ("claude/claude-opus-4-6-adaptive", "claude-opus-4-6"),
82
85
  ("claude/claude-sonnet-4-6-adaptive", "claude-sonnet-4-6")]:
83
86
  PROVEN[cid] = ("model_and_effort", model, f"cl-{model}-max")
87
+ # Targets resolve on vendor/model/effort alone, so claude-sonnet-5-non-reasoning
88
+ # (also effort high) must never be listed here: two verified rows at one
89
+ # selector make every sonnet-5 high dispatch ambiguous-runtime-mapping.
90
+ for effort in ["xhigh", "high", "medium", "low"]:
91
+ PROVEN[f"claude/claude-sonnet-5-{effort}"] = (
92
+ "model_and_effort", "claude-sonnet-5", f"cl-claude-sonnet-5-{effort}")
84
93
 
85
94
  # Fable is listed so its failures reach unproven[] rather than vanishing. Its
86
95
  # probes were refused for quota on 2026-09-07 and again on 2026-09-09; the
@@ -92,11 +92,22 @@ def _text(record: dict) -> str:
92
92
 
93
93
 
94
94
  def sweep(vendor: str, root: Path, *, repo: Path = build_overlay.REPO, home: Path | None = None,
95
- run_probe=probe_module.probe, manager=session_manager, log=print) -> dict:
96
- """Probe one vendor into root/evidence. Never raises for a probe that merely failed."""
95
+ run_probe=probe_module.probe, manager=session_manager, log=print,
96
+ only_missing: bool = False) -> dict:
97
+ """Probe one vendor into root/evidence. Never raises for a probe that merely failed.
98
+
99
+ only_missing probes just the rows root/evidence has no record for: the
100
+ executable is unchanged, so what it already answered still stands.
101
+ """
97
102
  home = home or Path.home()
98
103
  entries = plan(vendor)
99
104
  report = {"vendor": vendor, "outcome": "done", "passed": [], "failed": [], "detail": ""}
105
+ if only_missing:
106
+ entries = [entry for entry in entries
107
+ if not (root / "evidence" / f"{entry['name']}.json").is_file()]
108
+ if not entries:
109
+ report["detail"] = "every row already has probe evidence"
110
+ return report
100
111
  if (vendor in KEYCHAIN_VENDORS and sys.platform == "darwin"
101
112
  and os.environ.get("OMNILANE_PROBE_ANY_SESSION") != "1"):
102
113
  name = manager()
@@ -148,7 +159,9 @@ def sweep(vendor: str, root: Path, *, repo: Path = build_overlay.REPO, home: Pat
148
159
  return report
149
160
  report["failed"].append(entry["config_id"])
150
161
  log(f"{entry['name']} fail {str(record.get('verdict_reason'))[:120]}")
151
- if not report["passed"]:
162
+ # With only_missing the rows not re-probed still hold, so a new row that
163
+ # fails is a finding about that row, not about the vendor.
164
+ if not report["passed"] and not only_missing:
152
165
  report.update(outcome="failed", detail="no selector of this vendor passed")
153
166
  return report
154
167
 
@@ -61,6 +61,7 @@ def current_anchors() -> dict[str, dict[str, Path]]:
61
61
  def detect(overlay: dict, anchors: dict[str, dict[str, Path]]) -> dict[str, dict]:
62
62
  """Per vendor: what no longer matches the live overlay, and the signer it recorded."""
63
63
  report = {}
64
+ registry_snapshot = build_overlay.REGISTRY["snapshot"]["id"]
64
65
  for vendor in VENDORS:
65
66
  recorded = [entry for entry in overlay.get("evidence", []) if entry.get("vendor") == vendor]
66
67
  runner_name = build_overlay.RUNNERS[vendor]
@@ -81,11 +82,19 @@ def detect(overlay: dict, anchors: dict[str, dict[str, Path]]) -> dict[str, dict
81
82
  reasons.append(f"{runner_name} is not pinned")
82
83
  elif sha256(runner) != recorded_runner["sha256"]:
83
84
  reasons.append(f"{runner_name} changed")
85
+ cli_changed = any(runner_name not in reason for reason in reasons)
86
+ # The overlay is bound to one registry snapshot and dispatch refuses any
87
+ # other, so a re-scored registry strands a host whose CLIs never moved.
88
+ snapshot_only = not reasons and overlay.get("snapshot_id") != registry_snapshot
89
+ if overlay.get("snapshot_id") != registry_snapshot:
90
+ reasons.append(f"the score registry moved from {overlay.get('snapshot_id')} "
91
+ f"to {registry_snapshot}")
84
92
  report[vendor] = {
85
93
  "drifted": bool(reasons),
86
94
  "reasons": reasons,
95
+ "snapshot_only": snapshot_only,
87
96
  "cli": str(cli) if cli else None,
88
- "cli_changed": any(runner_name not in reason for reason in reasons),
97
+ "cli_changed": cli_changed,
89
98
  "recorded_cli_path": recorded_cli["path"] if recorded_cli else None,
90
99
  "recorded_codesign": recorded_signer(recorded_cli),
91
100
  }
@@ -105,6 +114,8 @@ def recorded_signer(entry: dict | None) -> dict | None:
105
114
  def gate(vendor_report: dict, approved: bool) -> tuple[bool, str]:
106
115
  if not vendor_report["cli"]:
107
116
  return False, "the CLI is not installed"
117
+ if vendor_report.get("snapshot_only"):
118
+ return True, "only the score registry changed; existing probe evidence is reused"
108
119
  if not vendor_report["cli_changed"]:
109
120
  return True, "only omnilane's own runner script changed"
110
121
  current = cli_provenance.facts(vendor_report["cli"])
@@ -339,7 +350,9 @@ def resign(args, log=print) -> int:
339
350
  root.mkdir(parents=True)
340
351
  shutil.copytree(source / "evidence", root / "evidence")
341
352
  shutil.copy2(live, root / "live-overlay.BEFORE.json")
342
- sweeps = {vendor: probe_sweep.sweep(vendor, root, log=log) for vendor in proceed}
353
+ sweeps = {vendor: probe_sweep.sweep(vendor, root, log=log,
354
+ only_missing=report[vendor]["snapshot_only"])
355
+ for vendor in proceed}
343
356
  summary["sweeps"] = sweeps
344
357
  unfinished = [vendor for vendor, result in sweeps.items() if result["outcome"] != "done"]
345
358
  for vendor in unfinished:
@@ -393,7 +406,12 @@ def resign(args, log=print) -> int:
393
406
  staged["evidence"][index] = old
394
407
  staged_path.write_text(json.dumps(staged, indent=2, ensure_ascii=False) + "\n")
395
408
  try:
396
- before, after = verified(live), verified(staged_path)
409
+ if overlay.get("snapshot_id") == staged["snapshot_id"]:
410
+ before = verified(live)
411
+ else:
412
+ # Bound to another registry snapshot, the live overlay does not load at all.
413
+ before = {vendor: 0 for vendor in VENDORS}
414
+ after = verified(staged_path)
397
415
  except (aa_policy.PolicyError, OSError, ValueError) as error:
398
416
  log(f"omnilane: the staged overlay does not load ({error}); live overlay untouched")
399
417
  return EXIT_ROLLED_BACK