clearotron 0.3.3-beta.0 → 0.3.3-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/onboard.mjs +80 -6
- package/build-info.json +2 -2
- package/docs/architecture/04-configuration-reference.md +28 -10
- package/driver/CHANGELOG.md +12 -0
- package/driver/citation-census.json +3 -3
- package/driver/config-inventory.mjs +1 -1
- package/driver/dev-portal.mjs +3 -3
- package/driver/driver.config.mjs +80 -9
- package/driver/engine/CONTRACT.md +3 -2
- package/driver/engine/anthropic-agent.mjs +34 -7
- package/driver/engine/mcp/probe-server.mjs +12 -3
- package/driver/package.json +1 -1
- package/driver/pipeline-knockout.mjs +3 -3
- package/driver/pipeline.mjs +14 -14
- package/driver/plan-run-agreement-verdict.mjs +49 -0
- package/driver/portal-service.mjs +8 -4
- package/driver/progress.mjs +14 -3
- package/driver/publish/index.mjs +1 -1
- package/driver/publish/report-data.mjs +4 -3
- package/driver/reference-score.mjs +10 -2
- package/driver/settle-stamp.mjs +10 -3
- package/driver/status-snapshot.mjs +2 -2
- package/driver/suite-census.json +61 -19
- package/mcp-server/CHANGELOG.md +4 -0
- package/mcp-server/lib/brief.mjs +11 -5
- package/mcp-server/package.json +1 -1
- package/package.json +2 -2
- package/portal-ui/package.json +1 -1
- package/providers/oauth-mcp-bridge/CHANGELOG.md +4 -0
- package/providers/oauth-mcp-bridge/package.json +1 -1
- package/scripts/e2e.mjs +1 -1
- package/scripts/live-surface-check.mjs +6 -14
- package/scripts/repo-writes.mjs +1 -1
- package/scripts/report-sections-render-check.mjs +7 -3
- package/scripts/score.mjs +7 -1
- package/shared/scroll-settle.mjs +67 -0
package/bin/onboard.mjs
CHANGED
|
@@ -98,7 +98,7 @@ import {
|
|
|
98
98
|
// REGISTER_PROVIDER frozen at first import) is the one `preflightCandidate` below already cache-busts
|
|
99
99
|
// around, and it is cache-busted whether or not this static import happened first.
|
|
100
100
|
import { config, ENGINE_BINARIES, DEFAULT_ENGINE_ID, RESEARCH_PROVIDERS, SERP_PROVIDERS, resolveEngineProgram, ON_A_WINDOWS_DRIVE,
|
|
101
|
-
enginesFolder, engineInstallArgs, engineInstallCommand } from "../driver/driver.config.mjs";
|
|
101
|
+
enginesFolder, engineInstallArgs, engineInstallCommand, olderThanFloor } from "../driver/driver.config.mjs";
|
|
102
102
|
import { resolveAuthMode, CLOUD_SWITCH, CLOUD_SETTINGS, CLOUD_SECRETS, cloudsSwitchedOn } from "../driver/engine/auth.mjs";
|
|
103
103
|
import { isInsideCheckout } from "../shared/inside-checkout.mjs"; // — one copy of the rule, and it is testable
|
|
104
104
|
import { packagedBuild as sharedPackagedBuild } from "../shared/packaged-build.mjs"; // — one reader of build-info.json, reachable from the driver
|
|
@@ -1105,6 +1105,25 @@ function programProblem(bin, named) {
|
|
|
1105
1105
|
return null;
|
|
1106
1106
|
}
|
|
1107
1107
|
|
|
1108
|
+
// WHAT A COPY BELOW THE FLOOR DOES TO THE MODELS WAS MEASURED ON CLAUDE ALONE (2026-09-22): it refuses the
|
|
1109
|
+
// newest model of a tier and serves the one before. Codex's floor records no such measurement, so an old
|
|
1110
|
+
// Codex is named by its version, the floor and the fix, and nothing is said about which model it runs.
|
|
1111
|
+
export const floorRefusesNewestModel = (eng) => Boolean(eng?.package) && eng.package === ENGINE_BINARIES["anthropic-agent"]?.package;
|
|
1112
|
+
|
|
1113
|
+
/**
|
|
1114
|
+
* What doctor and setup say after the version of a copy below the floor: what it costs, where that was
|
|
1115
|
+
* measured, and the fix. ONE COPY FOR BOTH SCREENS, so the two cannot drift apart again (ruling 285).
|
|
1116
|
+
*
|
|
1117
|
+
* The fix follows where the copy came from. A copy Clearotron installed moves with `update`. Any other
|
|
1118
|
+
* copy wins over one setup installs, by design, so installing another does not replace it: the fix is
|
|
1119
|
+
* to update it, or to name a newer one in the engine's setting.
|
|
1120
|
+
*/
|
|
1121
|
+
export function belowFloorTail(eng, bin) {
|
|
1122
|
+
const cost = floorRefusesNewestModel(eng) ? `The newest ${eng.product} models need a newer version. Searches run on an older model instead, or stop at the first step if they name the newest one exactly. ` : "";
|
|
1123
|
+
const fix = bin?.source === "installed" ? `Update it with \`${invoke("update")}\`.` : `Update it, or set ${eng.env} to a newer copy.`;
|
|
1124
|
+
return cost + fix;
|
|
1125
|
+
}
|
|
1126
|
+
|
|
1108
1127
|
/**
|
|
1109
1128
|
* What setup found of one engine's program, as its menu row says it; "" when nothing was looked for. A copy
|
|
1110
1129
|
* that cannot run is a problem, not an absence, and installing another is not always the fix, so the row
|
|
@@ -1113,7 +1132,14 @@ function programProblem(bin, named) {
|
|
|
1113
1132
|
*/
|
|
1114
1133
|
export function foundWords(eng, bin) {
|
|
1115
1134
|
if (!bin) return "";
|
|
1116
|
-
if (bin.executable && !bin.relative)
|
|
1135
|
+
if (bin.executable && !bin.relative) {
|
|
1136
|
+
// A COPY THAT RUNS IS NOT NECESSARILY ONE THIS BUILD CAN USE. Below the floor the program refuses
|
|
1137
|
+
// the newest model of a tier and serves the one before it, so the row cannot say "found" and leave
|
|
1138
|
+
// it there; it says which problem, and choosing the engine shows the fix, as the other problems do.
|
|
1139
|
+
if (olderThanFloor(bin.version, eng.floor) === true)
|
|
1140
|
+
return `problem: the copy of ${eng.product} here is version ${bin.version}; Clearotron needs ${eng.floor} or newer — choose it to see the fix`;
|
|
1141
|
+
return bin.version ? `found on this computer (version ${bin.version})` : "found on this computer";
|
|
1142
|
+
}
|
|
1117
1143
|
const p = programProblem(bin, bin.explicit);
|
|
1118
1144
|
if (p?.kind === "incomplete") return `problem: the copy of ${eng.product} here is incomplete and won't run — choose it to see the fix`;
|
|
1119
1145
|
if (p?.kind === "setting") return `problem: this computer is set to use a copy of ${eng.product} that isn't there — choose it to see the fix`;
|
|
@@ -1129,6 +1155,12 @@ export function foundWords(eng, bin) {
|
|
|
1129
1155
|
*/
|
|
1130
1156
|
export function cannotRunLine(eng, bin, setting = "") {
|
|
1131
1157
|
const set = namedSetting(eng, setting);
|
|
1158
|
+
// ANSWERED BEFORE THE REST, because a copy below the floor RUNS: it is executable, nothing rejected
|
|
1159
|
+
// it, and every clause below is about a copy that cannot start. Left to them it would fall through to
|
|
1160
|
+
// the general clause and be described as unusable, which sends the reader to look for a broken
|
|
1161
|
+
// install they do not have. The fix here is a version, not a repair.
|
|
1162
|
+
if (bin?.executable && !bin?.relative && olderThanFloor(bin.version, eng.floor) === true)
|
|
1163
|
+
return `${eng.product} on this computer is version ${bin.version}. Clearotron needs ${eng.floor} or newer. ${belowFloorTail(eng, bin)}`;
|
|
1132
1164
|
const p = programProblem(bin, set);
|
|
1133
1165
|
if (p?.kind === "incomplete") return `The copy of ${eng.product} at ${p.path} is incomplete: its installation stopped before the program was added. Setup can install a working copy.`;
|
|
1134
1166
|
if (p?.kind === "setting") return `This computer is set to use ${eng.product} at ${set}, and nothing there can run. Setup can install ${eng.product} and use that instead.`;
|
|
@@ -1761,10 +1793,26 @@ export async function runCheck() {
|
|
|
1761
1793
|
const bin = resolveEngineBin(binSetting, { engine: engineId });
|
|
1762
1794
|
// WHICH COPY, AND ITS VERSION, because the machine's own install and the one Clearotron installed
|
|
1763
1795
|
// are both legitimate and behave differently: the first updates itself, the second moves with
|
|
1764
|
-
// `clearotron update`. The version comes from the copy's own package.json when npm installed it
|
|
1765
|
-
//
|
|
1796
|
+
// `clearotron update`. The version comes from the copy's own package.json when npm installed it.
|
|
1797
|
+
//
|
|
1798
|
+
// AND IS ASKED FOR WHEN THERE IS NONE TO READ, which is the ordinary case for a copy the machine
|
|
1799
|
+
// installed by another route — a vendor's native installer leaves no package.json. Doctor used to
|
|
1800
|
+
// stop there and print "version not read", which was honest and made the floor comparison below
|
|
1801
|
+
// unanswerable on the route most machines take: this command could not tell an operator whether
|
|
1802
|
+
// their own copy was new enough for the models a run asks for, which is the whole of what it was
|
|
1803
|
+
// asked to check.
|
|
1804
|
+
//
|
|
1805
|
+
// It is the same short call setup's engine menu already makes (menuVersion): `--version`, two
|
|
1806
|
+
// seconds, no session and no network. That is within "calls nobody" — what that promises is no
|
|
1807
|
+
// provider call and no spend, not that nothing on this machine may be asked its own version.
|
|
1808
|
+
const versionOf = (b) => {
|
|
1809
|
+
if (b.version) return b.version;
|
|
1810
|
+
if (!b.executable || b.relative || !b.path) return null;
|
|
1811
|
+
try { return menuVersion(b.path) ?? null; } catch { return null; }
|
|
1812
|
+
};
|
|
1813
|
+
const seen = versionOf(bin);
|
|
1766
1814
|
const copyWords = (b) => `${b.source === "installed" ? "the copy Clearotron installed"
|
|
1767
|
-
: b.source === "explicit" ? `set in ${engSpec.env}` : "on PATH"}${
|
|
1815
|
+
: b.source === "explicit" ? `set in ${engSpec.env}` : "on PATH"}${seen ? `, version ${seen}` : ", version not read"}`;
|
|
1768
1816
|
// ── NATIVE WINDOWS IS ANSWERED HERE, BEFORE ANY PATH IS RESOLVED OR REPORTED ──────────────────
|
|
1769
1817
|
//
|
|
1770
1818
|
// `resolveEngineBin` tests a candidate with `accessSync(X_OK)` and `isFile()`. Windows has no
|
|
@@ -1786,7 +1834,24 @@ export async function runCheck() {
|
|
|
1786
1834
|
// and needs no engine, which is why four reports published on that same Windows box.
|
|
1787
1835
|
const platformRefusal = platformEngineRefusal();
|
|
1788
1836
|
if (platformRefusal) problem(platformRefusal);
|
|
1789
|
-
|
|
1837
|
+
// ── AND WHETHER THAT COPY IS NEW ENOUGH FOR THE MODELS IT WILL BE ASKED FOR ──────────────────
|
|
1838
|
+
//
|
|
1839
|
+
// The floor governed the copy setup INSTALLS and nothing else, and a copy already on the machine
|
|
1840
|
+
// wins over that one — so the route most machines actually take was the unchecked one. A program
|
|
1841
|
+
// below the floor does not fail: it refuses the newest model of a tier and serves the previous one,
|
|
1842
|
+
// and the search still finishes and the report still arrives. This is the cheap place to learn it.
|
|
1843
|
+
//
|
|
1844
|
+
// Three outcomes, because the comparison is three-valued: older, not older, and could not be
|
|
1845
|
+
// compared. The last is said rather than passed over — doctor's own contract is that a failure to
|
|
1846
|
+
// look is not a clean result — and it is the ordinary state for a copy the machine installed by
|
|
1847
|
+
// another route, where there is no package.json to read and doctor spawns nothing to ask.
|
|
1848
|
+
else if (bin.executable && !bin.relative && olderThanFloor(seen, engSpec.floor) === true)
|
|
1849
|
+
problem(`${bin.path} — ${copyWords(bin)}. Clearotron needs ${engSpec.floor} or newer. ${belowFloorTail(engSpec, bin)}`);
|
|
1850
|
+
else if (bin.executable && !bin.relative) {
|
|
1851
|
+
ok(`${bin.path} — ${copyWords(bin)}`);
|
|
1852
|
+
if (olderThanFloor(seen, engSpec.floor) === null)
|
|
1853
|
+
info(` Its version could not be checked against the ${engSpec.floor} Clearotron needs.`);
|
|
1854
|
+
}
|
|
1790
1855
|
// A copy that is there and cannot run is a broken install, not an absence: the vendor's placeholder
|
|
1791
1856
|
// left by an install that skipped its step, most often. Named with the reason and the fix.
|
|
1792
1857
|
else if (!binSet && bin.rejected?.length) problem(`no usable \`${engSpec.fallback}\`: ${bin.rejected.map((x) => `${x.path} is ${x.why}`).join("; ")}`);
|
|
@@ -3914,6 +3979,15 @@ try {
|
|
|
3914
3979
|
if (!bin.executable) { problem(`${bin.path ?? resolve(p)} is ${bin.rejected?.[0]?.why ?? "not an executable file"}.`); continue; }
|
|
3915
3980
|
}
|
|
3916
3981
|
ok(`found ${bin.path}${bin.source === "installed" ? `, the copy Clearotron installed${bin.version ? ` (${bin.version})` : ""}` : ""}`);
|
|
3982
|
+
// THE MENU ROW SAID "choose it to see the fix", AND THIS IS WHERE IT IS SHOWN. A copy below the floor
|
|
3983
|
+
// runs, so neither branch above is taken for it, and setup used to print "found" and carry on with the
|
|
3984
|
+
// row's promise unkept. It still carries on — an operator may have a reason to sit below the floor — but
|
|
3985
|
+
// not before saying so. The version is the one the menu already asked for (cached), or the package's.
|
|
3986
|
+
{
|
|
3987
|
+
let seenVersion = bin.version ?? null;
|
|
3988
|
+
if (!seenVersion) { try { seenVersion = menuVersion(bin.path) ?? null; } catch { seenVersion = null; } }
|
|
3989
|
+
if (olderThanFloor(seenVersion, eng.floor) === true) warn(cannotRunLine(eng, { ...bin, version: seenVersion }, process.env[eng.env]));
|
|
3990
|
+
}
|
|
3917
3991
|
// THE TERMS SENTENCE TRAVELS WITH THE PROGRAM, NOT WITH THE INSTALL OFFER. It was said only when this
|
|
3918
3992
|
// step offered to install the CLI, and a copy Clearotron installed on an earlier run skips that offer, so it is
|
|
3919
3993
|
// said here too. Using it, rather than installing it, is what accepts the vendor's terms.
|
package/build-info.json
CHANGED
|
@@ -124,14 +124,26 @@ through anything containing `/`):
|
|
|
124
124
|
|
|
125
125
|
| Alias | Full catalog id |
|
|
126
126
|
|---|---|
|
|
127
|
-
| haiku | `anthropic/claude-haiku
|
|
128
|
-
| sonnet | `anthropic/claude-sonnet
|
|
129
|
-
| opus | `anthropic/claude-opus
|
|
127
|
+
| haiku | `anthropic/claude-haiku` |
|
|
128
|
+
| sonnet | `anthropic/claude-sonnet` |
|
|
129
|
+
| opus | `anthropic/claude-opus` |
|
|
130
130
|
| gemini | `google/gemini-3.1-pro-preview` |
|
|
131
131
|
| gemini-flash | `google/gemini-3-flash-preview` |
|
|
132
132
|
| deepseek-v4-pro | `together/deepseek-ai/DeepSeek-V4-Pro` |
|
|
133
133
|
| azure | `azure-openai/gpt-5.4` |
|
|
134
134
|
|
|
135
|
+
The three tiers record a **tier, not a version**, because a tier is what a run asks for: the tier goes
|
|
136
|
+
to the program as the vendor's alias and the vendor answers with its newest model of that tier. This
|
|
137
|
+
id is what a dispatch row and a token-rollup row carry as the model *asked for*; what actually served
|
|
138
|
+
the turn is recorded beside it, and the report names that. A version here would be a claim about a
|
|
139
|
+
request nobody made, and wrong the day a newer model of the tier shipped.
|
|
140
|
+
|
|
141
|
+
One consequence, accepted when this was decided: per-model totals are keyed on what was asked for,
|
|
142
|
+
and the native-language lanes call the API directly, where a model id is required and a tier word is
|
|
143
|
+
not accepted. So one model reached by a stage and by those lanes appears under two keys —
|
|
144
|
+
`anthropic/claude-haiku` and `anthropic/claude-haiku-4-5`. They are different requests, and the split
|
|
145
|
+
says so.
|
|
146
|
+
|
|
135
147
|
The bottom four are **legacy names that no stage declares and no engine can run** — they resolve at
|
|
136
148
|
level 1 and then throw at level 2 (below). They are catalogue entries, not available tiers.
|
|
137
149
|
|
|
@@ -140,12 +152,18 @@ level 1 and then throw at level 2 (below). They are catalogue entries, not avail
|
|
|
140
152
|
as aliases, so each tier follows the vendor's newest model; to hold one still, set
|
|
141
153
|
`ANTHROPIC_DEFAULT_OPUS_MODEL` (or `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`,
|
|
142
154
|
`ANTHROPIC_DEFAULT_FABLE_MODEL`).
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
id the program reports it served.
|
|
155
|
+
An id that names a family and a version — `anthropic/claude-opus-5-5`, `claude-sonnet-5`, the dated
|
|
156
|
+
`claude-haiku-4-5-20251001` — is passed to the CLI as that model, so naming an exact model runs it
|
|
157
|
+
and keeps running it when a newer model of the tier ships. A family with no version
|
|
158
|
+
(`anthropic/claude-opus`) is the tier, not a model: the CLI has no model by that name, so it follows
|
|
159
|
+
the family like the bare alias. Telemetry keeps the level-1 catalog id as the model asked for, and
|
|
160
|
+
the attempt row records the id the program reports it served.
|
|
161
|
+
|
|
162
|
+
**The CLI must be new enough for the model.** Each release carries its own list of accepted models,
|
|
163
|
+
and one that predates a model refuses it outright while the tier alias goes on serving the previous
|
|
164
|
+
generation — a 400 at the first turn, or a run that quietly used an older model. Setup installs
|
|
165
|
+
`2.1.280` or newer for this reason; a copy already on the machine is used as it is, so `doctor`'s
|
|
166
|
+
version is the one to read before assuming which model a tier reaches.
|
|
149
167
|
|
|
150
168
|
**Anything else throws.** There is no regex fall-through to sonnet and no cross-provider
|
|
151
169
|
substitution: the `gemini`/`gemini-flash`/`deepseek-v4-pro`/`azure` mappings are gone with the
|
|
@@ -360,7 +378,7 @@ cannot be read as one list.
|
|
|
360
378
|
| `CLEAROTRON_DISPATCH_RECORD` | **on** | Write the verbatim message of every stage dispatch to `_driver/<stage>.attempt<N>[.repair<M>].dispatch.txt`, with `{file, sha, bytes, chars, kind}` on the attempt row. **Default ON** — `0`/`off`/`false`/`no` disarms it. Unlike `CLEAROTRON_DUMP_JSON` beside it, this is opt-OUT: the question it answers ("was the model given this?") is asked *after* the run that raised it, so a flag someone had to remember would be off on exactly the run that needed it. The files carry the company's identity verbatim and are deliberately not in the artifact table. |
|
|
361
379
|
| `CLEAROTRON_GATHER_SESSION_KEY` / `CLEAROTRON_GATHER_AGENT` / `CLEAROTRON_GATHER_SESSION_ID` | set per stage | Telemetry attribution into the provider-call ledger (set by the gather config; not operator-set). |
|
|
362
380
|
| `CLEAROTRON_RECORD_AXIS` | set per dispatch (unset ⇒ the stage is not fanned out) | Binds one fan-out turn of a recording stage to the single member it may write. `stageOnce` suffixes a fan-out stage's label with its axis, the gather config resolves `<stage>:<axis>` back to the base stage's tool group, and this carries the axis to the recording server. A call whose payload names a different member than the turn is bound to is REFUSED, so a seat cannot write into a sibling's file — without the binding every turn of the fan-out would record over member one. Set by the driver; not operator-set. |
|
|
363
|
-
| `PORTAL_READ_MODEL` | `
|
|
381
|
+
| `PORTAL_READ_MODEL` | `sonnet` | The model the portal's own compose-read turn uses. Distinct from the pipeline's tiers: this is a portal surface, not a stage. |
|
|
364
382
|
| `CLEAROTRON_ORDER_PROBE_SEED` | unset | Seed for `scripts/band-shape-probe.mjs`, so an ordering probe can be replayed. A diagnostic script's knob, not a run's. |
|
|
365
383
|
| `PROBE_TERM` | `DELTA` | The mark word `providers/uspto-local/bin/verify-index.mjs` searches when verifying a built local USPTO index. A diagnostic script's knob, not a run's. Change it when a row reports MEASURES NOTHING: that means the term had no exact hit, so the row's timing is not a result. |
|
|
366
384
|
|
package/driver/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# clearotron-driver
|
|
2
2
|
|
|
3
|
+
## 0.3.3-beta.1
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Fixed: a run can name a Fable model by its published id, not only by the tier word, which used to fail outright.
|
|
8
|
+
- Fixed: the portal's run list shows a run's rating, never the reviewer's sign-off word, while the run is still in progress.
|
|
9
|
+
- Fixed: naming an exact Claude model now runs that model, instead of quietly following the tier to a newer one.
|
|
10
|
+
- Fixed: a run now records the tier it asked for, not a model version nobody chose. The report still names the model that ran.
|
|
11
|
+
- Fixed: setup and doctor now report a Claude program too old for the models a search asks for, instead of passing it as fine.
|
|
12
|
+
- Fixed: setup installs a version of the Claude program new enough to run the current top-tier model, which older versions refuse.
|
|
13
|
+
- Fixed: the assistant's run summary names the rating again, where it had printed "[object object]" in its place.
|
|
14
|
+
|
|
3
15
|
## 0.3.3-beta.0
|
|
4
16
|
|
|
5
17
|
### Patch Changes
|
|
@@ -71,7 +71,7 @@ const shown = (names) => names.map((n) => n);
|
|
|
71
71
|
* and only one of them describes who gets the invoice.
|
|
72
72
|
*/
|
|
73
73
|
export function engineInventory(env = process.env) {
|
|
74
|
-
// THE SAME EXPRESSION AS THE RUN DOOR, character for character (driver.config.mjs
|
|
74
|
+
// THE SAME EXPRESSION AS THE RUN DOOR, character for character (driver.config.mjs,
|
|
75
75
|
// preflightEngineBinary). The obvious rewrite — `String(env.CLEAROTRON_AI ?? "").trim() || DEFAULT` —
|
|
76
76
|
// reads better and disagrees on a whitespace-only value: it falls back to the default while the door
|
|
77
77
|
// resolves `""` and refuses with "that is not an engine". The page would then name a known engine that
|
package/driver/dev-portal.mjs
CHANGED
|
@@ -145,7 +145,7 @@ function scanRuns(workspaceRoot) {
|
|
|
145
145
|
const s = JSON.parse(readFileSync(join(dir, "status.json"), "utf8"));
|
|
146
146
|
out.push({ runId: s.runId ?? `${slug}-${runName}`, slug: s.slug ?? slug, codename: s.codename ?? null,
|
|
147
147
|
agent: s.agent ?? null, state: s.state ?? null, stepN: s.stepN ?? null, stepLabel: s.stepLabel ?? null,
|
|
148
|
-
stepTotal: s.stepTotal ?? null,
|
|
148
|
+
stepTotal: s.stepTotal ?? null, signoff: s.review?.signoff ?? s.verdict ?? null, tier: s.tier ?? null, sendPending: s.sendPending ?? null,
|
|
149
149
|
markName: s.markName ?? null, updatedAt: s.updatedAt ?? null, archived });
|
|
150
150
|
} catch { /* not a run dir / unreadable status — skip */ }
|
|
151
151
|
};
|
|
@@ -293,8 +293,8 @@ $("#f").addEventListener("submit",async(e)=>{e.preventDefault();const fd=new For
|
|
|
293
293
|
const r=await fetch("/dev/enqueue",{method:"POST",headers:{"content-type":"application/json"},body:JSON.stringify(b)});
|
|
294
294
|
$("#fout").textContent=JSON.stringify(await r.json(),null,2);loadRuns();});
|
|
295
295
|
async function loadRuns(){const r=await(await fetch("/dev/runs")).json();
|
|
296
|
-
$("#runs").innerHTML=r.length?"<table><tr><th>run</th><th>state</th><th>step</th><th>
|
|
297
|
-
'<tr><td>'+(x.markName??x.slug)+' · '+(x.codename??"?")+'</td><td class="'+(x.state==="delivered"?"ok":x.state==="failed"?"err":"warn")+'">'+(x.state??"?")+(x.sendPending?" (sendPending)":"")+'</td><td>'+(x.stepLabel??"")+'</td><td>'+(x.
|
|
296
|
+
$("#runs").innerHTML=r.length?"<table><tr><th>run</th><th>state</th><th>step</th><th>sign-off</th><th>updated</th></tr>"+r.map(x=>
|
|
297
|
+
'<tr><td>'+(x.markName??x.slug)+' · '+(x.codename??"?")+'</td><td class="'+(x.state==="delivered"?"ok":x.state==="failed"?"err":"warn")+'">'+(x.state??"?")+(x.sendPending?" (sendPending)":"")+'</td><td>'+(x.stepLabel??"")+'</td><td>'+(x.signoff??"")+'</td><td>'+(x.updatedAt??"").slice(0,19)+'</td></tr>').join("")+"</table>":"no runs yet";}
|
|
298
298
|
async function loadOutbox(){const r=await(await fetch("/dev/outbox")).json();
|
|
299
299
|
$("#outbox").innerHTML=r.length?r.map(x=>'<details><summary>'+x.file+' <span class="warn">'+(x.packet?.kind??(x.legacyAgent?"delivered (legacy)":"?"))+'</span></summary><pre>'+JSON.stringify(x.packet??{legacyAgent:x.legacyAgent},null,2)+'</pre></details>').join(""):"outbox empty";}
|
|
300
300
|
// ── Searches panel (Phase 3a): registry levels + saved recipes via the /recipes/* proxy. EVERY
|
package/driver/driver.config.mjs
CHANGED
|
@@ -656,10 +656,29 @@ export const config = {
|
|
|
656
656
|
// stage definition and the id stamped on a token-rollup row are the same fact rather than two spellings
|
|
657
657
|
// of it. resolveModel below is the only reader that matters; an engine with its own resolveModelId
|
|
658
658
|
// overrides it, and anything already in catalog form passes through untouched.
|
|
659
|
+
// ── A TIER IS WHAT WAS ASKED FOR, SO A TIER IS WHAT IS RECORDED ─────────────────────────────────────
|
|
660
|
+
//
|
|
661
|
+
// These three named a VERSION — `anthropic/claude-opus-5` — and nothing ever asked for one. A stage
|
|
662
|
+
// names a tier, the tier goes to the program as the vendor's own alias, and the vendor answers with its
|
|
663
|
+
// newest model of that tier. The version written here was a claim about a request nobody made, and it
|
|
664
|
+
// was wrong the day a newer model shipped: a delivered run served throughout by the generation after
|
|
665
|
+
// Opus 5 recorded, on every attempt row, a request for Opus 5, and its token line accounted under the
|
|
666
|
+
// name of a model that did not run. Measured on that run, 2026-09-22.
|
|
667
|
+
//
|
|
668
|
+
// Owner's ruling, 2026-09-23: record the tier. What a run asked for is a tier, what it was served is
|
|
669
|
+
// recorded separately and already is, and the report names the model that ran — none of that moves.
|
|
670
|
+
//
|
|
671
|
+
// WHAT IT COSTS, RULED ON AND ACCEPTED RATHER THAN DISCOVERED LATER. Per-model totals are keyed on what
|
|
672
|
+
// was ASKED for, and the direct-API lanes must name a version because they call the API rather than the
|
|
673
|
+
// program — the API takes model ids, not tier words. So one model reached by both routes now lands in
|
|
674
|
+
// two buckets: `anthropic/claude-haiku` from a stage, `anthropic/claude-haiku-4-5` from those lanes.
|
|
675
|
+
// That is a real split in a per-model total and it was accepted with the ruling: the two are genuinely
|
|
676
|
+
// different requests, and keying the totals on what actually SERVED each turn is the change that would
|
|
677
|
+
// fix it properly, which is larger than this and not what was ruled.
|
|
659
678
|
export const MODELS = {
|
|
660
|
-
haiku: "anthropic/claude-haiku
|
|
661
|
-
sonnet: "anthropic/claude-sonnet
|
|
662
|
-
opus: "anthropic/claude-opus
|
|
679
|
+
haiku: "anthropic/claude-haiku",
|
|
680
|
+
sonnet: "anthropic/claude-sonnet",
|
|
681
|
+
opus: "anthropic/claude-opus",
|
|
663
682
|
gemini: "google/gemini-3.1-pro-preview",
|
|
664
683
|
"gemini-flash": "google/gemini-3-flash-preview",
|
|
665
684
|
"deepseek-v4-pro": "together/deepseek-ai/DeepSeek-V4-Pro",
|
|
@@ -675,11 +694,17 @@ export const MODELS = {
|
|
|
675
694
|
//
|
|
676
695
|
// A BARE Anthropic id (dated or not — "claude-haiku-4-5-20251001", "claude-opus-5") normalises to the
|
|
677
696
|
// catalog form too. The direct-API lanes (jx completions/judge/nativeread, driver.config JX_PROVIDERS)
|
|
678
|
-
// name their model that way because that is what the Messages API takes, so without this
|
|
679
|
-
//
|
|
680
|
-
//
|
|
681
|
-
//
|
|
682
|
-
//
|
|
697
|
+
// name their model that way because that is what the Messages API takes, so without this one model named
|
|
698
|
+
// in two spellings — dated and undated — would key apart in a rollup. The date suffix is dropped;
|
|
699
|
+
// anything that does not look like a bare claude id is returned untouched, so a genuinely unknown model
|
|
700
|
+
// still keys as-is rather than being guessed at.
|
|
701
|
+
//
|
|
702
|
+
// WHAT THIS NO LONGER DOES, SAID PLAINLY BECAUSE THE PARAGRAPH ABOVE USED TO CLAIM IT. It used to unite
|
|
703
|
+
// a stage's rows with those lanes' rows, because the tier resolved to a versioned id and so did they.
|
|
704
|
+
// The tiers now resolve to a tier (MODELS), and these lanes still name a version, so the same model
|
|
705
|
+
// reached both ways keys in two places. That split was ruled on and accepted (see MODELS) — it is not
|
|
706
|
+
// an oversight here, and closing it by collapsing a version to its tier would throw away the one thing
|
|
707
|
+
// these rows can still say about which model was asked for.
|
|
683
708
|
export function resolveModel(model) {
|
|
684
709
|
if (!model) return model;
|
|
685
710
|
if (MODELS[model]) return MODELS[model];
|
|
@@ -1976,7 +2001,12 @@ export const ENGINE_BINARIES = {
|
|
|
1976
2001
|
// newer", with no ceiling. Setup installs it into the engines folder (enginesFolder, below the table)
|
|
1977
2002
|
// when the reader picks this engine, and the resolver uses it only when the machine has no copy of its
|
|
1978
2003
|
// own. The package's own `bin` field names the program, so no path inside it is written down here.
|
|
1979
|
-
|
|
2004
|
+
// 2.1.280 is the floor because it is the oldest release that can run the current generation of this
|
|
2005
|
+
// vendor's top tier: below it the API refuses the model id outright ("version 2.1.280 or newer is
|
|
2006
|
+
// required"), and the tier alias quietly goes on serving the previous generation. Measured 2026-09-22
|
|
2007
|
+
// on 2.1.263 — the alias returned the older model and the pinned id was refused — so a floor that
|
|
2008
|
+
// only asks for a program that starts is a floor that passes a machine this engine cannot run on.
|
|
2009
|
+
package: "@anthropic-ai/claude-code", floor: "2.1.280",
|
|
1980
2010
|
// WHAT THE INSTALL TAKES ON DISK, in MB, which setup states before it asks to install. MEASURED, not
|
|
1981
2011
|
// declared by the vendor: the engines folder after a fresh install of this package into an empty
|
|
1982
2012
|
// folder, on npm 10.9.8 and on 11.19.1, 2026-09-14. A later release can be larger or smaller, so setup
|
|
@@ -2047,6 +2077,47 @@ export function enginesFolder({ env = process.env, home = homedir() } = {}) {
|
|
|
2047
2077
|
return String(env[ENGINES_DIR_ENV] ?? "").trim() || join(home, ".local", "share", "clearotron", "engines");
|
|
2048
2078
|
}
|
|
2049
2079
|
|
|
2080
|
+
/**
|
|
2081
|
+
* Is the copy of an engine's program on this machine older than the version this build asks for?
|
|
2082
|
+
*
|
|
2083
|
+
* THE FLOOR GOVERNED ONE ROUTE OF TWO. Setup passes it to npm, so a program it installs cannot land
|
|
2084
|
+
* under it; a copy already on the machine wins over the installed one by design, and nothing compared
|
|
2085
|
+
* its version to anything. The two routes are not equally common — most machines have their own copy —
|
|
2086
|
+
* so the check that existed covered the case that mostly does not arise.
|
|
2087
|
+
*
|
|
2088
|
+
* What that costs is not a crash. The program carries its own list of the models it accepts, so one
|
|
2089
|
+
* below the floor refuses the model a tier names and serves the previous generation instead: the run
|
|
2090
|
+
* completes, the report is delivered, and the only sign is a model id in the record that nobody chose.
|
|
2091
|
+
* Measured 2026-09-22 on 2.1.263, where the current top tier's id came back a 400 and the tier alias
|
|
2092
|
+
* answered with the generation before it.
|
|
2093
|
+
*
|
|
2094
|
+
* THREE-VALUED, AND THE THIRD VALUE IS THE POINT. `null` means "these two cannot be compared" — no
|
|
2095
|
+
* floor declared, nothing read from the copy, or a version this cannot parse — and it is never "fine".
|
|
2096
|
+
* A caller must say it could not look rather than print a pass, which is the absence-read-as-a-pass
|
|
2097
|
+
* class that the rest of this file keeps naming. Only a version that parses and sorts below the floor
|
|
2098
|
+
* comes back `true`.
|
|
2099
|
+
*
|
|
2100
|
+
* PURE, both arguments injected, so a test drives an old version and a current one without a program.
|
|
2101
|
+
*
|
|
2102
|
+
* @param {string|null|undefined} version what the copy reports, e.g. "2.1.273"
|
|
2103
|
+
* @param {string|null|undefined} floor the engine's declared floor, e.g. "2.1.280"
|
|
2104
|
+
* @returns {boolean|null} true = older than the floor; false = at it or newer; null = not comparable
|
|
2105
|
+
*/
|
|
2106
|
+
export function olderThanFloor(version, floor) {
|
|
2107
|
+
const parts = (v) => {
|
|
2108
|
+
const m = /^\s*v?(\d+)\.(\d+)(?:\.(\d+))?/.exec(String(v ?? ""));
|
|
2109
|
+
return m ? [Number(m[1]), Number(m[2]), Number(m[3] ?? 0)] : null;
|
|
2110
|
+
};
|
|
2111
|
+
const [got, want] = [parts(version), parts(floor)];
|
|
2112
|
+
if (!got || !want) return null;
|
|
2113
|
+
// Part by part, never as text: "2.1.99" sorts above "2.1.280" as a string, and that comparison would
|
|
2114
|
+
// read a machine two hundred releases behind as being ahead of the floor.
|
|
2115
|
+
for (let i = 0; i < 3; i++) {
|
|
2116
|
+
if (got[i] !== want[i]) return got[i] < want[i];
|
|
2117
|
+
}
|
|
2118
|
+
return false;
|
|
2119
|
+
}
|
|
2120
|
+
|
|
2050
2121
|
/** The npm arguments that install, or refresh, an engine's program in `dir`: "this version or newer". */
|
|
2051
2122
|
export function engineInstallArgs(spec, dir = enginesFolder()) {
|
|
2052
2123
|
return ["install", "--prefix", dir, "--no-fund", "--no-audit", `${spec.package}@>=${spec.floor}`];
|
|
@@ -153,8 +153,9 @@ and could not be: the telemetry logged the alias that was ASKED FOR, so an arm r
|
|
|
153
153
|
gemini and ran sonnet. Both tiers are gone — the failover chain was deleted in and both stages
|
|
154
154
|
declare an anthropic tier in `STAGES` — and every engine's model map now **refuses** an alias it cannot
|
|
155
155
|
run (`claudeModel`, `openaiModel`). On the anthropic engine a tier goes as the vendor's alias, a catalog
|
|
156
|
-
id
|
|
157
|
-
|
|
156
|
+
id naming a family and a version (`anthropic/claude-opus-5-5`, `claude-haiku-4-5-20251001`) goes as
|
|
157
|
+
that model so a pin holds, a `claude-*` id naming a family with no version goes as its family's alias
|
|
158
|
+
because the CLI has no model by that name, and anything else throws. To hold a tier on one model, set the vendor's own
|
|
158
159
|
`ANTHROPIC_DEFAULT_OPUS_MODEL` / `_SONNET_MODEL` / `_HAIKU_MODEL`, or `ANTHROPIC_DEFAULT_FABLE_MODEL` for
|
|
159
160
|
`fable`, which no stage asks for unless an override names it, as `CLEAROTRON_SYNTHESIS_MODEL=fable` does; each
|
|
160
161
|
reaches the CLI through the stage's environment.
|
|
@@ -156,11 +156,17 @@ const engineMaxBufferChars = () => Math.max(1024, Number(process.env.CLEAROTRON_
|
|
|
156
156
|
// An alias with no claude equivalent now FAILS LOUD, exactly as `openaiModel` has always done for a
|
|
157
157
|
// non-GPT id. That is the issue's requirement in one line: an unhonoured model override is an error,
|
|
158
158
|
// not a substitution.
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
159
|
+
// THE CATALOG IDS ARE NOT LISTED HERE ANY MORE, and removing them is what makes one rule cover every
|
|
160
|
+
// id. Four sat here and two of them disagreed with the other two: `anthropic/claude-opus-5` and
|
|
161
|
+
// `anthropic/claude-sonnet-5` went over as those models, while `anthropic/claude-sonnet-4-6` and
|
|
162
|
+
// `anthropic/claude-haiku-4-5` went over as their tier's alias. A caller naming an exact model got it
|
|
163
|
+
// or lost it depending on which of the four they happened to name, and nothing said which.
|
|
164
|
+
//
|
|
165
|
+
// The rule below now answers all four the same way, and no run changes: a stage names its TIER, and the
|
|
166
|
+
// tier words above are still the whole of what a run passes. These ids reach this function only when a
|
|
167
|
+
// caller names one — an override or an experiment arm — and there, being given the model you named is
|
|
168
|
+
// the behaviour the rest of this function already promises.
|
|
169
|
+
const CLAUDE_MODEL = { opus: "opus", sonnet: "sonnet", haiku: "haiku", fable: "fable" };
|
|
164
170
|
export function claudeModel(model) {
|
|
165
171
|
if (!model) return undefined;
|
|
166
172
|
if (CLAUDE_MODEL[model]) return CLAUDE_MODEL[model];
|
|
@@ -168,8 +174,29 @@ export function claudeModel(model) {
|
|
|
168
174
|
// that is a NAMING form of a model claude can actually run, not a substitution of a different one.
|
|
169
175
|
// The family must be named IN the id: a `claude-*` id whose family this build does not recognise
|
|
170
176
|
// throws too, rather than riding the old else-arm into sonnet.
|
|
171
|
-
|
|
172
|
-
|
|
177
|
+
// FABLE IS READ HERE TOO, and its absence was a live defect rather than a gap in readiness: the bare
|
|
178
|
+
// `fable` alias is in the table above, so `CLEAROTRON_SYNTHESIS_MODEL=fable` worked and hid it, while
|
|
179
|
+
// the pinned id every vendor page names — `claude-fable-5-1` — threw on its way to a program that runs
|
|
180
|
+
// it. Measured 2026-09-22: the program accepts that id and reports serving `claude-fable-5-1`.
|
|
181
|
+
const fam = /opus/i.test(model) ? "opus" : /haiku/i.test(model) ? "haiku" : /sonnet/i.test(model) ? "sonnet"
|
|
182
|
+
: /fable/i.test(model) ? "fable" : null;
|
|
183
|
+
if (fam && /^(?:anthropic\/)?claude-/i.test(model)) {
|
|
184
|
+
const bare = String(model).replace(/^anthropic\//i, "").toLowerCase();
|
|
185
|
+
// A CONCRETE ID GOES TO THE PROGRAM AS ITSELF, AND THAT IS WHAT MAKES A PIN A PIN. It used to come
|
|
186
|
+
// back as the bare family alias, so a caller who named an exact model got whichever model the tier
|
|
187
|
+
// pointed at — the same model on the day it was written, a different one the day a newer one
|
|
188
|
+
// shipped, and nothing to read in between. A silent un-pinning is the substitution this function
|
|
189
|
+
// exists to refuse, in the one form it still allowed.
|
|
190
|
+
//
|
|
191
|
+
// A FAMILY WITH NO VERSION IS THE TIER, not a model: `claude-opus` is what an operator types for a
|
|
192
|
+
// deployment of that tier, and the program has no model by that name. It keeps following the family.
|
|
193
|
+
//
|
|
194
|
+
// Measured against the program rather than assumed (2026-09-22): it accepts `claude-sonnet-5`,
|
|
195
|
+
// `claude-haiku-4-5-20251001` and `claude-fable-5-1` and reports serving each of them, so passing an
|
|
196
|
+
// exact id through costs nothing that the alias was buying. Where a caller names an id the program
|
|
197
|
+
// does not know, it says so and the turn fails loudly — which is the honest end of a bad pin.
|
|
198
|
+
return /^claude-(?:[a-z]+-\d|\d)/.test(bare) ? bare : fam;
|
|
199
|
+
}
|
|
173
200
|
throw new Error(`anthropic-agent: no claude model mapped for "${model}" — this engine runs claude only. Pass opus/sonnet/haiku/fable or a concrete claude-* id. (It used to substitute sonnet silently and log the alias you asked for: #238 corruption 3.)`);
|
|
174
201
|
}
|
|
175
202
|
|
|
@@ -9,8 +9,17 @@
|
|
|
9
9
|
// server, asks it to call `ping` once, and passes only when the reply carries what `ping` returned.
|
|
10
10
|
//
|
|
11
11
|
// WHAT IT RETURNS IS THE PROOF. Its one argument is a random word the probe mints for each
|
|
12
|
-
// turn and gives only to this process, so a reply that carries it cannot be the model guessing.
|
|
13
|
-
//
|
|
12
|
+
// turn and gives only to this process, so a reply that carries it cannot be the model guessing. It touches
|
|
13
|
+
// no file, no network and no run.
|
|
14
|
+
//
|
|
15
|
+
// IT IS DECLARED AS THE REGISTER SEARCH IS DECLARED, NOT AS WHAT IT DOES. codex decides from a tool's
|
|
16
|
+
// annotations whether a call needs approval, and `codex exec` refuses every call that does ("MCP tool call
|
|
17
|
+
// requires approval, but approval policy is never") unless its sandbox is bypassed. A read-only tool never
|
|
18
|
+
// needs approval, so a read-only `ping` passed on hosts where `register_execute_plan` — marked not
|
|
19
|
+
// read-only and open-world on every register server — was refused on every call and no search could run.
|
|
20
|
+
// The probe exists to answer for the tools a search calls, so its tool carries that tool's annotations,
|
|
21
|
+
// and a test keeps the two equal. The rule, in codex's own source, identical from 0.150.1 to 0.156.1:
|
|
22
|
+
// read-only → no approval; otherwise approval when destructive, or open-world, or either left unmarked.
|
|
14
23
|
import { serve } from "./stdio-server.mjs";
|
|
15
24
|
|
|
16
25
|
serve({
|
|
@@ -19,7 +28,7 @@ serve({
|
|
|
19
28
|
name: "ping",
|
|
20
29
|
description: "Return the word this check is waiting for.",
|
|
21
30
|
inputSchema: { type: "object", properties: {}, additionalProperties: false },
|
|
22
|
-
annotations: { readOnlyHint:
|
|
31
|
+
annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true },
|
|
23
32
|
handler: async () => {
|
|
24
33
|
const word = String(process.argv[2] ?? "").trim();
|
|
25
34
|
return word ? word : { isError: true, text: "ping: this server was started without a word to return" };
|
package/driver/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "clearotron-driver",
|
|
3
3
|
"private": true,
|
|
4
4
|
"type": "module",
|
|
5
|
-
"version": "0.3.3-beta.
|
|
5
|
+
"version": "0.3.3-beta.1",
|
|
6
6
|
"license": "AGPL-3.0-only",
|
|
7
7
|
"description": "Deterministic driver for the trademark clearance workflow: orchestration in code (fan-out, fan-in barrier, gating, retries); the model does judgment leaves only, through a reasoning CLI spawned per stage.",
|
|
8
8
|
"engines": {
|
|
@@ -639,7 +639,7 @@ export async function knockoutInner(ctx, job, opts = {}) {
|
|
|
639
639
|
// for reconcile-runs' exact liveness test. The stepper and the identity are separate calls now.
|
|
640
640
|
...identitySeed(),
|
|
641
641
|
stepIndex: 0, stepLabel: STEPS[0], stepN: 1, stepTotal: STEPS.length,
|
|
642
|
-
|
|
642
|
+
review: null, url: null, failedStage: null, reason: null, deliveredAt: null,
|
|
643
643
|
// A5/A3: a re-run of a previously-terminal knockout may reopen the state ONLY because the resume
|
|
644
644
|
// guard cleared the sentinel (ctx.stateReset threads that authority); startedAt is no longer
|
|
645
645
|
// seeded anywhere — writeRunStatus backfills it first-write-wins.
|
|
@@ -1092,11 +1092,11 @@ export async function knockoutInner(ctx, job, opts = {}) {
|
|
|
1092
1092
|
const deliveredAt = new Date().toISOString();
|
|
1093
1093
|
// `tier` beside `verdict` — the same band word under the name the clearance lane records it by, so a
|
|
1094
1094
|
// reader of either lane's status finds the rating in one place. `verdict` stays as it was.
|
|
1095
|
-
writeRunStatus(ctx, { state: "delivered",
|
|
1095
|
+
writeRunStatus(ctx, { state: "delivered", tier: overall, statement: published.statement, url: published.url, reports: published.reports.map((r) => ({ mark: r.mark, url: r.url })), deliveredAt, sendPending: true, stepIndex: STEPS.length - 1, stepLabel: STEPS[STEPS.length - 1], stepN: STEPS.length, stepTotal: STEPS.length });
|
|
1096
1096
|
// — the knockout lane's pool copy learns its terminal state the same way,
|
|
1097
1097
|
// for the same reason: publish returns the pool dir, and `state: "delivered"` is decided after it
|
|
1098
1098
|
// returns. Same seam, same best-effort contract, no lane-specific exception to write down.
|
|
1099
|
-
const stamp = writeSettleStamp(published.poolRunDir, { state: "delivered",
|
|
1099
|
+
const stamp = writeSettleStamp(published.poolRunDir, { state: "delivered", tier: overall, deliveredAt, runId: published.runId ?? run.runId, lane: "knockout" });
|
|
1100
1100
|
if (!stamp.written) note(`delivery: settle stamp not written (${stamp.reason})`);
|
|
1101
1101
|
const archived = archive(run);
|
|
1102
1102
|
rollupStatus(run.studioRoot);
|