@sema-agent/client-core 0.83.6 → 0.84.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +176 -0
- package/README.md +20 -13
- package/dist/abortableSleep.d.ts +10 -0
- package/dist/abortableSleep.js +37 -0
- package/dist/adapt/wireShapes.js +3 -2
- package/dist/adapter/activeRunSelfHeal.js +7 -7
- package/dist/adapter/downstream/eventToSdkMessage.js +6 -2
- package/dist/adapter/downstream/terminalToSdkResult.d.ts +1 -0
- package/dist/adapter/downstream/terminalToSdkResult.js +10 -5
- package/dist/adapter/downstream/wiringManifestView.d.ts +3 -1
- package/dist/adapter/downstream/wiringManifestView.js +1 -0
- package/dist/agentSession/engineAgentRegistryRead.d.ts +18 -0
- package/dist/agentSession/engineAgentRegistryRead.js +167 -0
- package/dist/agentsWireCaps.js +10 -1
- package/dist/attachmentsWireCaps.js +2 -2
- package/dist/controlRouter.js +1 -1
- package/dist/decideFailureNote.d.ts +2 -1
- package/dist/decideFailureNote.js +5 -3
- package/dist/detachWire.d.ts +1 -0
- package/dist/detachWire.js +13 -5
- package/dist/displayUntrusted.js +114 -18
- package/dist/engineAgentAbsence.d.ts +37 -0
- package/dist/engineAgentAbsence.js +142 -0
- package/dist/engineErrorCodes.d.ts +2 -0
- package/dist/engineErrorCodes.js +9 -5
- package/dist/engineNoticeCodes.d.ts +43 -0
- package/dist/engineNoticeCodes.js +118 -5
- package/dist/engineWireSdk.d.ts +2 -0
- package/dist/fleet/fleetProjection.js +3 -2
- package/dist/frozenSet.d.ts +1 -0
- package/dist/frozenSet.js +20 -0
- package/dist/handsSeam.d.ts +3 -0
- package/dist/handsSeam.js +19 -0
- package/dist/hitl/askGateWire.d.ts +1 -1
- package/dist/hitl/askGateWire.js +1 -1
- package/dist/hitl/hitlHostSurface.js +1 -1
- package/dist/hitl/parkResolver.d.ts +1 -0
- package/dist/hitl/parkResolver.js +9 -3
- package/dist/hitl/planReviewWire.d.ts +30 -3
- package/dist/hitl/planReviewWire.js +139 -32
- package/dist/hitl/sessionPolicyDeliverable.d.ts +2 -0
- package/dist/hitl/sessionPolicyDeliverable.js +14 -3
- package/dist/hitl/sessionPolicyWire.d.ts +40 -0
- package/dist/hitl/sessionPolicyWire.js +210 -0
- package/dist/hitl/toolApprovalWire.js +2 -2
- package/dist/index.d.ts +24 -13
- package/dist/index.js +13 -13
- package/dist/liveInitToolFace.js +28 -10
- package/dist/model/catalogLoader.js +3 -3
- package/dist/model/tierVocabulary.js +1 -1
- package/dist/request/taskRequest.js +15 -5
- package/dist/resumeRefusalCopy.d.ts +11 -0
- package/dist/resumeRefusalCopy.js +39 -1
- package/dist/rewindWireCaps.d.ts +0 -1
- package/dist/seam.d.ts +2 -1
- package/dist/seatContract.js +2 -2
- package/dist/systemReminderTag.d.ts +5 -0
- package/dist/systemReminderTag.js +20 -9
- package/dist/toolResult.js +2 -6
- package/dist/toolRoster.d.ts +16 -0
- package/dist/toolRoster.js +46 -5
- package/dist/wireErrorTriage.js +1 -11
- package/dist/workflowClient.js +4 -3
- package/dist/workflowMonitor.d.ts +1 -1
- package/docs/INTEGRATION-CLIENTS.md +701 -37
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.
|
|
38
|
+
**Version:** 0.84.1
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -67,7 +67,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
67
67
|
against — the tables live upstream precisely so this package does not keep a second copy that can
|
|
68
68
|
fall behind. The browser bundle really bundles the SDK through (the portability guard would
|
|
69
69
|
exit 3 rather than quietly mark it external).
|
|
70
|
-
- The declared floor is `>=11.3.0` (raised from `>=11.2.1` in 0.80.0: a deny decision may now name its settler — `settledBy: "policy"` is typed on both the durable decide body and the live respond body from 11.3.0 on, and the gate record's settlement vocabulary carries its thirteenth word `policy_refused`, which this package reads to place a refusal in the policy bucket rather than the person's; the package now compiles against 11.3.0 and no consumer ships 11.2.x any more, so the older floor lost its witness; before that raised from `>=11.0.1` in 0.79.0: `capabilities.mcpProbe`, the MCP probe face (`mcpCapabilities` / `probeMcp` with `McpProbeFace`) and the write receipt's third liveness arm (`stillLive: "unknown"`) are typed from 11.2.x on, 11.2.1 adds `TaskRequest.approverPosture` and the `mandated` approval-frame key to the types and the runtime key anchor, the package now compiles against 11.2.1, and no consumer ships 11.0.x / 11.1.x any more, so the older floor lost its witness; before that raised from `>=9.8.1` in 0.78.0: `DeniedBy` carries its tenth word `read_boundary`, `rules.write` answers a `stillLive`-discriminated body, `RemovalLiveness` / `RuleWriteRequest` / `RuleWriteResult` / `RuleWriteBehavior` are exported from the SDK root and `TaskRequest.excludeAllTools` is typed from 11.x on, the package now compiles against 11.0.1, and no consumer ships 9.8.x any more, so the older floor lost its witness; before that raised from `>=9.7.1` in 0.75.0: `Capabilities.deviceExecutor.management` is typed from 9.8.x on, the package now compiles against 9.8.1, and no consumer ships 9.7.x any more, so the older floor lost its witness; before that raised from `>=9.6.0` in 0.74.0: `Capabilities.approvalsStreamLive` / `.executionLane`, `LivePendingRow.frame`, the `live_*` approval-stream events and `gates[].toolCallId` are typed from 9.7.x on, and no consumer ships 9.6.0 any more, so the older floor lost its witness; before that raised from `>=9.4.0` in 0.71.0: the `tool_disclosure` / `tool_progress` frames and `ToolApprovalFrame.readRootCandidate` are typed there; earlier: raised from `>=8.8.0` in 0.69.0: the `reasoning_end` frame and `McpStatusPanel.lastLegMcp` are typed from 9.4.0 on), and it is *witnessed*: the guard checks that an actually
|
|
70
|
+
- The declared floor is `>=12.0.1` (raised from `>=11.3.0` in 0.84.0: the per-session background listing (`sessions.background`), its capability bit `capabilities.background.listFace` and the closed background-status vocabulary are typed from 12.x on, and the package reads all three, so it no longer compiles against 11.x; before that raised from `>=11.2.1` in 0.80.0: a deny decision may now name its settler — `settledBy: "policy"` is typed on both the durable decide body and the live respond body from 11.3.0 on, and the gate record's settlement vocabulary carries its thirteenth word `policy_refused`, which this package reads to place a refusal in the policy bucket rather than the person's; the package now compiles against 11.3.0 and no consumer ships 11.2.x any more, so the older floor lost its witness; before that raised from `>=11.0.1` in 0.79.0: `capabilities.mcpProbe`, the MCP probe face (`mcpCapabilities` / `probeMcp` with `McpProbeFace`) and the write receipt's third liveness arm (`stillLive: "unknown"`) are typed from 11.2.x on, 11.2.1 adds `TaskRequest.approverPosture` and the `mandated` approval-frame key to the types and the runtime key anchor, the package now compiles against 11.2.1, and no consumer ships 11.0.x / 11.1.x any more, so the older floor lost its witness; before that raised from `>=9.8.1` in 0.78.0: `DeniedBy` carries its tenth word `read_boundary`, `rules.write` answers a `stillLive`-discriminated body, `RemovalLiveness` / `RuleWriteRequest` / `RuleWriteResult` / `RuleWriteBehavior` are exported from the SDK root and `TaskRequest.excludeAllTools` is typed from 11.x on, the package now compiles against 11.0.1, and no consumer ships 9.8.x any more, so the older floor lost its witness; before that raised from `>=9.7.1` in 0.75.0: `Capabilities.deviceExecutor.management` is typed from 9.8.x on, the package now compiles against 9.8.1, and no consumer ships 9.7.x any more, so the older floor lost its witness; before that raised from `>=9.6.0` in 0.74.0: `Capabilities.approvalsStreamLive` / `.executionLane`, `LivePendingRow.frame`, the `live_*` approval-stream events and `gates[].toolCallId` are typed from 9.7.x on, and no consumer ships 9.6.0 any more, so the older floor lost its witness; before that raised from `>=9.4.0` in 0.71.0: the `tool_disclosure` / `tool_progress` frames and `ToolApprovalFrame.readRootCandidate` are typed there; earlier: raised from `>=8.8.0` in 0.69.0: the `reasoning_end` frame and `McpStatusPanel.lastLegMcp` are typed from 9.4.0 on), and it is *witnessed*: the guard checks that an actually
|
|
71
71
|
installed SDK at that line still exports every value-level symbol this package imports and still
|
|
72
72
|
declares `TaskStats.costMicroUsd` (the key `costOrNull` reads). A floor nobody ever ran is a
|
|
73
73
|
promise, not a contract.
|
|
@@ -297,14 +297,14 @@ guard still cross-checks the table by name).
|
|
|
297
297
|
| `scripts/run-segment-authority-single-source-test.mjs` | The authoritative-segment replacement verdict, single-sourced. `text_end.content` and the `text_delta` stream stopped being byte-identical the day the engine started redacting the former through the same filter as the result, so every consumer now has to decide six ways what to do with the segment it has half-emitted — and until this release that decision existed **twice**: once here for the transcript lane, once in the shell for the print lane, hot-fixed a version apart. The verdict is now one pure function both lanes call, and the guard pins it on the quantity that actually decides the outcome: whether the authoritative text still *starts with* the bytes that already left, not whether a flush has happened — the latter is a precondition, and anchoring on it withholds a perfectly ordinary answer. Each of the six forms is checked with its counter-case, the prefix length is pinned to UTF-16 code units against a non-ASCII sample whose UTF-8 byte count differs (slicing by bytes leaves the very thing being redacted on screen), and the withheld-segment ledger is compared by normalised equality rather than substring, because a short redaction marker quoted in an unrelated later answer would otherwise suppress that answer entirely. The same file pins the session-level memory-capture declaration to one mint point — the wire value is a single-member closed set, and a consumer that spells it wrong gets a loud refusal rather than a silently dropped privacy request — and pins the SDK URL/health transit to be the **same function reference**, since wrapping it would discard the one guarantee the transit exists for. A last section strips comments with the TypeScript parser and asserts the second expression has not grown back |
|
|
298
298
|
| `scripts/run-print-bash-iserror-test.mjs` | The print lane's Bash `is_error` authority (structured over regex). A second section pins where the denial classification word lands on this lane: on the message envelope, never inside the tool-result block, because that block is forwarded verbatim to the provider on compaction and a self-minted key there is the shape of an old, real defect. A word outside the upstream table — or an empty string, a non-string, or nothing at all — mints no key rather than a guess, and the word never moves the error flag, because attribution does not decide anything |
|
|
299
299
|
| `scripts/run-bash-benign-exit-interpretation-test.mjs` | Benign non-zero Bash exits (`returnCodeInterpretation`) stay non-errors across all three derivation arms, and the annotation transits to the card |
|
|
300
|
-
| `scripts/run-sdk-floor-test.mjs` | The SDK version floor — and, more to the point, that the *installed* type declarations still carry the keys this package reads — including, from 0.80.0, the three declarations that justify the floor itself: the key naming who settled a refusal on both decision legs, and the thirteenth word in the settlement vocabulary. They are found through the syntax tree rather than by searching text, because this guard's own comment stripper blanks string contents and would have made that check permanently, silently green. |
|
|
300
|
+
| `scripts/run-sdk-floor-test.mjs` | The SDK version floor — and, more to the point, that the *installed* type declarations still carry the keys this package reads — including, from 0.80.0, the three declarations that justify the floor itself: the key naming who settled a refusal on both decision legs, and the thirteenth word in the settlement vocabulary. They are found through the syntax tree rather than by searching text, because this guard's own comment stripper blanks string contents and would have made that check permanently, silently green. From 0.84.0 the floor is 12.0.1 and the guard witnesses the declarations the package now reads: the per-session background listing method, the `background.listFace` capability bit, and the closed background-status vocabulary that the registry classification switches over exhaustively. Every installed SDK the guard can see at or above the floor must carry those declarations too, so lowering the floor to a line where they do not exist fails. |
|
|
301
301
|
| `scripts/run-engine-caps-ledger-test.mjs` | A per-key disposition ledger for `GET /v1/capabilities`. The SDK's `Capabilities` grew from 74 keys to 93 in one release and nothing on the board could see it: this package consumes that table through four synchronous readers, and *nineteen new positions arriving while the package does not move* is exactly the disease shape this repo keeps logging on other axes — the fact is already on the wire, the package boundary is the cell that swallows it, and no client can read it however they write their side. So the ledger is reconciled **element-wise against the SDK interface in both directions**: a key the SDK added with no ledger row is red (someone must classify it), and a row for a key the SDK removed is red too (a registration that no longer does anything). Each row then has to survive its own claim — a `read` row names the source file, and the **code** there (comments stripped) must really mention the key, because prose asserting an alignment is the classic way these guards go hollow; a `not_read` row must have **zero** read sites in the tree, so wiring one up while the ledger still says the package ignores it is red rather than invisible. The census behind those two directions recognises five call shapes, each of which really occurs here — a reader whose base argument carries its own parentheses, a direct `caps.<key>`, a narrowing cast, an own-property read helper, and a `*_CAP` constant — and proves it on fabricated samples first, since a census that recognises one shape reports "nothing here" for the other four. What the guard deliberately does **not** judge is whether a position *ought* to be read: that is a design call, and the ledger only pins that every capability was looked at once by a person and that what they wrote down does not contradict the code |
|
|
302
302
|
| `scripts/run-sql-engine-capability-test.mjs` | The SQL-posture read face and the four-state capability reader underneath it. One capability cell here carries **four different things**, and each one points an operator somewhere else: nothing has been observed yet in this process (a one-shot doctor run is always in that state), the response arrived but carries no such key (an older engine), the engine explicitly answered `null` — *this deployment has no SQL backend*, which is a **positive fact** rather than an absence — and a full reading. Fold any two together and the screen states something flatly, confidently, and wrongly, so every positive control here is paired with a control pointing the opposite way, and the four sentences the doctor row can print are checked to be pairwise distinct and non-implying. The reading itself is narrowed no tighter than the mint: `txnMode: null` is a **legal value** — two of the three engines always report it that way, and the upstream type note names reading it as "optimistic" as the error — so treating it as malformed would throw away the entire reading for ordinary deployments, which is the same disease this repo logged when a consumer's domain was narrower than the producer's. A response that cannot be parsed **clears** the cell rather than leaving the previous engine's answer in place, and a separate invalidation port exists for the case the generation latch cannot catch — a same-port respawn whose new probe never succeeded, where the stale reading would otherwise be answered as current fact. Untrusted values (the isolation string is read back from a database server variable) are sanitised and bounded before display, and the bound is applied **before** escaping so a visible escape never gets cut in half. Finally the export names are themselves a guard: the shell still carries a copy that is meant to go red on the package's same-named export and be swapped out, so renaming anything here would silently disarm that lock |
|
|
303
303
|
| `scripts/run-web-search-backend-capability-test.mjs` | The deployment-default WebSearch backend read face (`capabilities.webSearch.backend`, engine ≥7.82.1). Same four-state discipline as the SQL and write-protection cells, with two things that are specific here and therefore guarded: a **missing key** (an older engine) and an explicit **`"none"`** (the engine says this deployment has no default search backend) point an operator in opposite directions — "cannot tell" versus "not configured" — and must never be folded; and the `none` sentence has to say both halves of the contract at once: the default scenario mounts no WebSearch tool, **and** a caller-supplied `webSearch` setting can still mount it on a single-user lane, because the capability advertises the deployment default, not whether this request has search. The backend word is read as an **open set** — the engine's closed set is typed from its own provider tuple and grows with it, so hand-copying three words here would turn a newly configured backend into "unreadable" (the narrower-than-the-mint disease this repo already logged once). `webSearch: null` is malformed rather than `none` (the mint never emits `null`), extra members never cross, an unparseable response clears the cell, a stale probe generation is dropped, the invalidation port clears to "not observed", and the open-set word is sanitised and bounded before display |
|
|
304
304
|
| `scripts/run-terminal-cause-projection-test.mjs` | The `7.64.0` wire reshape, projected. A run's ending stopped being eight parallel flat keys and became **one tagged cause** (`completed \| failed \| blocked \| paused`), and a tool call's gate stopped being four orthogonal words and became **one record** (`disposition` / `settlement?` / `origin?`). Both are read in exactly one place in this package, and this guard pins them at **two levels**, because the dangerous seam is "the reader was updated, the consumer was not": each terminal arm is checked on the reader *and* on the `subtype` / `is_error` / `errors[]` the projector actually emits. Two properties carry most of the weight. First, a terminal word this reader does not know is **never** laundered into an empty success — it lands on an `unknown` arm carrying the word verbatim, while a payload with no terminal word at all (the mock lane) keeps the success arm exactly as before, which is the one and only case the reader answers `null`. Second, the three window words (`approval_window_expired`, `denial_limit_window_expired`, `park_sla_expired`) must each be told apart by a different predicate: the previous generation collapsed all three onto one `timeout`, and re-merging them would throw away the discrimination this reshape just restored. Two byte generations are read by one reader, keyed on the discriminator upstream nailed (`"terminal" in result`): the current cause form, and the **flat** form that a current engine still emits on two lanes — replayed persisted bytes, which the service passes through verbatim rather than back-filling, and the service's own rejection envelope. A cause-form payload that also carries stale flat keys must ignore them entirely: keeping one compatibility read is what gives a single fact two sources. The same file also pins the MCP delivery verdict and HTTP status riding the wiring manifest, the four-state write-protection reading (where three of the four states mean *cannot tell*, and none of them may be printed as "there is no table"), and the park-reopen fetch identity: that predicate is asserted through the **real entry point**, since the defect being fixed was precisely a call site wired to a different predicate than the one that routed the row there. From 0.80.0 one of those three boundaries flips: the key naming **who settled a refusal** stopped being a dead byte and became part of the wire, so the check stopped scanning the build output for the word and started reading the request bodies the two decision legs actually send. A refusal attributed to the deployment's own policy carries the word; one attributed to a person, one with no attribution at all, and one carrying a word the vocabulary does not hold carry nothing — the wire has no slot for “a person decided this” other than the key's absence, so inventing one would be minting a word upstream does not have. The allow family never carries it on any of its routes, because that combination is refused before the approval is judged while the side effects of allowing have already landed, and the three refusals nobody was asked about (a card that failed, a user who walked away, an interruption) carry nothing either. A deployment that signs the bodies it accepts does not sign that word, and there is no capability bit to ask beforehand, so a refusal on exactly that ground is answered by re-sending the same decision once with that one key removed — byte-for-byte the same otherwise — rather than letting an optional note take the whole denial down with it. The guard measures that along three axes: the decision still lands and is reported as decided with the attribution handed back and a separate flag saying it never reached the wire; a caller who aborted in between gets no second request; every other refusal code, and every decision that never carried the key, send exactly once. The classification of a second failure is made from what the second body actually carried, not from what the card asked for. |
|
|
305
305
|
| `scripts/run-auto-mode-unavailable-test.mjs` | The fact behind "you are being asked because the auto-mode classifier could not run", and the one place its sentence is minted. The cause table is a **copy**, reconciled word for word in both directions against the installed engine's own bytes — it narrowed upstream, and the guard follows rather than keeping the old shape: a table checked against something nobody ships any more is the oldest way for a guard to be green and wrong. The retirement is held from both sides — the removed table must really be gone upstream, and the removed reader and word must really be gone here — while the word that left keeps arriving cleanly from an older engine, because the reader takes the cause as an **open set**: the vocabulary belongs upstream, so a copied list here would discard a legal value the day one is added, and the value discarded is precisely "this outage is a NEW kind". The reader's one exclusion is the word the engine says it never stamps here — the classifier did run and did answer, just outside its contract, so reading it as a failure would invent an event the engine denies. That exclusion used to be derived from a second table which no longer exists; the reason for it never lived in that table, so it is now stated where it actually comes from, pinned as a **named** set (a magic literal scattered through the reader reds) and cross-checked against the engine's own verdict declaration and against the reader having exactly one such comparison. One reader serves both the live ask and its durable parked twin, since the two carry the same key path and a second copy is how two ledgers drift apart. Absence is pinned as absence — most asks never consulted a classifier at all — and the sentences are checked mutually distinct, prototype-safe, and walked end to end: an unknown word reaches the sentence a person reads (the fallback that names it verbatim) and the status reading (unavailable for this round, never a fallback to "available"), with counter-controls proving neither assertion is vacuous |
|
|
306
|
-
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails |
|
|
307
|
-
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source |
|
|
306
|
+
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails. From 0.84.0 it also covers the reader for the two read-directory grant notices: it recognises only those two codes, needs the tool call id to match a card, passes the rejection reason through as written, and treats only the granted notice as evidence that a directory was added; a granted notice without both the directory and the spelling the engine now holds, or with a scope other than `exact`, is not read at all, and the scope word is pinned to the engine's type at compile time. The server also mints a few notices of its own through the same channel; those codes live in a second table with their own audiences, kept apart from the engine mirror (which must stay equal to the engine's catalog) and reconciled against the server's published package when one is supplied, so a user-facing server notice is no longer filed under operations. One dispatcher returns the typed facts for every code that has a reader, discriminated by code and tagged with its audience — only the user-audience codes belong on a user surface — and the guard ties the dispatch table to the module's own exported readers in both directions, so a reader cannot be exported without a row and a row cannot be dropped without the guard failing. |
|
|
307
|
+
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source. From 0.84.0 the roster-derived sentence source no longer throws on a value it does not recognise, including a reading of the manifest's `hands` section passed by mistake: it answers the same "not stated" sentence as the `hands` reader, from one shared source, and its six known sentences do not change. |
|
|
308
308
|
| `scripts/run-permission-rule-issue-codes-test.mjs` | The rule-lint refusal codes an engine reports when it will not compile a permission rule. The SDK publishes neither a schema nor a type for them, so the package mints the table from the engine's own bytes and the guard pays the cost of that copy instead of leaving it to somebody remembering: it parses the codes the engine actually mints and reconciles them against the table in both directions, so a code added upstream (the user would see a bare code) and a code only the package believes in (a branch that can never fire) both fail. It also reconciles the table plus a small retired ledger against the engine's declared union, which is deliberately not the same set — one member was renamed and its old name is still declared — so reviving a code the engine will never mint again is impossible and a future stale member shows up immediately. Sentences are pinned one per code, mutually distinct, and split by family: a rule that is wrong and a rule that is legal but unsupported on this lane are different next steps and may not share a sentence. The engine's own message rides along as prose — sanitized and capped after escaping, never matched on |
|
|
309
309
|
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, ten words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because an unknown word means different things in each: a denial layer this build does not know may have been added by a newer engine or may come from a damaged record, so its sentence says it cannot tell which instead of asserting damage; the asker vocabulary is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine). Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer; it must not say the question is asked every time — an answer for this one call may come from the person, a hook or an automatic check the deployment runs — and its wording is checked against the engine package's own description of the mandate. A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. Since 0.83.2 a sixth list covers the word a mandated question stands on (`APPROVAL_MANDATE_WORDS`, six words): it must equal the engine's own list word for word and in order, membership is exact, the card reader `readApprovalMandate` answers only for an own key holding one of the six words, and each word has one fixed sentence explaining why the question must be confirmed — six distinct sentences that never point the reader at writing a rule, never promise a question every time, never claim only a person may answer, and repeat no other sentence the package mints. The list is also pinned against the engine's type at compile time in both directions, while the published build references no engine package at all: every `.js` and `.d.ts` file in the build is scanned, and the same scan is first shown to fire on references planted in a scratch directory. |
|
|
310
310
|
| `scripts/run-engine-identity-test.mjs` | The engine generation anchors on `/health` (`pid`, `instanceId`, `startedAt`; engine >=7.67.0). `/health` is the one unauthenticated door and its heartbeat is always green, so "another host restarted the shared engine" used to be discoverable only by having some authenticated request hit a 401 first — a path that misreads a restart as a network fault. The reader narrows each anchor independently (one malformed field never hides the other two) and always hands back a reading object rather than an absence, because the caller is asking which anchors answered, not whether there was a response. The comparison is a three-word verdict, not a boolean: `unknown` when the two readings share no comparable anchor at all — an empty intersection means nothing could be compared, never that nothing changed — and the boolean convenience is pinned so that only `true` is an assertion. Any comparable anchor differing decides `changed`, so a reading whose `startedAt` matches while its `instanceId` does not cannot be waved through as the same life; precedence only decides which anchor gets named in the diagnosis |
|
|
@@ -356,7 +356,7 @@ guard still cross-checks the table by name).
|
|
|
356
356
|
| `scripts/run-notif-fleet-honesty-test.mjs` | [2393] the five notification/fleet disciplines a green type-check cannot see, each proven by reverting the fix. (1) The workflow-side dedup `return` keeps a count and a trace — without it "suppressed by design" and "a real completion swallowed because the runId minting changed" are the same observation. (2) `seq` normalisation has exactly one mint point, so a 0-based or fractional wire `seq` cannot make the watcher lane and the frame lane key the same completion differently (which would feed the model twice). (3) The TTL sweep defers to a probe arm that is still inside its own deadline — an entry recorded as "abandoned" must not be delivered a moment later — while an arm that has outlived its deadline never blocks the sweep, so the headless exit gate keeps its liveness. (4) The reset hook really clears every ledger it claims to (the sticky `prompt` ledger leaked across cases). (5) The fleet ledger counts all three drop paths (malformed / unknown frame type / isolation drop), and the panel projection's settled recycling is anchored on the settle instant and skips still-present rows, so the dedup token is never carried off with the entry (which would re-emit `end`) |
|
|
357
357
|
| `scripts/run-public-surface-test.mjs` | The outward promises: the npm export surface baseline (an **exact set**, both directions — a new export that never entered the baseline is one nobody watched leave, and deleting it later would not be red), the peer floor witness, and this README's claims |
|
|
358
358
|
| `scripts/run-client-core-message-branching-test.mjs` | §B8 (branching on error **text**) and §B10 (truthiness standing in for existence when the value can be `0`). AST + type-checker census over `src/`, a named ALLOW list carrying owner and expiry, a known-site floor, and two fixed corpora with a known verdict judged by the same classifier on every run |
|
|
359
|
-
| `scripts/run-registry-test.mjs` | Every ratchet number the suites compare against lives in exactly one place, `scripts/registry.json`: each cell must be a finite non-negative integer, and each one carries a dated ledger of every change (`from` / `to` / `when` / `why`) whose last entry for that cell must equal the number in force — a number moved without an entry, or an entry written without moving the number, is red in both directions.
|
|
359
|
+
| `scripts/run-registry-test.mjs` | Every ratchet number the suites compare against lives in exactly one place, `scripts/registry.json`: each cell must be a finite non-negative integer, and each one carries a dated ledger of every change (`from` / `to` / `when` / `why`) whose last entry for that cell must equal the number in force — a number moved without an entry, or an entry written without moving the number, is red in both directions. Every suite that reads it is checked to read it: no ratchet name may sit next to a numeric literal in their own source (an AST check, so an accounting comment or a test string mentioning the number is fine), each names the cell it reads, and each keeps the step-down channel — when a measurement comes in *under* a ceiling the suite prints one `RATCHET-SLACK` line instead of passing in silence, so slack cannot quietly accumulate under a ceiling nobody lowered. Negative control: a cell rewritten as a string, a fractional or negative cell, a missing cell, a cell with no ledger entry, a ledger entry missing a field, and a ledger tail that disagrees with the number in force each have to make the check speak |
|
|
360
360
|
| `scripts/run-client-core-failloud-test.mjs` | §C1/§C2: an empty `catch` with no comment anywhere inside it, a pure-swallow `catch` nobody reasoned about, and `void <write>` that really returns a Promise with no `.catch`. The exemption instrument is a comment saying why *this* failure may die; the documented-swallow count is a ratchet that only goes down |
|
|
361
361
|
| `scripts/run-client-core-typeshape-test.mjs` | Type discipline as a guard rather than a build side effect: the set of enabled strict knobs (one silently switched off is red), `tsc --noEmit`, and export-surface ratchets for inline anonymous shapes (≥3 members), `unknown` leaving the surface, and bare `unknown` returns — zero slack in either direction |
|
|
362
362
|
| `scripts/run-client-core-singleton-test.mjs` | Module-level singletons ⇄ `docs/refactor/p1-scan/singleton-manifest.json`, **both directions**: an unregistered singleton is red (registering it forces someone to answer "what if this got duplicated"), a stale entry is red, and the `dupRisk: high` count only goes down |
|
|
@@ -376,7 +376,7 @@ guard still cross-checks the table by name).
|
|
|
376
376
|
| `scripts/run-additive-key-passthrough-test.mjs` | The one disease shape behind two legs: a **closed whitelist / flattening arm** dropping a fact that is already on the wire, while both sides of the seam look correct. (1) The `task_progress` projection carries a registered **key ledger** — a frame populated with every key the service really projects is pushed through the shipped `eventToSdkMessage`, and the set of wire keys that survive must equal the registered pass-through list **name for name in both directions**, so quietly forwarding one more key is as red as quietly dropping one. `model` (the child run's model id, minted by core as `prepared.model.id` and projected by the server since 7.52.1) is the key this batch adds, with the same conditional the server itself applies: a non-empty string or no key at all — an empty string is neither a model id nor "unknown". The ledger is also checked against the fenced list in `docs/INTEGRATION-CLIENTS.md` §3d, so a doc that still says seven keys while the code forwards eight is red rather than merely stale. (2) The decide-failure arms carry the server's S-02 `currentPending` pointer key from a 409 `approval_stale` refusal onto the outcome the host reads. The reader is structural rather than `instanceof`, because the client is host-injected and the class identity is not this package's to assume; a half triple never mints (half a pointer cannot relocate anything), an empty string is not presence, and `checkpointToken` never transits. Both the allow and the deny leg are driven end to end through the real durable approval path — as is the accept-session leg, where a refusal carrying the pointer key must now re-raise instead of silently re-sending the human's answer for the **old** card as a plain approve (one decide call, pointer preserved), while a legacy 400 still falls back exactly as before — and all three flattening points must call the one shared reader — the same-shape residue check that makes "fixed one arm and left the twin" red instead of invisible. (3) The same disease growing on the REQUEST side: the `.mcp.json` → server-spec projection rebuilds each server key by key, and the settings schema deliberately leaves some keys parse-transparent — whatever JSON the file carries reaches the engine untouched, because validating them where the whole domain parses all-or-nothing would let one bad declaration take every server down silently. The whitelist had no row for the newest of them, so an operator's per-tool declarations — the ones the write fence reads — were stripped at the package boundary while both sides looked correct. The criterion is not "is that key handled" but the transparent-key table read out of the INSTALLED schema at runtime, reconciled name-for-name against this leg's ledger, so the day upstream adds a third one this turns red and forces an explicit decision. Behaviour is pinned on both transports, by object identity rather than deep equality (a rebuild would be a second judge), and malformed values must transit UNCHANGED rather than be refused here — the engine refuses them loudly and names the server, whereas a package-side judge can only swallow a declared protection quietly. Absence still mints no key, unknown keys still never reach the wire (the fix is the dropped key, not the gate), and the one transparent key this leg deliberately does not forward is a ledger entry with its own exit condition: it belongs to the deployment plane, and the day the request-plane type declares it the entry's premise is gone and the gate says so |
|
|
377
377
|
| `scripts/run-esc-halt-plan-test.mjs` | The Esc stop decision every client shares: fire the **turn-level** halt first, and escalate to a **run-level** cancel in exactly two cases — the engine itself answered with a 409 from the closed code set (it is saying "there is no in-flight turn here; use cancel for a run-level stop"), or that shot came back with no verdict at all *and* the shell can independently prove a permission card was on screen. Everything else does not escalate. The asymmetry is the whole point and every negative control guards the same direction — deciding *not* to escalate costs the user one more choice on a busy-session card (recoverable), deciding to escalate wrongly tears down a run that was alive and takes every in-flight tool with it (not). So: the closed code set is a **frozen** value, not a `ReadonlySet` — type-level immutability does not stop a consumer's `.add()`, and the guard proves it by really trying to mutate the exported value and then checking the verdict did not drift; the escalation gate is the **conjunction** of that closed set and the 409 status, since honouring the code alone lets a 500 that merely quotes it drive a destructive call; `interrupt.not_held` and `steering.not_running` are deliberately outside the set (the first means *this replica* has no live face — the run may be perfectly alive on another); an unreadable code falls to the no-escalation side; a `parked` flag never overrides a verdict the engine did give, and only strict `true` counts when it did not. The first shot is unconditional by construction — it does not consult `parked`, because the 409 it earns is exactly the verdict the gate wants — and the verdict itself is a closed machine-readable reason word, not display copy. A third escalating case was added once tearing the stream stopped reaping the run: with detach armed, a shot that never lands leaves the run going all the way to the end of the turn, so the Esc the user pressed has no effect at all and nothing on screen says so — the old behaviour had a silent backstop (tearing the stream ended the run) and that backstop is gone. The new fact is held to the same three disciplines as `parked`: it is read only where the engine gave no verdict, it is judged **after** `parked` so an existing host's reason word does not change under it, and only strict `true` counts. Absence is proven to be a no-op rather than asserted — the guard carries its own reference implementation of the previous version's table, runs the full grid through both, requires zero divergence when the new field is omitted, and first shows the comparison really does report a difference on the one cell where the two versions are meant to differ |
|
|
378
378
|
| `scripts/run-peer-frame-projection-test.mjs` | The three engine-injected lanes design/385 puts on the **one** `task_notification` carrier, which are not the same kind of thing at all: a delegated child's uplink (`agentMessage`), another session's message drained from this session's own box (`crossSessionMessage`), and a receipt about one of *this* session's own outbound messages (`crossSessionNotice`). The engine renders none of them inside a `<task-notification>` shell, so a client that projects them as the generic completion card shows "background task finished" while the model read a colleague's sentence — two faces describing different events. The discriminator is pinned to the **typed carrier being present**, never to the `summary` text: those carriers can only be minted by the engine's injection legs (the external `notify()` input is a strict subset of the payload and can wear none of them), while `summary` is filled by every notification there is — so anchoring on text would let any background task impersonate a colleague's message by writing `<agent-message from="…">` into its own summary, and a positive control asserts exactly that payload still projects as the generic card. Fail-closed has two tiers rather than one: a broken **required** field (empty `from`, a non-string `body`, a notice `kind` outside the closed set) returns absence so the caller falls back to the generic card — an honest downgrade where the user still sees the notification — while a broken **optional** field drops only itself, because losing an attribution note and losing a colleague's whole message are not the same magnitude. The provenance side record is **required and must agree on four points** (`kind` matches the lane; `from`/`taskId`/`seq` are present and equal the carrier/payload — each equality is anchored on a core mint site and pinned by the cli wire-anchor A-K24), so a carrier signed with a trusted name but a disagreeing provenance falls back to the generic card; peer bodies pass the same authority-envelope neutralization core applies (`<task-notification>` etc. are defused) so a colleague's text can never seed the resume dedup ledger. Lane precedence copies the engine renderer's own order, because the model already read the frame in that order and a client ordering of its own would put a card on screen that disagrees with the frame the model saw. Rendering and parsing of the transcript line live in the same module and are round-tripped in both directions, including a body carrying a forged closing tag (a parser fooled there hands half a message to the next row) and a quote inside the sender label (which must not forge a second attribute); the notice lane is deliberately kept **out** of the parser, since recognising it would mean anchoring the `[Cross-session …]` prefix and a user typing that same line would be rendered as engine speech. Hostile carriers are read as own **data** descriptors only and accessors are never invoked at all — `catch` catches throwing, not never returning — proven by a counting getter that must stay at zero calls, alongside a revoked proxy and a prototype-only carrier; and four legacy payload shapes assert the no-carrier path is byte-identical to before, which is the executable form of "zero difference for an older host". A re-supplied cross-session message — same task id, status and sequence as the first delivery, handed to the model again after compaction — is not rendered a second time, because the first delivery is still on the user's screen; a record with the next sequence number still renders |
|
|
379
|
-
| `scripts/run-wiring-manifest-projection-test.mjs` | The two end-user facts carried on the engine's `wiring_manifest` frame (`modelGate`: which tools this run's model gate removed and the verbatim restore hint; `autoMode`: whether auto mode is actually armed and the engine's own reason word). Projection: both sections ride as `_sema_`-prefixed superset keys, verbatim, and no SDK-named key is minted; a frame where neither section is well-formed projects to `none/not_in_slice` (no empty arm); `modelGate` needs all three keys and treats `removed: []` as a bad value rather than a reading; `autoMode` needs a boolean plus a non-empty reason that agrees with it, and the reason word is never mapped onto the capabilities vocabulary; the frame is flat (a nested `manifest:{}` wrapper is not a supply); `eventId` rides like every other arm. Adapter: exactly one chrome event on the main lane, a sub-flow frame (any `parentToolCallId`, `null` included) yields nothing, and an absent `eventId` leaves the key absent. Added at receiving time because the shell-side gate could not see this package's behaviour: two mutations (empty `removed` accepted, sub-flow gate removed) had passed the package suite untouched 0.71.0 adds sections F–I: the fourth/fifth/sixth manifest sections (`tools` via the roster reader, `hooks[]` rows dropped one by one when malformed, `lsp` absent unless `mounted` is a boolean), the `tool_roster_delta` arm (narrowed `delta`, `malformed` when `fromDigest`/`roster` cannot be read, host applies it against its own digest), the `context_usage` arm (finite-gated scalars plus `sections[]` rows dropped one by one), and the `WiringManifestMcpEntryView` rename with `MAX_AGENT_SKILLS` gone from the surface |
|
|
379
|
+
| `scripts/run-wiring-manifest-projection-test.mjs` | The two end-user facts carried on the engine's `wiring_manifest` frame (`modelGate`: which tools this run's model gate removed and the verbatim restore hint; `autoMode`: whether auto mode is actually armed and the engine's own reason word). Projection: both sections ride as `_sema_`-prefixed superset keys, verbatim, and no SDK-named key is minted; a frame where neither section is well-formed projects to `none/not_in_slice` (no empty arm); `modelGate` needs all three keys and treats `removed: []` as a bad value rather than a reading; `autoMode` needs a boolean plus a non-empty reason that agrees with it, and the reason word is never mapped onto the capabilities vocabulary; the frame is flat (a nested `manifest:{}` wrapper is not a supply); `eventId` rides like every other arm. Adapter: exactly one chrome event on the main lane, a sub-flow frame (any `parentToolCallId`, `null` included) yields nothing, and an absent `eventId` leaves the key absent. Added at receiving time because the shell-side gate could not see this package's behaviour: two mutations (empty `removed` accepted, sub-flow gate removed) had passed the package suite untouched 0.71.0 adds sections F–I: the fourth/fifth/sixth manifest sections (`tools` via the roster reader, `hooks[]` rows dropped one by one when malformed, `lsp` absent unless `mounted` is a boolean), the `tool_roster_delta` arm (narrowed `delta`, `malformed` when `fromDigest`/`roster` cannot be read, host applies it against its own digest), the `context_usage` arm (finite-gated scalars plus `sections[]` rows dropped one by one), and the `WiringManifestMcpEntryView` rename with `MAX_AGENT_SKILLS` gone from the surface From 0.84.0 the `hands` section (whether this leg was assembled with the engine's built-in file and shell tools) is projected as well: a strict boolean `mounted` plus the engine's reason word passed through as written, absence kept as "not reported" rather than folded to either answer; a three-state reader and one sentence source share their opening and closing with the roster-derived sentence, whose bytes do not change. The sentence source answers the not-reported sentence for anything it does not recognise, including a roster-derived reading, and never throws. |
|
|
380
380
|
| `scripts/run-submit-wiring-manifest-test.mjs` | The non-streaming submit receipt can carry the run's opening wiring manifest (`TaskResult.wiringManifest`, additive on newer servers). `readSubmitWiringManifest` answers one of three: the key is absent on the receipt itself (older server, or a deployment whose engine never produced that frame) — not the same as unreadable; the key is present but cannot be read (not an object, or none of the nine sections survive); or a manifest view. The view is the same shape the streaming lane's chrome event carries (minus its two envelope keys) and is assembled by the same code path, so both lanes agree byte for byte on the same object. Liveness fields ride through untouched — this reader never mints a liveness verdict — and an operator-shaped receipt with extra governance sections reads to the same view as a tenant-shaped one. A zero-tool roster is a real reading, not an absence. |
|
|
381
381
|
| `scripts/run-rule-offers-reader-test.mjs` | The narrowing reader behind the "don't ask again" options, now a public entry point rather than a card-port-only one. Hosts that render the frame themselves (a browser has no three-way terminal card) previously had to rebuild this reader on their side, and what it carries is a **redemption-safety** judgement, not a convenience: the batch arm is redeemed by **index**, so a reader that compacts the array after dropping a malformed entry makes the k-th option a person clicked and the k-th rule the server writes two different rules. So: a bad entry is dropped **on its own** (one bad option must not make a real one disappear) while every surviving entry keeps its **original wire index** — pinned from both ends, with the bad entries leading and trailing. A batch's *members* are the opposite: any malformed member drops the whole batch, because a conjunctive batch is one "yes" to all of them and a batch missing a member is a different grant; its honest-remainder count is a reading, not decoration, so a non-integer or negative value drops the batch rather than rendering a fabricated zero. An empty array, a non-array, an over-cap array and an all-bad array all read as **absence** rather than an empty list, because an empty list renders as "there is an option lane with nothing in it". The two wire generations are ordered by a rule, not a preference: the newer key wins outright, a newer key that is **present but unreadable** does not fall back to the retired key (borrowing the older material would pass someone else's options off as this request's), and a `null` newer key reads as absence so a relaying layer that serialises "missing" as null cannot delete the whole lane on older engines. The public entry is finally reconciled against **both** card-port legs on the same material, byte for byte, so the exported reader and the one the card sees can never become two. Two upstream vocabularies used to be **hand-copied** here, and both had fallen behind: a match word outside the copied pair dropped an otherwise valid option outright, and a batch carrying a directory-read member — a member kind the copy did not know — dropped the whole batch. Both tables now come from one place upstream and are re-exported verbatim, pinned in both directions: every word in the table must be accepted (a narrower copy reds on the words it never learned) and a word constructed to be outside it must still be refused (a reader widened to "any string" reds too), with the retired-key normalising leg sharing the same narrowing so the fix cannot land on one leg only. A member whose kind is genuinely unknown still drops **the whole batch and only that batch** — never one member, because a conjunctive batch one member short renders "yes to N" as "yes to N−1", and never the card, because the honest single beside it is intact — while a member from before the discriminant existed normalises to the historical kind rather than being refused. The additive per-segment reasons ride through verbatim, drop only the row that is malformed, and stay **absent rather than empty** when nothing survives, since an empty list would read as "confirmed nothing uncovered" while the count remains the only source of truth |
|
|
382
382
|
| `scripts/run-resume-refusal-copy-test.mjs` | The **words** a client says when a resume is refused, minted once here instead of three times. The facts behind them already lived in this package; the sentences did not, so each client wrote its own — and those sentences answer a safety question (was my decision consumed, can this token still be redeemed), which is exactly the kind of answer that must not vary by client. Two closed sets meet here and the guard pins their relationship in both directions, because it is a premise rather than a coincidence: one set answers *can waiting help* (the codes the server mints a wait on), the other answers *what should a person be told*, they **intersect in exactly one code**, and each keeps a member the other must not have — a placement mismatch is never waitable no matter what arrives on the response, since its remedy is a changed argument rather than elapsed time, and a full governance window needs no prose because "you can wait" is the whole message. The overlapping code delegates its wait and its disposition to the existing reading rather than judging again: nine shapes of input drive both entry points and the two readings must agree byte for byte, the absent case included, because two judges always diverge somewhere. The wait is narrowed to the domain the server mints it in, which is **stricter than the shell's own copy was** — a zero now reads as no window rather than as "retry now", and the wake-up it would retry is an at-most-once action with real side effects. The third sentence is chosen by the disposition, never by the engine's prose: rewriting the message to either upstream branch's exact wording, with the window untouched, must leave all three sentences unchanged, while adding a window must change the third one and only the third one. The engine's folded resume refusal (`resume_blocked_by_policy`) gets its own reading — the original code as named, absent or unreadable — and a wording that neither claims nothing was consumed nor predicts whether a retry would pass. |
|
|
@@ -386,13 +386,13 @@ guard still cross-checks the table by name).
|
|
|
386
386
|
| `scripts/run-approval-frame-chrome-arms-test.mjs` | The two in-stream approval frames finally reaching every host through the shared pipeline instead of one shell's private branch — the shape of a layering defect: hosts that only consume the package could not rebuild their pending cards after a reconnect, and did not clear a card the engine had withdrawn. The payload is deliberately carried as the **envelope** the upstream types declare rather than the first-version card: the stream parser applies no predicate, so narrowing here would let a legitimately newer frame pass as the older shape and invite consumers to read keys a newer card never promised. The guard therefore pins that every open key survives untouched, that an unknown version still passes through, and that narrowing is left to the host's own predicates — with the fallback being a generic card and a person, **never** an automatic denial. A frame whose version cannot be read at all is reported as malformed rather than dropped in silence, because both frames carry user-visible decisions and state changes. Both arms are registered as **required** host duties, and their duty text names the load-bearing rules a host would otherwise have to rediscover: which predicate to narrow with, that the reconnect preamble — not a replayed historical frame — is the authority on which cards exist, and that a withdrawal frame can be lost entirely. Unlike the sibling arms, these carry **no** sub-stream cutoff: an approval raised under a delegated call still has to reach a person, and filtering it by ownership is the host's job, not a reason to discard it. Finally the upstream bytes that justify the envelope discipline are checked to still be there, since the whole design rests on them |
|
|
387
387
|
| `scripts/run-terminal-status-vocabulary-test.mjs` | One place that decides whether a run has **ended** and whether it ended badly — written because that judgement had already been hand-copied three times, so the day the engine added a word for *the agent itself reported it cannot continue*, every copy missed it and a panel settled a self-reported failure as a success. The distinction the table exists for is pinned from both sides: that word belongs in it, while the two words meaning *waiting for a person to decide* deliberately do **not** — reading those as endings would bury a run that is actively waiting on the reader. A word this client does not know answers *no*, and the guard states plainly that *no* is not evidence of success: proving success means reading the positive side, so negating this predicate is the very mistake that caused two earlier incidents. The fleet lane gets the same treatment from the other direction: a workflow parked on a durable approval used to fall through to *running*, leaving the person with no hint that a card was waiting, and it now lands on the same rendered word the task lane already used — same fact, same word, checked end to end on a real row. Why the word was added directly rather than carried as a private superset key is checked mechanically against the upstream declaration being open, so the day it closes this reds and the decision gets revisited. The residue sweep is the point: the source tree must contain **no** further inlined copy of the judgement, each of the three former sites is checked to really read the single predicate, and the one reviewed exemption carries its reason **and** a liveness assertion, so an exemption whose justification expires cannot quietly keep standing |
|
|
388
388
|
| `scripts/run-terminal-word-source-test.mjs` | Two tables of ending words, kept apart by **who owns them** — because they used to be one. The engine's own closed set of reasons a run ended, and the server's set of row states a run can finish in, overlap in three words but not in all of them: one word for *something outside stopped it* exists only on the server side, and one for *it paused and can be resumed* exists only on the engine side and means very nearly the opposite of an ending. Merged into a single list, those two sources became indistinguishable, so a new word on either side looked the same as a new word on the other, and the safest-looking move — folding the unknown word into a known one — is the exact mistake that has caused incidents here before. The engine-owned table is checked as a **copy, not an opinion**: it is reconciled word-for-word and in order against the installed engine package, read from both its declaration and its runtime bytes with the two required to agree, so the day upstream adds a fifth reason this reds before anything ships. The two dividing words are each pinned from both sides, including against the upstream declaration directly rather than only against this package's own list. Why the table is copied rather than re-exported is itself an assertion with an expiry: the day upstream publishes the set as a value, this guard reds and the decision gets revisited. The renamed tables leave **no alias** behind, since an alias would let a reader keep consuming the merged list and the split would have bought nothing |
|
|
389
|
-
| `scripts/run-terminal-table-provenance-test.mjs` | Several tables answering *has this ended*, which until now only asserted their own current wording rather than that the wording was right — a snapshot equality passes forever even the day upstream adds a word this package never learns about. Each is reconciled against a named upstream source instead, one comparator shared across all of them rather than one copy per table: a notification's terminal words are the engine's own closed set minus its one live word; a sub-agent tick's terminal words are the engine's own inline status literal minus *running*; a run row's terminal words must **cover every** engine reason a run can end — missing one is the exact failure mode that once let a client retry a connection until its budget ran out while the ending sat unread in the row the whole time — plus one explicitly named legacy word the engine's current declaration no longer carries; a fleet row's terminal words are pinned to **exact equality** with the transport's own status set minus its known non-terminal words (an adversarial pass found the earlier one-directional form let a real terminal word be quietly deleted from this side and still pass), and separately pinned against that set's current member count so the day it changes a person has to look. A sixth table has
|
|
389
|
+
| `scripts/run-terminal-table-provenance-test.mjs` | Several tables answering *has this ended*, which until now only asserted their own current wording rather than that the wording was right — a snapshot equality passes forever even the day upstream adds a word this package never learns about. Each is reconciled against a named upstream source instead, one comparator shared across all of them rather than one copy per table: a notification's terminal words are the engine's own closed set minus its one live word; a sub-agent tick's terminal words are the engine's own inline status literal minus *running*; a run row's terminal words must **cover every** engine reason a run can end — missing one is the exact failure mode that once let a client retry a connection until its budget ran out while the ending sat unread in the row the whole time — plus one explicitly named legacy word the engine's current declaration no longer carries — kept on purpose to read rows stored before that word was retired, since retiring a word the engine writes does not remove rows already stored with it; a fleet row's terminal words are pinned to **exact equality** with the transport's own status set minus its known non-terminal words (an adversarial pass found the earlier one-directional form let a real terminal word be quietly deleted from this side and still pass), and separately pinned against that set's current member count so the day it changes a person has to look. A sixth table, a workflow's terminal words, has two upstream producers and is pinned to **exact equality** with their union minus the one live word: the stored run record's status set, and the statuses the in-process task registry gives a workflow handle (read from the engine's own source; it emits *cancelled* when a workflow without a store is stopped — a version of this table reconciled against the first producer alone dropped that word, so a stopped workflow was never marked delivered and its completion could be delivered twice). The table is separately checked against every non-terminal word gathered — including one meaning *durably paused*, sourced from the engine's own outcome vocabulary rather than any of the other unions, after the same adversarial pass found a caller-documented non-terminal word this boundary had missed — and a behavioral regression drives the installed engine's real task registry through *stop* then *read the output*, feeds that result through both projection paths, and requires each to mark the run delivered and clear it from the waiting count. A seventh pair, found by the same pass sweeping the whole tree for the same shape of hand-copied table, answers a related but distinct question — whether a session's claim on a run has been released or is still held — and is pinned as two complementary halves of one upstream set: released-minus-one-named-legacy-word and held must partition the transport's status set exactly, so a real state going missing from either side is caught the same way a fabricated one would be. Every extraction in this guard parses real syntax rather than pattern-matching quoted text, so a comment mentioning a word never counts as that word being present, and single- and double-quoted members are read identically |
|
|
390
390
|
| `scripts/run-workflow-park-truth-projection-test.mjs` | The read face for *which approvals a workflow run left parked* — and the credential that must never ride along with it. Upstream strips the redemption token from that response, and this package's reader is built so the token **cannot** come back: each row is assembled field by field from the three identity keys, never copied wholesale, so an extra key appearing upstream is structurally unable to reach anything this package hands a UI. The guard proves that rather than asserting it — a poisoned row carrying a secret is read, and the secret is searched for across the **entire** serialized result, with the same search proven to find it in the input so a blind search cannot pass; renaming the credential key does not help it through, because the rule is *only these three*, not a blocklist; and the reader's own source is checked to contain no object spread, since one such line would quietly void all of it. The other half is an absence distinction with opposite consequences: a record with **no** parks field at all was written by an older engine and proves nothing about whether approvals are waiting, while an empty list is a positive statement that none are — collapsing those two would let a run whose parked approvals cannot be proven be resumed anyway, so they are kept literally distinguishable, and a payload whose rows are all unreadable answers *unknown* rather than *none*. The four refusal codes for this family are checked code by code against the engine's real bytes, never matched by name prefix, and the older umbrella code they were split out of is asserted to still be **alive** — treating the whole code as retired would make a family of real refusals vanish silently **0.68.3 (core 7.18.0):** two more keys ride the same projection duty as `parks` itself: `originUnconfirmed: true` on a row (never `false`; absence is the confirmed state) and `resumeAdmissionIncomplete: true` on the run (presence means "not a resume base"). Dropping either would turn a refused record back into an admissible one, so the guard pins both, including that neither folds into the other **0.69.1 (CC-12):** both keys now also ride the projected `WorkflowRunState`, so a host that only sees the projection can render them |
|
|
391
391
|
| `scripts/run-retired-vocabulary-census-test.mjs` | Whether a retirement really happened. When upstream removes a family, a downstream package can cut it out or keep a courteous alias — and the alias is the worse outcome: three clients keep writing branches for something nobody emits, and a status line advertises a state it can never reach. Choosing the clean cut only means something if a guard holds it, since a comment saying *retired* is not an exit code. Each registered entry is held two ways: the name must be gone from **code positions** in this package (comments stripped first, because the explanation is supposed to stay) and off the published surface, and — the half that keeps this from being self-congratulation — it must really be gone **upstream**, since that is the entire reason it was removed here; if it comes back, the disposition deserves reconsideration rather than silence. The scanner proves it can speak by finding a symbol that is genuinely present before any absence is believed, and distinguishes a mention inside a comment from one in a string literal, which is exactly the form being cleared. A closing check runs the other way: the retirement **story** must remain in the comments, including a promise this package made earlier and has now had to withdraw — deleting the history alongside the code is a bad way to satisfy *zero hits*, and leaves the next reader with code that has no reason |
|
|
392
392
|
| `scripts/run-classifier-status-test.mjs` | What state the auto-mode classifier is in **on this session** — the question a doctor line, a model settings page and a permission card’s status row all ask, and a different question from the one the approval card asks (*why am I being asked right now*), so the sentences are pinned mutually distinct from that face’s as well as from each other. The session-level half of this reading — a breaker record the engine used to keep — was **retired upstream**, and the guard now holds that retirement from **both** sides: the engine's own declarations must really no longer carry it (a fact coming back would mean the removal here was the wrong disposition, and that deserves a conversation rather than silence), and this package must carry no alias, no state word and no leftover narrowing for it — a reading kept alive for something nobody emits any more is a promise the interface cannot keep, and it left the doctor line advertising a state it can never reach. What remains is ordered by the quantity that actually decides whether the classifier is running: the fact from **this round** first, then whether this leg is armed — a decider is minted per run, so a later leg can be armed again. Not armed, and a section that never arrived, both answer **undefined** rather than *available*; that arming question has its own field and answering it twice grows a second ledger. Arming and availability are also **two words, not one**: the engine says a decider was minted *for this leg*, which is an assembly-time fact, while whether that decider answers any given round is a **per-call** one — so an armed leg reads `armed` and only a positive per-call fact (an ask whose origin is the classifier's own denial-bound fallback, which by construction stands *after* the classifier ran) reads `available`. Every other ask origin is refused as evidence and for a stated reason rather than out of caution: several are ones the classifier is structurally forbidden to answer, and for the rest a surviving ask is precisely the case where it did **not** resolve one — so reading availability off them would be a guess. The projection is a **whitelist**, so an older engine still sending the retired member loses it at the boundary while the two live facts beside it ride through untouched. Rendering never throws and never impersonates: a state word this client does not know — including the retired one, which a restored view can still carry — reaches an honest fallback that names it verbatim, carries no invented explanation of a mechanism that no longer exists, and is proven distinct from all three real sentences; prototype keys reach that same fallback rather than a function body, checked against a real out-of-table word so the comparison cannot hold vacuously |
|
|
393
393
|
| `scripts/run-compaction-boundary-projection-test.mjs` | The compaction divider and the one frame that makes its anchor resolvable. The trigger word is passed through as an **open set** instead of being folded to two: the engine deliberately stopped flattening its third value (a compaction that was not optional — a prompt-too-long recovery or trim pressure) and carries what the hook layer saw, so folding it again at the package boundary re-introduces exactly what upstream had just removed, while a consumer branching on *is it manual* keeps its behaviour byte for byte. Only an unreadable word (absent, empty, non-string) falls back — that is *could not read it*, not *read it and did not recognise it*. Two superset keys ride the metadata and neither fabricates: the preserved-segment anchor is minted only when its id really reads out, because half an anchor sends the host looking up an empty string in its map, and the clamp ratio is a **disclosure** whose real zero is a fact rather than an absence. The clamp ratio also carries a registered exit condition — the service really sends it while the SDK arm has no seat for it yet, so the read is defensive and this guard reds the day that seat appears, forcing a re-check instead of leaving a cast to rot. The committed-message frame moves out of *deliberately not projected*: that classification was true about transcript rows and false about **positioning**, since the engine states that consumers build their own id-to-message map from this frame to place the divider — projecting the anchor without it hands the host something it cannot resolve. It becomes a neutral internal arm and an optional chrome ledger event, never a transcript row (the frame carries no body, so minting one would put words in the engine's mouth), with both required ids narrowed and a malformed frame recorded rather than half-minted |
|
|
394
394
|
| `scripts/run-cost-absence-projection-test.mjs` | Telling **declared free** apart from **never priced**, in both directions, because the package was getting each one wrong in the opposite way. The engine separates them on the wire — an absent cost means some spend had no price table, an explicit zero means the model declared itself free — and the result projector used to require a *positive* number, so a genuinely free run could not say so; while the per-model mirror folded absence to zero, so an unpriced run told a billing consumer it cost nothing. The total is now reported as the engine stated it, with absence and non-finite values alone reading as unknown, and a negative passed through rather than corrected, since a refund is a legal figure and the package is not a second accountant. The per-model figure keeps the CC shape intact — that field is a required number and *unknown* is simply not expressible in it — so the value stays zero and a **companion superset bit** carries the distinction, which means the two are read together and a reader that only ever looked at the number is unchanged; the bit is minted only in the absent case and never as `false`, since a key present with a false value reads as a third state. The same mint point serves both the wire's per-model split and the synthesised current-model row, so neither can drift. Alongside it the cache-write figure stops being a hardcoded zero and reads the field the wire has always carried, in both the flat usage and the synthesised row, and all four flat token slots move from a null-coalesce to a finite-number guard — the stats object has an open index signature and the wire is JSON, so a string or an infinity would otherwise land in a slot the types promise is a number, compiling green and surfacing only when something sums it |
|
|
395
|
-
| `scripts/run-permission-denial-projection-test.mjs` | The terminal result's **permission-denial list** being the wire's real one rather than a hardcoded empty array. The session vocabulary carries a list of tool calls that were denied; the projector used to mint `[]` in both the success arm and the error envelope, which folded two different statements into one — *nothing was denied on this run* and *this frame carries no such ledger at all* (an older engine, a rejection envelope, a failure event that arrives without stats) looked identical. Each denied gate on the wire's human-review ledger now becomes one record, in wire order, carrying the keys the wire can actually honour: the tool name when it reported one, and a superset field with the engine's own short, redacted one-line summary of the call's input. **Two lists, deliberately.** The reference shape requires three fields on every element — tool name, call id, and the full input object — and the wire's ledger carries only the first. Filling the other two with an empty string and an empty object would be invention; putting a half-filled element into the reference array would break the element contract, and a strict consumer validating the stream drops the *whole* result message rather than one field. So the reference array admits only fully-formed records — empty today, and filling itself the day the wire grows the two missing fields, with no code change — while every record the wire really has rides a superset carrier beside it. A contract check pins today's absence, so that day turns this guard red on purpose. The companion bit means *this reference list cannot be claimed complete*: no ledger, an unreadable row, an unrecognised decision word (a rejected plan is not a denied tool call, and a row with no decision at all is not a judgement), or a record that could not be fully formed. Only its absence lets a reader say *zero denials*; it is never minted as `false`. Rows that cannot be read drop themselves rather than the whole ledger, and both arms go through one mint point so they cannot drift. Since 0.73.4 the third CC key is sourced from the same stream's `tool_start` frame, joined by call id: a row joins only when the frame was seen on this stream, its arguments are a plain object, and
|
|
395
|
+
| `scripts/run-permission-denial-projection-test.mjs` | The terminal result's **permission-denial list** being the wire's real one rather than a hardcoded empty array. The session vocabulary carries a list of tool calls that were denied; the projector used to mint `[]` in both the success arm and the error envelope, which folded two different statements into one — *nothing was denied on this run* and *this frame carries no such ledger at all* (an older engine, a rejection envelope, a failure event that arrives without stats) looked identical. Each denied gate on the wire's human-review ledger now becomes one record, in wire order, carrying the keys the wire can actually honour: the tool name when it reported one, and a superset field with the engine's own short, redacted one-line summary of the call's input. **Two lists, deliberately.** The reference shape requires three fields on every element — tool name, call id, and the full input object — and the wire's ledger carries only the first. Filling the other two with an empty string and an empty object would be invention; putting a half-filled element into the reference array would break the element contract, and a strict consumer validating the stream drops the *whole* result message rather than one field. So the reference array admits only fully-formed records — empty today, and filling itself the day the wire grows the two missing fields, with no code change — while every record the wire really has rides a superset carrier beside it. A contract check pins today's absence, so that day turns this guard red on purpose. The companion bit means *this reference list cannot be claimed complete*: no ledger, an unreadable row, an unrecognised decision word (a rejected plan is not a denied tool call, and a row with no decision at all is not a judgement), or a record that could not be fully formed. Only its absence lets a reader say *zero denials*; it is never minted as `false`. Rows that cannot be read drop themselves rather than the whole ledger, and both arms go through one mint point so they cannot drift. Since 0.73.4 the third CC key is sourced from the same stream's `tool_start` frame, joined by call id: a row joins only when the frame was seen on this stream, its arguments are a plain object, and the scan over them finishes within budget. Since 0.84.1 arguments the transport redacted on the way (a string leaf carrying a replacement token, or a cycle / depth placeholder) join as well: the reference list receives the very object the transcript's tool call shows, and that entry, on both lists, carries a superset flag saying the input is a redacted view rather than the original — redacted arguments that are over budget, not a plain object, or never seen on this stream still stay off the reference list, and the flag is never minted as false; both halves have positive controls (a fully joined list drops the discriminator, a partially joined one keeps it), the ledger's own input wins when present, the snapshot is per-stream and capped with a one-way overflow latch, and an id seen with two different argument objects never joins. Later sections add the second stream-local join and the two discriminators the headless exit-code rule needs. "Which layer denied this" is not on the denial ledger at all — it is on the gate record of the same call's close-out frame, so it is joined by call id under the same law as the arguments: the closed word table is checked on the collecting side, the ledger's own value wins if it ever arrives, a word from outside the table is not stamped, and a row that cannot be joined keeps the key absent rather than claiming nobody denied it. The classification word is carried on both lists under the same name and the same value, so a consumer needs one reader, not two. The "this run produced no tool output and was denied" flag is present only when three independent things hold at once — the denial evidence is read from the full list rather than the strict one, which can be empty for reasons that have nothing to do with denials; this stream saw no successful tool close-out; and this stream can honestly claim to have watched the run from its first frame. A stream that reconnected mid-run cannot make the last claim, so it mints nothing rather than a false negative, and the flag is never minted as false. From 0.80.0 that classification has a **second source**. It used to come only from this package's own decision path, so a refusal the engine settled on its own — a deployment policy answering the card on an unattended lane, with no client involved — left the field empty even though the same stream's gate record said exactly what had happened. The engine's own settlement word now fills it when, and only when, the local one is absent: the package's own attribution always wins, because letting a replayed frame overwrite it would let the wire change what the host itself said. The word is read literally in both directions and never reverse-engineered, and the separate field naming *which layer* refused is left exactly as the wire wrote it — the two answer different questions, and rewriting one to match the other would make them say the same thing twice. A later section reconciles the terminal list against the denied calls seen in the same stream, so denials that never reached a human (rules, hooks, classifiers, write protection) are listed too: complete rows join the CC list, rows missing a field stay on the extended list and mark it incomplete. A further section feeds the same raw tool-result frame through the projector into both lanes and requires the denial category on the interactive transcript record, on the non-interactive frame and from the reader on the raw frame to agree, including frames whose settlement, classifier cause or classifier attribution is present but unreadable — a shape the narrowed gate view on the internal frame cannot show — and pins that the reader gives the same answer on the internal frame as on the raw one. |
|
|
396
396
|
| `scripts/run-cost-reconcile-projection-test.mjs` | The **end-of-run cost reconciliation** reaching consumers at all. The engine splits a run's spend on the wire — the task's own cost, which deliberately excludes delegated sub-agents, the delegated total itself, and the within-task compaction subtotal that sits inside the own figure — and states two reconciliation identities for them. The package used to project none of it, so a cost view could only ever see one number and under-reported both delegated and compaction spend. Both structures are now projected onto the result as superset fields in the wire's integer micro-currency unit, read key by key, with unreadable keys dropped individually, an entirely unreadable structure omitted rather than emitted empty, and unknown categories passed through since the vocabulary belongs upstream. The delegated cost stays **absent when it was never priced**, never a fabricated zero. The same reader also feeds a terminal chrome arm carrying the three parts plus the reconciled total, so the two faces can never compute different answers; the reconciled total is minted only when both sides are known, and otherwise a discriminator bit says which side is unknown. **The reference field for total cost keeps its meaning** — it remains the task's own spend and the delegated total is not folded into it — because that is a shape the wider ecosystem reads; the reconciled figure is offered beside it, not in place of it. A frame that carries no stats emits no arm at all, and the existing rule that in-stream per-turn usage is not published for sub-flows is pinned unchanged, since delegated spend arrives once, at the end. The bit that says those figures are a lower bound is **per stream**, not per context: the emit context belongs to the caller and may be reused across streams, so a gap observed on one run is no evidence at all about the next one — the observation is held for the duration of one stream and handed to both projection faces by value, and the guard drives a reused context both sequentially and concurrently to prove neither direction leaks |
|
|
397
397
|
| `scripts/run-task-progress-terminal-projection-test.mjs` | The one tick that says a delegated child **finished**. The engine fires exactly one final beat carrying a terminal face, and says in the same breath why it exists — so a consumer sees the row finish instead of watching it vanish after the last running beat — but the package's projection whitelist had no seat for that field and its adapter still carried the older premise in a comment, so the terminal beat arrived byte-identical to another running one: the panel row stayed up waiting for a defensive sweep (which only ever settles rows bound to a card still open this turn) or for a separate notification frame. The status now rides through as an **open set** with the vocabulary left upstream, while the question *which words are terminal* is answered by a closed pair on the adapter side — an unrecognised new word takes the running path, because guessing it terminal ends a row that is still working whereas one extra running beat merely renders late. A terminal beat settles the row directly under the lane proof its binding gives it (not the main lane a notification would use, and not by card id, since the engine is naming a child rather than closing a card), freezes the inline group-row twin in the same beat so a later sweep cannot reset the real tool count, clears the session-resident ledger, and fires the stop hook only for a child whose start really fired. It does not mark the row live or emit a second progress beat, and it shares the settled-row ledger with the other two settle legs so a replay or a double-delivery cannot produce a second end. Three things are pinned **unchanged**: a running beat, an absent status (older engines never send the field, and reading absence as terminal would make every child row disappear on its first beat), and the workflow lane gate, which still runs before any of this |
|
|
398
398
|
| `scripts/run-assistant-arm-identity-test.mjs` | The identity keys on an assistant row, and an explicit account of the two that are **deliberately not** there. What the renderer received was a bare role-and-content object, so a dozen consumer sites downstream were each estimating what the message envelope should have told them. The id is taken from the engine's own event id rather than minted locally, because it has to be **the same value** on the live leg and on a durable replay — a freshly minted one would make a replayed message look new to a host's dedup and to rewind — and when the wire carries none the key is simply absent rather than filled with a random stand-in wearing an identity it does not have; it is also kept distinct from the envelope's own local render key, which is a different identity. The model name comes from what the host pinned when it opened the stream (the request was the host's to build) and is never guessed, since a wrong model name is worse than none once a billing or capability face looks it up. Usage and stop reason are **not** minted on this arm, and the reason is frame order rather than effort: content arms arrive before the turn's closing frame, so at the moment the arm is emitted the engine has not yet said what the round cost — anything put there would be an estimate, which is the very thing this work exists to remove — and synthesising a follow-up assistant update when the real figure lands is also refused, because that shape does not exist upstream and would place a message in the transcript the engine never sent. Their real values leave through the turn's own neutral arm as two superset keys, the usage one reusing the **same single mint point** the footer rollup already folds so the two faces cannot diverge, and the stop reason passed through verbatim as an open set — the machine signal for *was this turn cut short*, previously blind on both the stream and the trace. The existing behaviours beside them are pinned too: no arm at all when usage is wholly absent, and the sub-flow cut-out that keeps a child's turn from driving the leader's face |
|
|
@@ -412,7 +412,7 @@ guard still cross-checks the table by name).
|
|
|
412
412
|
| `scripts/run-registrar-tables-test.mjs` | The four registrar table bodies — this Guards table and the three census tables in the repository's negative-control record — are **generated** from `scripts/gates-manifest.json`, the one file that describes a suite. Every row's text must equal what the generator emits, the manifest's suite set must equal the suites on disk, each entry must declare how it is negative-controlled (rehearsed, blind, or behavioural, with the census taker itself declared as such since it does not appear in its own tables), and each row's outward prose is scanned against the published-surface word list — the same list the packaging-hygiene guard uses, shared rather than copied — before the generator may write it into this file. Adding a guard is therefore one manifest entry plus one generator run instead of six hand edits across three files, and a description that drifts in one place and not the others stops being expressible. The row count is no longer what is compared: the earlier arrangement checked the census tables by **length**, so rows naming the wrong suites reconciled green. Positive controls run entirely on in-memory copies — a changed description, a dropped entry, an added entry, a changed class and a hand-edited row on disk each have to make the same judgement speak — and the quieter halves are pinned too: nothing outside a table body may move, a line inside one that is not a recognisable row makes the generator refuse rather than drop it, byte equality is backed by a column-count check (a cell holding a bare pipe splits a row into extra columns, and a code span does not protect it), and the malformed rows kept byte-for-byte as they are found are registered individually, so the registration turns red the day it stops being needed rather than outliving its reason — audited in both directions, since a registration pointing at a row that is no longer malformed and one pointing at a guard that was reclassified or deleted are both exemptions nobody reads |
|
|
413
413
|
| `scripts/run-mcp-probe-face-test.mjs` | The engine's **run-free MCP status face** (server ≥7.93.0, sdk 11.2.0): the `capabilities.mcpProbe` bit read the same four-state way as its eleven sibling capability readers, and two call ports on top of it — read the deployment's own declared servers, or probe a caller-supplied list. Presence of the bit is carried by the engine version, so an **absent key means an older engine** (that route answers a coded 404) and is read as *not reported*, never as *no*; an explicit `false` is the engine's own no and is reported without inventing a reason (whether caller-supplied declarations are accepted is a different bit's question); a non-boolean is malformed and is dropped rather than folded into a no. The availability verdict answers only *should this call go on the wire*: an explicit no means zero requests, and both kinds of *cannot tell* are sent anyway, so an older engine answers with its own coded refusal instead of being silently skipped. One failure judge serves both ports and asks **provenance before status**: a 4xx, 501 or 503 that carries no machine code proves nothing about who answered and is reported as *no verdict*; twelve coded refusals each get their own arm (three identity codes, one of which sits outside the `auth.` family so a prefix fallback would miss it; three different treatments ride the same 400), and a coded answer this version does not recognise lands in *cannot tell*, never in *this engine has no such face*. Retry-after seconds ride 429 and 503 and are absent rather than 0 when the engine gave none. The 200 body is narrowed through the **same per-row narrower** as the streaming `wiring_manifest.mcp[]` leg and the session panel's replay leg, liveness cell included; on this face rows are paired with the submitted declarations **by index**, so a dropped row makes the whole answer unreadable rather than a half table, a row count that differs from the submitted count is reported as misaligned, and an honest empty `servers: []` is kept apart from *non-empty but nothing readable*. `probedAt` and `ttlSec` pass through untouched — this package mints no freshness verdict — the declaration list is handed over as-is with its length snapshotted once, a list that serialises itself differently from what was counted is refused locally with zero requests, and no port ever retries a dial. |
|
|
414
414
|
| `scripts/run-sdk-wire-transit-test.mjs` | The package's pass-through of a few SDK names (`sdkWireTransit`), pinned: every value re-export is the **same reference** as the SDK's own (a client class the host recognises with `instanceof`, the two approval-frame predicates, the three session-bundle calls and the SDK error class), not a look-alike wrapper — wrapping would discard the one anti-drift guarantee a pass-through has; every type re-export is present by name in the emitted declaration file; the gate's list and the source file's export lists are compared in both directions so a name added to one without the other turns red; and a name the SDK does not export must fail the same test, so the gate is not vacuously green. |
|
|
415
|
-
| `scripts/run-sdk-registry-transit-test.mjs` | The package's cloud control-plane surface (`sdkRegistryTransit`), which lives behind its own `./registry` subpath entry point rather than on the root barrel, and this guard holds both halves of that decision. Upstream publishes the same surface behind a subpath of its own, because the subject differs: the engine-wire surface speaks for one engine's service credential, this one for a person's rotating token, and their refresh and error semantics were deliberately never merged. Keeping it behind a second entry point means a client that never touches the control plane neither resolves nor type-checks it. Every value re-export is therefore read from the file the subpath entry actually resolves to, and must be the **same reference** as upstream's own (the control-plane client class, the three config reads, the health probe, the feedback call, the auth-path constant, the content-address helper, and the two typed error classes a host recognises with `instanceof`); every type re-export is compared with the emitted declaration file in both directions; the gate's list and the source file's export lists are likewise compared both ways; and the root entry is checked to carry none of these names, with the root barrel's own source checked to reference the entry file nowhere — a name leaking onto the root would put the cost of this surface back on clients that never asked for it. The subpath is then verified end to end: the installed SDK must really publish its own `./registry` entry and declare every transited name inside it, and this package's own `exports` must point that subpath at exactly the files the gate just judged. Portability is two checks rather than one, done with a parser rather than a text search: the upstream subpath's emitted JavaScript, walked recursively, must contain neither a `node:` specifier (static, side-effect, dynamic, `require` and re-export forms all exercised) nor a Node **global** — because the same package's third entry point is a Node-only surface that imports nothing at all and reaches for the `Buffer` global, so a specifier check alone would pass it as isomorphic, while a byte-level search of it reports two `node:` hits that live entirely inside a documentation example. A text scanner that merely strips comments first gets both directions wrong on ordinary JavaScript — a regular-expression literal containing a slash pair swallows the rest of its line, and the word in a string reads as a reference — so both scanners run off the syntax tree and are checked against fixtures for each failure direction as well as against that real material — including a dynamic import written with a template literal, which a check that accepts only quoted strings misses entirely, and a dynamic import whose target cannot be determined statically, which is refused rather than read as no edge at all. Zero-processing is likewise enforced with a syntax-tree allowlist rather than a keyword search: every top-level statement must be a named re-export carrying that one specifier, so an import followed by an in-place edit of the upstream prototype is refused with a file and line — that shape leaves the name lists untouched and even keeps the same-reference check green, since both sides are then the one object that was damaged. The last section measures, rather than merely notes, one declaration-level gap: two of the upstream subpath's declaration files reference a package the SDK lists only among its own dev dependencies. Moving such a check to a scratch directory is not isolation, because package resolution walks the ancestor directories, so the gate builds a sandbox served by a restricted compiler host and proves the isolation both ways — a decoy copy of the missing package placed one level above the sandbox must silence the errors for an unrestricted host and must not silence them for the restricted one. Then it installs this package into that same sandbox as a real consumer would, and pins the two readings that justify the entry-point split: a consumer that imports only from the root sees no unresolved-module errors at all, while a consumer that imports the subpath sees exactly the two, reported honestly rather than swallowed by this layer. The day upstream ships those declarations, that section turns red and the
|
|
415
|
+
| `scripts/run-sdk-registry-transit-test.mjs` | The package's cloud control-plane surface (`sdkRegistryTransit`), which lives behind its own `./registry` subpath entry point rather than on the root barrel, and this guard holds both halves of that decision. Upstream publishes the same surface behind a subpath of its own, because the subject differs: the engine-wire surface speaks for one engine's service credential, this one for a person's rotating token, and their refresh and error semantics were deliberately never merged. Keeping it behind a second entry point means a client that never touches the control plane neither resolves nor type-checks it. Every value re-export is therefore read from the file the subpath entry actually resolves to, and must be the **same reference** as upstream's own (the control-plane client class, the three config reads, the health probe, the feedback call, the auth-path constant, the content-address helper, and the two typed error classes a host recognises with `instanceof`); every type re-export is compared with the emitted declaration file in both directions; the gate's list and the source file's export lists are likewise compared both ways; and the root entry is checked to carry none of these names, with the root barrel's own source checked to reference the entry file nowhere — a name leaking onto the root would put the cost of this surface back on clients that never asked for it. The subpath is then verified end to end: the installed SDK must really publish its own `./registry` entry and declare every transited name inside it, and this package's own `exports` must point that subpath at exactly the files the gate just judged. Portability is two checks rather than one, done with a parser rather than a text search: the upstream subpath's emitted JavaScript, walked recursively, must contain neither a `node:` specifier (static, side-effect, dynamic, `require` and re-export forms all exercised) nor a Node **global** — because the same package's third entry point is a Node-only surface that imports nothing at all and reaches for the `Buffer` global, so a specifier check alone would pass it as isomorphic, while a byte-level search of it reports two `node:` hits that live entirely inside a documentation example. A text scanner that merely strips comments first gets both directions wrong on ordinary JavaScript — a regular-expression literal containing a slash pair swallows the rest of its line, and the word in a string reads as a reference — so both scanners run off the syntax tree and are checked against fixtures for each failure direction as well as against that real material — including a dynamic import written with a template literal, which a check that accepts only quoted strings misses entirely, and a dynamic import whose target cannot be determined statically, which is refused rather than read as no edge at all. Zero-processing is likewise enforced with a syntax-tree allowlist rather than a keyword search: every top-level statement must be a named re-export carrying that one specifier, so an import followed by an in-place edit of the upstream prototype is refused with a file and line — that shape leaves the name lists untouched and even keeps the same-reference check green, since both sides are then the one object that was damaged. The last section measures, rather than merely notes, one declaration-level gap: two of the upstream subpath's declaration files reference a package the SDK lists only among its own dev dependencies. Moving such a check to a scratch directory is not isolation, because package resolution walks the ancestor directories, so the gate builds a sandbox served by a restricted compiler host and proves the isolation both ways — a decoy copy of the missing package placed one level above the sandbox must silence the errors for an unrestricted host and must not silence them for the restricted one. Then it installs this package into that same sandbox as a real consumer would, and pins the two readings that justify the entry-point split: a consumer that imports only from the root sees no unresolved-module errors at all, while a consumer that imports the subpath sees exactly the two, reported honestly rather than swallowed by this layer. The day upstream ships those declarations, that section turns red; the registered count is then updated and the section stays as a guard against a regression. The separation itself rests on the root closure being computed correctly, so the portability guard that computes it was extended in the same change: a template-literal dynamic import is followed like any other edge, and an edge whose target cannot be resolved statically is refused on every one of the four graphs — without that, a single line in a third file already reachable from the root would put this surface back into the root runtime while every guard stayed green. Since 0.84.0 the upstream subpath ships those declarations itself: the measured count is now zero, the section stays as a guard (a new missing declaration package turns it red again), the isolation proof uses a synthetic probe that imports only the decoy, and the gate also checks that no upstream subpath declaration references the missing package any more. |
|
|
416
416
|
| `scripts/run-rule-removal-consequence-test.mjs` | The one sentence a rule-removal confirmation surface shows for what removing the rule will actually do: one line per behaviour the rule could have been enforcing, plus a neutral line for when that behaviour cannot be read back, so a caller that hits an out-of-set or missing value never falls back to a specific claim it cannot support. The guard compares the actual output against frozen text rather than merely checking that some string came back, so a dropped word or a swapped clause is caught the day it lands, and the four sentences are pinned pairwise distinct. The line for a rule that was denying something is pinned to say the removal widens what can run rather than echoing the wording used for a rule that asks again — the two are opposite directions, and sharing a sentence between them would tell the person confirming the removal the opposite of what is about to happen. The neutral line is checked from the other side for the same reason: it must not contain a word that belongs to only one of the three behaviours, because that would answer on behalf of a state the caller was unable to determine. The lookup that turns a raw stored or transmitted value into one of the three behaviours reads by strict equality only, proven with a fully trapped proxy and a counting getter to show it never touches a property on whatever it is handed, so a value with a legitimate-looking word sitting on its prototype chain is rejected exactly like any other out-of-set value rather than being unwrapped. A closing self-check mutates one character out of each frozen sentence and asserts the exact-match comparison actually fails on it, so the guard cannot pass by checking only that a string of some kind came back |
|
|
417
417
|
| `scripts/run-mcp-engine-leg-test.mjs` | The two legs behind one row on the MCP detail card. In a two-process setup the servers are hosted by the **engine**, while the Status cell on the card reports **this client's own** connection to them, and the two are independent truths: this client failing to connect does not mean the server's tools are unavailable, and the engine leg on the same screen may be saying it completed an exchange moments ago. Two mints share the work. The first reads one server's liveness off the engine leg as six readings, and its load-bearing distinctions are three. **No roster that could be read end to end** (`null`, `undefined`, anything that is not an array, a length that is not a non-negative integer, a length beyond the scan bound, or rows that could not be read at all with nothing matched) is **not** the same as an **empty** roster, which is the engine leg's positive statement that it declared no servers; the same narrower that feeds this reader answers `undefined` when a non-empty section yields no readable row, so *I could not read it* never turns into *I know it is zero*, and a roster that was not read to the end never produces *this server is not on it*. **Could not be read is not the same as absent**: absence means only that the row carries no liveness key of its own, while a record that is present but unintelligible — a cell that is not an object, a word that is not a non-empty string, or a read that fails outright — is reported as unreadable and is decided before the observation beside it, since a record that may well have said the opposite is no evidence of reachability. **An ambiguous name is answered as ambiguous**: when a roster that was read end to end matches a name more than once, the reader reports the match count instead of picking one, because the upstream contract for the sibling MCP face states that names are not guaranteed unique, and picking optimistically would contradict the worst-fact-first verdict shown on the same screen. When a row's identity could not be read at all the reader makes **no claim about that name whatsoever**, not even a count, since the row it could not read may well be a second one carrying the same name — a row that is plainly not a row, such as a hole in a sparse array, is a different matter and leaves the roster complete. The observed word is passed through **verbatim as an open set**, so a fourth word one day arrives at consumers untouched. The second mint is the one sentence that goes under Status, and it appears **only** when this client's own connection really did fail — the other Status values already tell the truth, and a sentence on top of them would only muddy the verbatim health vocabulary. **The liveness observation outranks the tool roster, and having heard from the engine includes hearing that it could not tell, and hearing something unreadable**: a word meaning *looked and could not tell*, a word this version does not recognise, and a record that arrived but could not be read each get their own sentence rather than falling back to the roster, because falling back would say this client holds nothing at all while the diagnostics page shows that very record. A value that is not a reading at all is a different case and does fall back, since nothing then establishes that the engine sent anything. Only when liveness genuinely cannot speak does the roster get a turn, and the roster itself has four answers — tools were listed, nothing was said, the engine reported zero, and the count could not be read — because **unreadable, absent and zero are three different facts**. No sentence claims that nothing at all has been seen about the server, because on two of the readings that reach the roster the engine has plainly listed it; the sentence for a silent roster states only what this client holds. The nine sentences are pairwise distinct, each one names the engine leg, and **none of them renders the liveness word itself**. Membership of the word list is decided by the one shared predicate rather than a second copy, and the branch over the known words is pinned so that a new word upstream fails the build instead of silently taking the *not recognised* sentence. Neither mint ever throws, whatever a host hands it: every value is taken through one reader that accepts **own properties only** — an inherited key is not a wire fact, and one on a shared prototype could otherwise manufacture an engine observation or suppress a real one — reads each key exactly once, and keeps a read that fails apart from a value that is absent. The roster is walked by index rather than through the array's own `find`, the array test is guarded because it can throw on its own, and a length that overstates itself would otherwise spin forever |
|
|
418
418
|
| `scripts/run-cc-message-key-census-test.mjs` | The one rule behind CC-shaped messages this package emits: every top-level key on a `user` / `assistant` / `result` / `system` message must be a member the CC SDK mirror (`@sema-agent/agent-types`) declares for that same arm, or one of this package's `_sema_`-prefixed additive keys, or a row in a dated transition table that goes red the day its retire version arrives. Mint sites are found syntactically (identifier, quoted and computed `type` names alike) and their key sets are resolved syntactically too — inline literals, both branches of a conditional, the nullish-coalescing and logical or/and operators, `const` initializers and every `return` of a helper — so a type assertion, a `Partial<Pick<…>>` narrowing or a computed name cannot launder a key past the check, while a spread the tool cannot follow (a parameter, a `let`, a member access) fails the tool, never the product. Arms that carry a `subtype` discriminator are checked against that subtype's own member set, so a key declared only for another subtype does not pass on the strength of the arm-wide union. The runtime section drives the real adapter pipeline and pins the tool-result record's SDK-spelled `tool_use_result` (the transitional camelCase twin rides along by reference until 0.83.0). A second runtime section feeds the same tool-result frame to both the interactive adapter and the non-interactive frame builder and requires the structured-result key and the no-output marker to be present together, absent together and equal on the two outputs, while the transitional camelCase name stays on the transcript record only; a static section checks a per-key routing table for the internal tool-result frame against the key sets resolved at the three mint sites, in both directions, and requires the non-interactive frame's structured-result value to trace back to the very same syntax node the transcript record uses — so recomputing or copying that logic on the non-interactive path fails even when the values happen to agree. |
|
|
@@ -428,11 +428,18 @@ guard still cross-checks the table by name).
|
|
|
428
428
|
| `scripts/run-layering-shadow-export-test.mjs` | Same-name shadows across the first-party clients that consume this package (terminal, desktop, web and the admin console). Each client's product sources are read at the local clone's `origin/main` (or its HEAD when there is no such ref), without fetching, and parsed with the TypeScript parser; every top-level runtime export the client declares itself is compared with this package's public runtime exports. The guard prints which ref, commit and commit date it read for each client, and warns (without failing) when that commit is more than seven days old, because the result then only describes that older snapshot. A client-side declaration carrying the name of a package export means a piece of shared logic now lives in two places and can drift apart. It fails the guard unless it is listed in `scripts/layering-shadow-exemptions.json`, and a listed row must carry a retire-by version no more than three minor lines ahead (it fails once the package reaches it). It also fails once the client has removed the shadow and the row still stands. Re-exports of this package's own exports are the intended form and never count. A client tree that is not present is reported as a skipped section, not as a pass. The ruler proves itself on an in-memory fake client (planted shadows must be caught, legal forms must not), on a throwaway repository (a missing `origin/main` falls back to HEAD, a broken one is a fault rather than a silent fallback), and refuses to report zero on a client whose scan surface is empty. |
|
|
429
429
|
| `scripts/run-session-policy-deliverable-test.mjs` | Which of a batch of user-written permission rules can be written into a session’s own rule record without changing their meaning, and why each of the others cannot. The record holds whole tool names and command names only, so exactly one class maps across losslessly: a deny rule that names one tool with no qualifier. Everything else is withheld with one word from a closed six-word list — an ask rule (the record has no ask tier), a deny rule with a parenthesised qualifier (recording just the name could block more), a rule that names a server or agent peer without naming one of its tools (for every protocol namespace the engine knows, checked against the engine package's own table) or contains a wildcard (*) anywhere (an engine that compares exact names would block nothing), an entry that is not a tool name, and a name the engine refuses at the start of every run — a retired tool name, or one containing "__" without a protocol prefix, where the prefix check is case-sensitive (once such a name is in the record, every later run of the session fails at startup until that entry is removed; the retired-name list is checked entry by entry against the engine package's own list, and a withheld retired name carries its current name when the engine says it was renamed) — and each word has one sentence, which never echoes the rule itself; asking for the sentence never throws, even with a value that throws when turned into a string. A name with leading or trailing whitespace counts as not a tool name: the record compares exact bytes, so it would block nothing. The guard pins the batch semantics: the deliverable part is either the whole batch or empty, never a subset, so a caller cannot send half a change and report it as saved. An end-to-end check runs the engine package itself: every batch this function would deliver — the recorded vectors and a fixed-seed sample of generated names — is written into an in-memory session rule store and the next run must get past its start-up checks and reach the model, while every name withheld as refused — every retired name included — must indeed make that run fail at start-up, and every string literal in the judgement source that it withholds as refused must be one the engine package's own tables refuse. It also checks that malformed input never throws and never delivers anything (non-arrays, non-string entries, holes, a polluted array prototype, a length or index that throws, a changing index read once), that a batch which cannot be read at all is marked `unreadable: true` while an empty batch is not, so the two stay tellable apart, that each word is produced by some vector and nothing outside the list is produced, and — when a checkout of the previous in-client implementation is present — that this function gives the same answer on every recorded vector and on tens of thousands of generated rules and pairs, except for four deliberately stricter classes (whitespace-padded names; rules with a wildcard anywhere, which the previous implementation sent as exact names unless the wildcard was the whole tool part of a server rule; peer-wide rules outside the MCP namespace, which it did not recognise; and names the engine refuses at start-up, which it sent as ordinary names), whose disagreements are counted per class and must match an independent count exactly. |
|
|
430
430
|
| `scripts/run-plugin-hooks-projection-test.mjs` | Plugin hooks: each command hook an enabled plugin declares is decided one by one as running in the engine, running in this client, or not running at all, and the page of hooks sent with a request is built from the same per-turn plan the client uses to skip its own copies, so one hook never runs in two places. Governance is judged first and always wins — a managed hooks switch-off, an untrusted workspace, safe or bare mode, or a governance read that fails sends no plugin hook and does not list it as a gap; managed-hooks-only (set directly, or through a merged non-managed hooks switch-off) keeps only managed plugins; the plugin-only customization lock does not touch plugin hooks. A hook reaches the engine only when this client started the engine on this machine, the engine reports plugin-hook support, the entry is a command, the plugin declares no sensitive option, and the event still fits the engine's per-event limits; the gate walks that matrix cell by cell, including the limit boundaries and a session goal hook counting toward them. A fact that was never read is reported as not known rather than as a fact: a host that does not say where the engine runs gets a "not known whether this client started the engine" reason, an engine whose capabilities have not been read yet gets a "not known yet whether it supports plugin hooks" reason, and the plan's two engine facts are null in those cases, not false. Events the engine never fires run only if the client says it fires them itself, and hooks the upstream behaviour itself refuses (option references in a shell-form command, an unset option in exec form, malformed entries) run nowhere. Exec-form arguments are passed element by element with only saved non-sensitive option references filled in; path placeholders are left for the executor. Sensitive option values never reach the request: with a host that wrongly supplies one, every string in the plan, the request body, the notice, the labels and the log is searched for it across eight cells. A host without the plugin reader keeps the previous request body and gets exactly one warning per settings port; plugin data that throws while it is being read (a throwing getter, a revoked proxy) is treated like a failing reader — no plugin hooks this turn, settings hooks still sent, nothing thrown; the not-running notice names the plugin and events, never a command or an option value, and escapes control characters in names. Command hooks from settings that carry arguments (a non-empty `args` array, which is the exec form, or any other non-null value) are removed from the request until the engine reports support for arguments, because the engine would otherwise drop the arguments and run the bare command through a shell; an empty `args` array is not treated as carrying arguments when the command is made only of letters, digits and `_ . / : + -` (the shell runs the same executable), so such a guard still reaches the engine, while an empty array on a command with spaces or shell characters is removed; `args` on a prompt or http entry, a null `args`, or an entry with no type is left alone, and those go out unchanged. MCP tool hooks, which the engine cannot parse, are removed only from a request built from a plan, whose not-running notice the host shows; a request built without a plan still carries them on engine-fired events, so the engine rejects the whole request loudly instead of a guard hook silently not running — the gate checks both request bodies against the engine's own hooks schema. Without a plan, every removed hook of that kind on an engine-fired event produces one warning per settings port, event and reason. Malformed entries still pass through for the engine to reject loudly, and passing null where the options object goes behaves like passing nothing; a `plan` option that is not a plan is ignored rather than turning the whole page into nothing, and a plan passed directly in place of the options object is recognised and used. A `plugin` key written by hand on a settings hook is stripped before sending (even when its value is undefined), because only hooks that come from the plugin reader may carry plugin context; the settings document itself is left untouched and a debug line records the count. The two hand-copied tables, the engine-fired event list and the engine limits, are checked against their owners. |
|
|
431
|
-
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. Since 0.83.5 the outlet has two opt-in switches and a position read-out. `escapeBackslashes` (escape form only) writes every literal backslash as a pair, so each output decodes back to exactly one input (a real invisible character and its literal six-character spelling no longer look alike); a reference decoder round-trips thousands of random strings, the output is byte-identical to the default when the input has no backslash, a length cap is measured on the paired output and keeps the longest fitting prefix, credentials are washed exactly as in the default, and the switch is not idempotent by design (use it only at the final render). Zero-width joiners and non-joiners are kept only inside emoji sequences drawn as emoji (so not between symbols such as © or ™ that display as text) and between letters of scripts where they change the shaping (joining scripts such as Arabic, and the Brahmic family), each listed script checked both ways; the one other place a joiner is kept is right after a virama at the end of a word, the older spelling still found in Malayalam and Bengali text. Next to Latin, Cyrillic, CJK and other letters, next to modifier letters shared across scripts, at the start of a word, at the end of a word without a virama before it, or on their own they are now marked. `blanks` marks characters that look like a space but are not an ASCII space (no-break and other width spaces, the ideographic space, Hangul fillers, the blank Braille pattern) before whitespace folding, for names that must never look alike. `displayUntrustedMarks` returns the same text plus the position of every character mark; its text is compared with the outlet over thousands of inputs. With both credential nets off, the character face stays byte-identical to the previous release for input without joiners. The credential nets read escape sequences the way a terminal would when one is cut short: an unfinished colouring or character-set sequence interrupted by another one is dropped as a whole, a sequence never takes the `@` of an address as its final character, and a final character that starts a well-known secret literal (`sk-`, `ghp_`, `AKIA`, a JWT) is also read as the start of that literal; escape tokens this outlet writes are read as one unit, while look-alike text it never writes (an upper-case `\U`, or a code point it never marks) is read as plain text. A URL is cut before a run of non-ASCII blanks followed by a credential label or scheme word, Hangul fillers and the blank Braille pattern count as spaces around a label's separator, and a value that itself starts with a quoted label (`token= "password":"..."`) is left to that inner label. With `blanks` on, a blank written as an escape token right after a label is read as a blank when the value is judged, so `password:` followed by a no-break space and `missing` stays unmasked and an empty value gets no marker; the credential nets also read the text the way it looks after default whitespace folding and combine what each reading masks, so whatever the default form masks stays masked with `blanks` on (checked over a seeded corpus for the escape form, with paired backslashes and without folding; the exceptions are text that itself contains a literal six-character blank escape, which cannot be told apart from one the outlet wrote, and the dot and space marks, which cannot tell a marked blank from a real dot or space). A lone surrogate wedged between a label, its separator and its value no longer hides the value: the credential nets treat it exactly like a format character, in the machine-readable wash, in a single pass of the display outlet, and on the end-of-run rows and the result frame's `errors[]` on both lanes, while lone surrogates anywhere else are left byte-for-byte. |
|
|
432
|
-
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. |
|
|
433
|
-
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. |
|
|
431
|
+
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. Since 0.83.5 the outlet has two opt-in switches and a position read-out. `escapeBackslashes` (escape form only) writes every literal backslash as a pair, so each output decodes back to exactly one input (a real invisible character and its literal six-character spelling no longer look alike); a reference decoder round-trips thousands of random strings, the output is byte-identical to the default when the input has no backslash, a length cap is measured on the paired output and keeps the longest fitting prefix, credentials are washed exactly as in the default, and the switch is not idempotent by design (use it only at the final render). Zero-width joiners and non-joiners are kept only inside emoji sequences drawn as emoji (so not between symbols such as © or ™ that display as text) and between letters of scripts where they change the shaping (joining scripts such as Arabic, and the Brahmic family), each listed script checked both ways; the one other place a joiner is kept is right after a virama at the end of a word, the older spelling still found in Malayalam and Bengali text. Next to Latin, Cyrillic, CJK and other letters, next to modifier letters shared across scripts, at the start of a word, at the end of a word without a virama before it, or on their own they are now marked. `blanks` marks characters that look like a space but are not an ASCII space (no-break and other width spaces, the ideographic space, Hangul fillers, the blank Braille pattern) before whitespace folding, for names that must never look alike. `displayUntrustedMarks` returns the same text plus the position of every character mark; its text is compared with the outlet over thousands of inputs. With both credential nets off, the character face stays byte-identical to the previous release for input without joiners. The credential nets read escape sequences the way a terminal would when one is cut short: an unfinished colouring or character-set sequence interrupted by another one is dropped as a whole, a sequence never takes the `@` of an address as its final character, and a final character that starts a well-known secret literal (`sk-`, `ghp_`, `AKIA`, a JWT) is also read as the start of that literal; escape tokens this outlet writes are read as one unit, while look-alike text it never writes (an upper-case `\U`, or a code point it never marks) is read as plain text. A URL is cut before a run of non-ASCII blanks followed by a credential label or scheme word, Hangul fillers and the blank Braille pattern count as spaces around a label's separator, and a value that itself starts with a quoted label (`token= "password":"..."`) is left to that inner label. With `blanks` on, a blank written as an escape token right after a label is read as a blank when the value is judged, so `password:` followed by a no-break space and `missing` stays unmasked and an empty value gets no marker; the credential nets also read the text the way it looks after default whitespace folding and combine what each reading masks, so whatever the default form masks stays masked with `blanks` on (checked over a seeded corpus for the escape form, with paired backslashes and without folding; the exceptions are text that itself contains a literal six-character blank escape, which cannot be told apart from one the outlet wrote, and the dot and space marks, which cannot tell a marked blank from a real dot or space). A lone surrogate wedged between a label, its separator and its value no longer hides the value: the credential nets treat it exactly like a format character, in the machine-readable wash, in a single pass of the display outlet, and on the end-of-run rows and the result frame's `errors[]` on both lanes, while lone surrogates anywhere else are left byte-for-byte. Since 0.84.1 an address-shaped value (`scheme://…`) that sits right after an invisible single character — a format character, a lone surrogate, a non-whitespace control character, or the outlet's own escape token for one — is no longer let through as an address: the scheme stays and the host and path are replaced, with the query and fragment masked as before; `user:<password>@` followed only by stripped units and then a boundary, a port or another `@` is masked as userinfo. A value after real whitespace or after a colour sequence is still treated as an address, and clean text stays unchanged; forms that the previous release masked are checked not to leak on the same outlets. A `user:<password>@` candidate that is the value of a credential label or scheme word — including a label or scheme word split by stripped characters, and a quoted value that goes on past whitespace — is left to the label pass, so the whole value is masked as in the previous release; both reported shapes and frozen samples from the targeted pools are checked on every outlet. |
|
|
432
|
+
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. Since 0.84.0 the product table itself is also checked against the SDK's runtime list of known origin words, so a word the upstream does not know cannot sit in the table. |
|
|
433
|
+
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. From 0.84.0 the package also reads the registry itself: a reader lists the session's background registry only when the server advertises the listing, classifies its failures as unavailable (a failed read keeps the HTTP status and error code when the error carries them, and never invents either), and refuses a partly readable listing as a whole; a classifier turns the listing into the per-row reading, treating a registry status it cannot place as unknown and reading "not listed" only for keys shaped like registry handles and only when the host states the engine is a single process and names the row's session, because a key from another identity space proves nothing by being absent and, on a multi-replica deployment, a listing answered by another replica does not contain this session's agents at all (without that statement a missing row reads unknown and the row stays); and a predicate returns the agents the registry still counts as live that the host has no row for. A listing at the server's 500-row cap cannot show that an agent is gone either, so a missing row there reads unknown; the cap is checked against the engine's own list clamp and, when a server build is supplied, against the server's route. Each listing carries its session and a per-client sequence number, and a listing superseded by a later delivered read, or read for another session than the one the host names, counts for nothing. Only the exact listing object the reader returned can authorize removing a row or adding one: a spread copy, a structured clone, a filtered or a hand-built listing reads unknown and adds nothing, the listing, its row array and every row are frozen, the cap check uses the row count recorded when the page was read, and rewriting a listing's sequence number fools nothing. The fill predicate holds back agents first seen within the absence settle window — timed on one monotonic local clock from when this client's reader first saw the id, so a skew between the server's clock and this one, or reconnecting to a long-running agent, cannot skip the window — agents without a readable registration time and agents the host saw end since the read went out; the registry key of a row, progress or absence event is whichever of its two ids is shaped like a registry handle; and none of the reader, the classifier or the predicate throws. |
|
|
434
434
|
| `scripts/run-plan-review-dismissal-test.mjs` | An automatic reopen does not put back a plan-review card the user closed (the first-presentation path neither checks nor clears that record, so a replayed park frame still presents its card). A plan-review card the user dismissed (Esc, abort, or any answer that is not approve or reject) is recorded per session and run at the moment of dismissal, synchronously, before anything queued behind the card can be released; a reopen marked `trigger: 'automatic'` then refuses with `{ reopened: false, dismissedByUser: true }` instead of minting a new card the user's next keystroke would land on, while the user's own next action (`trigger: 'user'`) reopens it and clears the record. The record is keyed by gate instance when the host supplies an instance reader: a new plan gate on the same run is still surfaced, and anything that cannot prove the gate is new (no reader, a failed, empty, thrown or timed-out read) refuses on the conservative side. The asynchronous form re-checks after its reads and before minting — a decision handed over meanwhile (seen by the package, or reported by the host's optional hand-over predicate) answers as "your answer is on its way"; a record that changed meanwhile makes the stale evaluation mint nothing and answer from the current record: another close refuses as the user's close and keeps the newer record, a card already back on screen (the user's own action or a concurrent automatic reopen, waited for within the receipt window and re-read once the wait is over) answers `reopened: true`, and a record that is gone (session change, ledger overflow) answers a plain refusal without `dismissedByUser`; a hand-over predicate that throws refuses with a plain `{ reopened: false }`. Each run has at most one reopen on its way: a reopen that arrives while an earlier one's card is published but not yet settled joins it instead of minting a second card and retiring the first card's answer path. Instance readers are snapshotted when they resolve, so a host that hands over its own live set still gets a new gate recognised; and a decisive-looking host answer note does not clear the record for the very card the package's own responder already judged non-decisive (the label was not on that card). A successful reopen replaces only the record taken before the card was minted (with an on-screen marker, not a deletion), a decisive answer clears it, the per-session ledger is bounded, and a session change clears its bucket. Without `trigger` the reopen answers exactly as before, apart from joining a reopen already on its way. |
|
|
435
435
|
| `scripts/run-gate-interrupt-safety-test.mjs` | The two safety properties of running the suites themselves. The collecting runner no longer kills a suite outright when its time limit expires: it sends a termination signal first, waits a grace period for the suite to clean up, and only then kills it — and it records the suite as timed out however it exits, so a suite that exits cleanly after the signal is still not counted as passing. The limit and the grace period can be widened for a single suite in `gates-manifest.json` (an optional entry with a written reason; a malformed entry, including one with a misspelled field, stops the runner before any suite starts instead of silently falling back to the default, and an entry filed under a misspelled name is reported by this guard rather than ignored), and the negative-control suite is widened there. Every property is exercised on byte-for-byte copies of the real runner and the real negative-control suite in a throwaway directory: a suite that honours the signal finishes within the grace period, one that ignores it is killed when the grace period ends, and the summary lines are unchanged; once a suite has exited the runner waits at most a short drain window for its output pipes, so a child process that inherited them and outlives the suite neither turns a passing suite into a timeout nor holds the runner past the limit, the grace period and that window; the negative-control suite, interrupted in the middle of a rehearsal by any of the three signals or by the runner's own time limit, restores the file byte-for-byte, leaves no backup behind, stops the rehearsed guard together with anything it started, regenerates the build output (checked on a small project in the throwaway directory), and exits with 128 plus the signal number; a rebuild that would overrun the grace period is cut short and its compiler stopped; a backup left behind by an earlier run — next to a later target, loose in the source tree, or inside a symlinked dependency directory — makes it refuse to start without touching anything, naming the file and how to restore it. |
|
|
436
|
+
| `scripts/run-system-reminder-open-tag-test.mjs` | The opening-tag locator for the engine's `<system-reminder>` envelope, for hosts that split a message into envelopes and body text themselves rather than only stripping or unwrapping it. `findSystemReminderOpenTag(text, from?)` returns the start and end of the next opening tag at or after `from` (UTF-16 indices, `end` one past the tag) or `null`, and it recognises exactly the two shapes the engine mints — a bare tag, or a tag with a single `mark` attribute whose value is 22 base64url characters — through the very same matcher the strip and unwrap entry points use, so there is no second grammar to drift. The guard pins both engine shapes to the exact index, refuses the non-engine shapes listed in the integration guide plus near relatives (and does not let them swallow a real tag that follows), refuses an opening tag truncated at the end of the text, answers `null` without throwing for a non-string text and for a `from` outside the integer range 0..length, and shows that the locator ignores block context (a tag inside a block body, an unclosed tag and a nested inner tag are all located — pairing with a close is the caller's loop, as in the strip entry point). On twenty thousand seeded random strings mixing both shapes, truncations, near relatives and nesting, the set of tags the locator finds equals the set derived purely from the observable answers of the strip and unwrap entry points, and a strip and an unwrap rebuilt on the locator agree with the real ones byte for byte. Two timing cells show a single call over many near-miss tags and a full walk with `from` moving forward both stay linear |
|
|
437
|
+
| `scripts/run-detach-durable-off-verdict-test.mjs` | The verdict behind the durable-off hint, now a public entry point without the process-level gate. `isDetachDurableOff400(err)` answers whether an error is the engine's refusal of the detach-on-disconnect header on a deployment with no durable run ledger: status 400 and the engine's own refusal sentence in the message, and nothing else — in particular not the error code that refusal carries, because the engine uses the same generic precondition code for unrelated refusals on the same submit path, and treating those as this one would silently resubmit a turn without the header. Browser and desktop hosts, which never arm the terminal's detach ledger, can now ask the same question instead of keeping their own copy. The guard pins the positive case on the real sentence (with or without the code, with surrounding text, and on the error the SDK actually throws when a stubbed engine answers 400), the negative case on the engine's other refusals that share the code (verbatim, and again through the SDK), non-400 statuses, and eighteen malformed inputs that must answer `false` without throwing. It shows the verdict does not read the process ledger and that the hint still returns nothing before the header has been sent. The verdict never throws: an error object whose `status` or `message` getter throws, a proxy whose trap throws and a revoked proxy all answer `false`, each property is read at most once (and `message` not at all unless the status is 400), and the hint therefore answers nothing for those objects instead of throwing as it did before. Against a frozen copy of the previous hint, every other input gives the same answer byte for byte, armed or not. When the engine's source tree is available, it also checks that exactly one refusal site with that code carries the sentence and that every other one is answered `false` |
|
|
438
|
+
| `scripts/run-parked-resume-startup-test.mjs` | The one reading and the one sentence for a parked resume that fails while starting (`422 parked_resume.startup_failed`). The engine has two unrelated reasons for it — the parked agent's session is gone or the engine rejected the decision itself, or the approval is for a tool the agent inherits from the task that started it and this server cannot hand that tool over — and the response carries **no field that tells them apart**; the reason lives only in the engine's prose, which this package does not branch on. So the reading narrows the cause only by what the caller itself sent: a deny can only be the first reason (the inherited-tool refusal is minted for approvals alone, and the guard pins that upstream premise), while an approval, or a caller that does not say, gets one sentence that lays out both ways forward instead of guessing. Either way the verdict is fixed: sending the same decision again is not a recovery path, and deny stays available. The two sentences are minted in one place, never claim the card is still waiting (in one of the two shapes it is not), and avoid every word the package's own "already decided" scan looks for, since they are appended to failure text that scan reads; all five decision exits append exactly that sentence, and any other failure keeps its text byte for byte. The code also takes precedence over the package's "already decided" word scan: the engine's own text for the gone-session shape contains "not found", which that scan used to read as "someone already settled this card" and answer with a silent reconnect; now a failure carrying this code is classified by the code (no reconnect, the sentence reaches the failure text), while failures with no code or another code are scanned exactly as before. The predicate that says which codes classify themselves is public, so a client with its own "already decided" chain can ask it before scanning words. |
|
|
439
|
+
| `scripts/run-plan-review-injected-wire-test.mjs` | The plan-review orchestration accepts a host-supplied engine client, so a browser host behind a same-origin relay — which cannot install a token-bearing engine target — runs the package's own decision path instead of rebuilding it. With a client injected, the decision, its in-flight latch, the resend-once-without-the-key handling of a `request.field_conflict` refusal, the post-decide status re-pull, the six-state effect classification and the outcome text are byte-for-byte what the installed-target path produces, proven against a real relay-form SDK client and a fake engine while the installed target points at a different fake engine that must receive nothing. The engine-version evidence behind the three-choice card is read only under the cache key the host names, never from the installed target; every request of the relay-form client leaves without an `authorization` header, checked beside a token-bearing client on the same spy so the absence is not the spy's blindness; and a planted token never reaches a log line, an outcome sentence, a queue item or the host callback on any of the failure paths. A decision may carry a `reason` (trimmed, dropped when blank, capped at the server's limit), an `onOutcome` callback hands the outcome back to hosts that have no model-prompt queue (called once per admitted decision, never for a latched duplicate, and its own failure never affects delivery), a reopen with an injected client no longer needs a host delivery function on a non-default session, and the two capability probes take the same injected client with a bounded wait. A decision that the injected client's own time limit cuts off is reported as sent without an answer within the time limit (it may have taken effect), in the same sentence the installed-target path uses when its limit is reached, while a connection that cannot be made is still reported as unreachable; a client built with the documented decision budget answers a slow approval exactly as the installed target does. The package's own reads through an injected client — the two probes and the post-decide re-pull — settle at their deadlines even while the client sleeps through a retry back-off, and abort the request underneath; a malformed injected connection is reported as not sent, never thrown. |
|
|
440
|
+
| `scripts/run-session-policy-refused-removal-test.mjs` | The way out for a session that already carries a tool name the engine refuses. The engine refuses to start a run when any rule applied to it names a retired tool, or a name containing "__" without a protocol prefix, so a session whose own rule record holds such a name fails at startup on every run. The guard pins three pieces. First, the refusal test itself: it is compared name by name, in both directions, with an independent reading of the installed engine's own tables (retired names and protocol prefixes) over a corpus that includes padded, wildcard and parenthesised forms, and it is shown to be wider than the verdict used before a write — a padded or wildcard form is withheld there for another reason yet still refuses the run, so a removal driven by that verdict would leave the session broken. Every corpus name is then written alone into the deny list and into the allow list of a real engine run: the test says "refused" exactly when the run fails at startup with the refusal code and the model is never called, while the same shapes in the command lists are not audited at all. Second, the removal verb: it takes only the refused entries out of those two lists and leaves every other entry, list and ordering byte-for-byte, keeping an emptied allow list as a present empty list; after it runs, the same session's next real engine run gets past startup, for every refused name in the corpus and in both lists. Against the installed engine as an ordinary caller, taking a deny entry out is refused as a loosening — reported as such, written exactly once, the record unchanged and the next run still refused, never a false success — while an allow-list-only removal, which narrows, succeeds; against a model of the newer contract the engine has announced, the removal succeeds, and the same model still refuses a removal of anything else. A concurrent writer causes exactly one re-read, judged again on the other writer's record, and a second collision is reported rather than retried; read failures, coded write refusals and unconfirmed writes (no verdict, a receipt that cannot be read, or a receipt that does not account for the removal, including one that still holds a removed entry) each land in their own arm, and running the verb again after an unconfirmed write reports nothing left to remove. Third, the wording: one sentence per outcome; the loosening refusal names operators and says the record is unchanged, the unconfirmed one says the change may already be in effect, and none carries a rule string or engine text. The startup-failure sentence answers to that one code only, lists every place the entry can be set without naming who may remove it, and a real run with the name in the caller's own policy rather than in the session record yields the same code — which is why the sentence may not claim the entry is in the session record. The verb also requires the tool roster reported by a run on the deployment: a name listed there, as a tool name or an alias and compared exactly as the engine compares it, is left in place even when it is a retired name, because the engine accepts that rule — with a real engine run that mounts a deployment tool under a retired name (or with that name as an alias), the removal keeps the deny rule and the next run still starts. Without a roster, or with one that cannot be read, nothing is read or written and the outcome says why; a readable empty roster is still a roster. The removal sentence speaks separately about names taken out of the deny list and out of the allow list. |
|
|
441
|
+
| `scripts/run-fixture-flat-done-ratchet-test.mjs` | Test fixtures that feed the engine's terminal `done` frame are kept in the shape the live engine actually sends. Since the terminal record became a tagged cause (`result.terminal`), the older flat shape (`result.status` plus loose keys) reaches a client only in two ways: a stored row replayed verbatim from before the upgrade, and the server's own refusal envelope — so a fixture written in the flat shape tests the replay path while claiming to test the live one. The suite finds every `{ type: 'done', result }` literal under `scripts/`, follows `result` to the object literal it really comes from (in place, through a variable, through a helper's parameter at each of its call sites, through a spread, or through the rows of an iterated array) and counts the flat ones that carry no one-line note saying they model a replayed row or a refusal envelope. That count may only go down: it is a ceiling kept in the ratchet registry, and a count under the ceiling prints a step-down line instead of passing in silence. A planted corpus with a known number of flat, noted and cause-shaped frames in each of those forms must be counted exactly, and a note that sits inside a string or gives no reason does not count |
|
|
442
|
+
| `scripts/run-authority-envelope-mirror-test.mjs` | The authority-envelope tags that a peer's message body is defused against before it is written into a transcript line. Tags the engine uses to speak with its own authority (reminders, completion notices and the like) must never survive inside a body a peer wrote, or a forged completion notice could be read back on resume as a real one and poison the dedup ledger. The package keeps its own copy of the engine's list, so the copy is reconciled against the installed engine package in both directions: a tag the engine treats as authority and the package lacks is red, and so is a tag the package defuses that the engine does not, since that rewrites ordinary text in a peer's message. The engine does not export this list from its package entry yet, so the check reads it from the engine's own module by path and says so; it also confirms the list is still derived from the engine's envelope registry. At the rendered output, every engine authority tag placed in a body (opening, closing, with attributes, upper case) comes out defused on all three peer lanes, and the engine's non-authority envelope tags come out untouched |
|
|
436
443
|
|
|
437
444
|
Each suite carries a floor that only moves up — a refactor that stops executing a group of
|
|
438
445
|
assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
|
package/dist/abortableSleep.d.ts
CHANGED
|
@@ -1 +1,11 @@
|
|
|
1
1
|
export declare function abortableSleep(ms: number, signal: AbortSignal): Promise<void>;
|
|
2
|
+
export type DeadlineSettled<T> = {
|
|
3
|
+
kind: 'value';
|
|
4
|
+
value: T;
|
|
5
|
+
} | {
|
|
6
|
+
kind: 'error';
|
|
7
|
+
detail: string;
|
|
8
|
+
} | {
|
|
9
|
+
kind: 'deadline';
|
|
10
|
+
};
|
|
11
|
+
export declare function settleWithinDeadline<T>(ms: number, run: (signal: AbortSignal) => Promise<T>): Promise<DeadlineSettled<T>>;
|
package/dist/abortableSleep.js
CHANGED
|
@@ -13,3 +13,40 @@ export function abortableSleep(ms, signal) {
|
|
|
13
13
|
signal.addEventListener('abort', onAbort, { once: true });
|
|
14
14
|
});
|
|
15
15
|
}
|
|
16
|
+
function errorDetail(error) {
|
|
17
|
+
try {
|
|
18
|
+
return String(error);
|
|
19
|
+
}
|
|
20
|
+
catch {
|
|
21
|
+
return '';
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
export function settleWithinDeadline(ms, run) {
|
|
25
|
+
const ac = new AbortController();
|
|
26
|
+
return new Promise((resolve) => {
|
|
27
|
+
let settled = false;
|
|
28
|
+
const timer = setTimeout(() => {
|
|
29
|
+
if (settled)
|
|
30
|
+
return;
|
|
31
|
+
settled = true;
|
|
32
|
+
resolve({ kind: 'deadline' });
|
|
33
|
+
ac.abort();
|
|
34
|
+
}, ms);
|
|
35
|
+
const finish = (out) => {
|
|
36
|
+
if (settled)
|
|
37
|
+
return;
|
|
38
|
+
settled = true;
|
|
39
|
+
clearTimeout(timer);
|
|
40
|
+
resolve(out);
|
|
41
|
+
};
|
|
42
|
+
let pending;
|
|
43
|
+
try {
|
|
44
|
+
pending = Promise.resolve(run(ac.signal));
|
|
45
|
+
}
|
|
46
|
+
catch (error) {
|
|
47
|
+
finish({ kind: 'error', detail: errorDetail(error) });
|
|
48
|
+
return;
|
|
49
|
+
}
|
|
50
|
+
pending.then((value) => finish({ kind: 'value', value }), (error) => finish({ kind: 'error', detail: errorDetail(error) }));
|
|
51
|
+
});
|
|
52
|
+
}
|
package/dist/adapt/wireShapes.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
|
+
import { frozenSet } from '../frozenSet.js';
|
|
1
2
|
export const FLUSH_INTERVAL_MS = 100;
|
|
2
3
|
export const IDLE_FLUSH_MS = 1500;
|
|
3
|
-
export const SUBAGENT_TOOL_NAMES =
|
|
4
|
-
export const WORKFLOW_TOOL_NAMES =
|
|
4
|
+
export const SUBAGENT_TOOL_NAMES = frozenSet(['Task', 'Agent', 'Fork']);
|
|
5
|
+
export const WORKFLOW_TOOL_NAMES = frozenSet(['Workflow', 'RunWorkflow', 'run_workflow']);
|
|
5
6
|
export const TASK_TOOL_NAMES = new Set(['TodoWrite', 'TaskCreate', 'TaskUpdate']);
|
|
6
7
|
export const REJECT_MESSAGE = "The user doesn't want to proceed with this tool use. The tool use was rejected (eg. if it was a file edit, the new_string was NOT written to the file). STOP what you are doing and wait for the user to tell you how to proceed.";
|
|
7
8
|
export const CANCEL_MESSAGE = "The user doesn't want to take this action right now. STOP what you are doing and wait for the user to tell you how to proceed.";
|
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
import { abortableSleep } from '../abortableSleep.js';
|
|
2
2
|
import { corruptStoredRowContent, corruptStoredRowWhere } from '../wireErrorTriage.js';
|
|
3
3
|
export const PLAN_REVIEW_GATE_KIND = 'plan_review';
|
|
4
|
-
export const PLAN_REVIEW_GATE_KINDS = [PLAN_REVIEW_GATE_KIND, 'dry_run_review'];
|
|
5
|
-
export const ASK_PARK_GATE_KINDS = ['human', 'irreversible_ask', 'policy_ask', 'tool_approval'];
|
|
4
|
+
export const PLAN_REVIEW_GATE_KINDS = Object.freeze([PLAN_REVIEW_GATE_KIND, 'dry_run_review']);
|
|
5
|
+
export const ASK_PARK_GATE_KINDS = Object.freeze(['human', 'irreversible_ask', 'policy_ask', 'tool_approval']);
|
|
6
6
|
export function askParkForeignGateKind(row) {
|
|
7
7
|
const kind = typeof row.gateKind === 'string' && row.gateKind.length > 0 ? row.gateKind : null;
|
|
8
8
|
return kind !== null && !ASK_PARK_GATE_KINDS.includes(kind) ? kind : null;
|
|
9
9
|
}
|
|
10
|
-
export const PLAN_REVIEW_STATES = ['needs_review'];
|
|
11
|
-
export const ASK_PARK_STATES = ['suspended'];
|
|
12
|
-
export const RUNNING_STATES = ['running'];
|
|
13
|
-
export const CLAIM_RELEASED_STATES = ['completed', 'failed', 'blocked', 'timeout'];
|
|
14
|
-
export const CLAIM_HELD_STATES = ['running', 'suspended', 'needs_review'];
|
|
10
|
+
export const PLAN_REVIEW_STATES = Object.freeze(['needs_review']);
|
|
11
|
+
export const ASK_PARK_STATES = Object.freeze(['suspended']);
|
|
12
|
+
export const RUNNING_STATES = Object.freeze(['running']);
|
|
13
|
+
export const CLAIM_RELEASED_STATES = Object.freeze(['completed', 'failed', 'blocked', 'timeout']);
|
|
14
|
+
export const CLAIM_HELD_STATES = Object.freeze(['running', 'suspended', 'needs_review']);
|
|
15
15
|
export function selfHealSubmissionDisposition(outcome) {
|
|
16
16
|
switch (outcome.kind) {
|
|
17
17
|
case 'decision-pending':
|