@sema-agent/client-core 0.83.6 → 0.84.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +78 -0
- package/README.md +10 -10
- package/dist/adapt/wireShapes.js +3 -2
- package/dist/adapter/activeRunSelfHeal.js +7 -7
- package/dist/adapter/downstream/eventToSdkMessage.js +6 -2
- package/dist/adapter/downstream/wiringManifestView.d.ts +3 -1
- package/dist/adapter/downstream/wiringManifestView.js +1 -0
- package/dist/agentSession/engineAgentRegistryRead.d.ts +18 -0
- package/dist/agentSession/engineAgentRegistryRead.js +167 -0
- package/dist/agentsWireCaps.js +2 -1
- package/dist/attachmentsWireCaps.js +2 -2
- package/dist/controlRouter.js +1 -1
- package/dist/engineAgentAbsence.d.ts +37 -0
- package/dist/engineAgentAbsence.js +142 -0
- package/dist/engineErrorCodes.js +6 -5
- package/dist/engineNoticeCodes.d.ts +17 -0
- package/dist/engineNoticeCodes.js +45 -0
- package/dist/fleet/fleetProjection.js +3 -2
- package/dist/frozenSet.d.ts +1 -0
- package/dist/frozenSet.js +20 -0
- package/dist/handsSeam.d.ts +3 -0
- package/dist/handsSeam.js +19 -0
- package/dist/index.d.ts +23 -13
- package/dist/index.js +12 -13
- package/dist/model/catalogLoader.js +3 -3
- package/dist/model/tierVocabulary.js +1 -1
- package/dist/request/taskRequest.js +15 -5
- package/dist/rewindWireCaps.d.ts +0 -1
- package/dist/seam.d.ts +2 -1
- package/dist/seatContract.js +2 -2
- package/dist/toolResult.js +2 -6
- package/dist/toolRoster.d.ts +16 -0
- package/dist/toolRoster.js +46 -5
- package/dist/wireErrorTriage.js +1 -11
- package/dist/workflowClient.js +4 -3
- package/dist/workflowMonitor.d.ts +1 -1
- package/docs/INTEGRATION-CLIENTS.md +248 -29
- package/package.json +3 -3
package/CHANGELOG.md
CHANGED
|
@@ -49,6 +49,84 @@
|
|
|
49
49
|
> 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
|
|
50
50
|
> commit 漏转在发布前就红,不再靠人记。
|
|
51
51
|
|
|
52
|
+
## 0.84.0(2026-09-27)
|
|
53
|
+
|
|
54
|
+
> 主题:🔴 **minor** —— peer sdk 地板 `>=11.3.0` → `>=12.0.1`,同版四件 additive 与一批陈旧逻辑清扫。① 后台代理**登记读数**(缺席行的下半场):读口 + 取代判定 + 归类口 + 登记键桥 + 补行谓词,宿主终于有真读数可以填进缺席行回收入参的 `registry` 位;② 读目录授权两枚**结论通告**的事实读口;③ 接线回执 **`hands` 段**(引擎对「这条腿有没有挂上它自带的文件 / shell 工具」的正面声明);④ 审批词表与云控制面子路径**追平 sdk 12**(读法零改);⑤ **清扫**:退役键 `rewind.rewindFiles` 构造期拒收(专句点名去处)、`WorkflowRunState.agentCount` 改可选、26 个内部件退出包根、九张公面判定表换成只读的 Set 子类、包内判定用的十五张数组运行期冻结、`resume_at` 文本兼容腿退役。根公面运行期导出 1278 → 1260(+8 −26);公面类型 +10;超集键 +1 `_sema_hands`;开发依赖引擎 `~7.33.1` 不变。🔴 换钉前先读下面 BREAKING 五条(终端有一处编译期会红:`agentCount` 改可选)。
|
|
55
|
+
|
|
56
|
+
### BREAKING
|
|
57
|
+
|
|
58
|
+
- **peer `@sema-agent/sdk` 地板 `>=11.3.0` → `>=12.0.1`**。本包读三处 12.x 才有的声明 —— 每会话后台登记列表 `sessions.background`、它的能力位 `capabilities.background.listFace`、七词闭集的后台状态型 —— 装着 11.x 会见 peer 警告,声明面也编译不过。sdk 12 唯一的 BREAKING(设备码回体 `verification_uri_complete` 由必填改可选)本包早按可选读。地板由门见证:门从装着的声明文件里逐条读出这三处,不信版本号。**谁要跟**:终端 / 网页端 / 桌面端 / 管理台都换钉 sdk `>=12.0.1`;端上若有把 `verification_uri_complete` 当恒在的串来读的代码,同批改(在场 ⇒ 一键 / 二维码;缺席 ⇒ 展示地址 + 用户码)。
|
|
59
|
+
- **退役的回退键 `rewindFiles` 退场**(型面 + 运行期)。`SeamRewindSpec.rewindFiles` 删除(本包 0.81.0 起已不产它;引擎自 core 6.0.0 起对「捕获」义忽略、对「回退」义拒收 —— 回退旗是 `restoreFiles`)。请求装配的 `rewind` 组现在恰四键(`resumeAt` / `resumeAtMode` / `restoreFiles` / `rewindFilesTo`):给 `buildTaskRequest` / `assembleTaskRequest` 传 `rewind: { rewindFiles: … }` **构造期抛 `TypeError`**,不再原样透传;消息是退役键专句 —— 点名 `rewind.rewindFiles` 是退役键(引擎 core 6.0.0 起)、要删掉、代之以什么(回退会话时一并还原文件 ⇒ `rewind.restoreFiles: true`,须与 `rewind.resumeAt` 同交;只还原文件 ⇒ `rewind.rewindFilesTo`);在场即拒,`true` / `false` 同判;混着别的表外子键时这一句先报。值为 `undefined` / `null` 仍算没给。**谁要跟**:四端都已不发这一位,零改动;仍在交它的宿主换钉前删掉(不删 = 每一轮构造期抛,一轮都发不出去),不要换成别的键:文件历史在首次触碰时即被跟踪,不需要请求旗;钉着 `rewindFiles === true` 的请求键对账测试那一格同批删。
|
|
60
|
+
- **`WorkflowRunState.agentCount` 改可选**(型面;与 0.73.0 起的 `totalTokens` 同律)。有键 ⇒ run 记录上的腿数(`agents` 为空数组时是真 `0`);缺键 ⇒ run 记录上没有可读的腿数组 —— 渲「—」或不渲这一段,**绝不渲 `0`**。`projectWorkflowRun` 遇到 `agents` 在场却不是数组时也不再抛。**谁要跟**:终端 —— 工作流详情 / 列表降级行上按必有数字读的地方(如 `run.agentCount > 0`)编译不过,改按在场判,为「不知道」补的那一枚壳侧超集位可以退役;网页端 —— 详情面「N agent(s)」缺席时会渲成空数字,缺席不渲这一段。桌面端 / 管理台零读点。
|
|
61
|
+
- **26 个内部件退出包根**(型面 + 运行期;0.71.3 预告的 `export *` 放大件收回;已知各端源码零具名 import):`DEFAULT_DENY_REASON`、`MAX_HOOK_NOTICE_TEXT_CHARS`、`STOP_NOT_LANDED`、`STOP_NOT_LOCAL`、`STOP_PARKED`、`STOP_PARK_ARBITER_UNREACHABLE`、`STOP_PARK_RESUME_WON`、`clearBgTerminalFacts`、`denyReasonForWire`、`engineWireDebugEnabled`、`isAskTool`、`listNotifiedRuns`、`listWorkflowCompletionCardsEnqueued`、`noteBgOwnerAbsence`、`notePlanReviewAnsweredFor`、`planReviewArmedKeyFor`、`projectDiagnosticsFrame`、`registerSubagentContentAlias`、`resolveOwnerContext`、`surfaceEditNotForwarded`、`surfaceRuleArmNotSent`、`surfaceRuleArmRejected`、`unregisterLocalQuestionResponder`、`waitForGateArmedFor`、`wireCycleSeq`、`wireParentId`。包内照旧使用,行为不变;包的 `exports` 只开根入口与 `./registry`,从内部路径深 import 不受支持。0.71.3 预告的 34 名里 8 名暂留:`hostTimersFor` / `engineSessionParamFor` 是多会话宿主契约的一部分(与 `hostSettingsFor` / `hostFsFor` / `hostSessionFor` 同族),撤回收回预告、长期留在公面;`unrefTimer` 留在公面(跨宿主可移植的定时器原语:没有 `unref` 的宿主上原样返回、不抛,三端宿主自建保活定时器共用这一份判断;门在根公面上钉它的行为);`shortTaskLabel` / `DENIAL_LIMIT_KINDS` / `engineCapNestedTrue` / `surfaceRememberNotApplied` 定 0.85.0 退出公面;`RULE_OFFERS_ABSENCE_REASONS` 留在公面(与 `@sema-agent/sdk` 根入口导出的是同一个数组对象)。**谁要跟**:四端零改动;管理台测试里把 `shortTaskLabel` 当「一定在公面上的名字」用的那一格,0.85.0 前换一个名字。
|
|
62
|
+
- **九张公面判定表换成只读的 Set 子类**(运行期行为面):`CONFIG_REFUSAL_CODES`、`DELEGATION_CAP_CODES`、`TERMINAL_FLEET_TASK_STATUSES`、`CONTROL_TOOL_VERBS`、`STRUCTURED_DETAIL_TYPES`、`INTERNAL_SDK_ARM_TYPES`、`AGENT_MEMORY_WORDS`、`SUBAGENT_TOOL_NAMES`、`WORKFLOW_TOOL_NAMES`。它们一直标着 `ReadonlySet<string>`,实际却是普通 Set,而本包自己的判定读的就是同一个实例 —— 任何一处 `.add(…)` 都会改掉所有调用方的判定。现在 `add` / `delete` / `clear` 抛 `TypeError`、表不变;`has`、`size`、迭代、`forEach`、`instanceof Set`、`new Set(table)` 复制与 `structuredClone` 都照旧,成员与次序不变,型面不变。🔴 **看得见的差别**:实例的 `constructor` 不再是 `Set`、原型也不是 `Set.prototype` —— 拿它跟普通 `new Set([...])` 做严格深比较(Node `assert.deepStrictEqual`,以及比较构造器 / 原型的测试断言)从此判不等;宽松深比较(`assert.deepEqual`)照旧相等。**谁要跟**:端上若有这类严格深比较,改比成员(`[...table]`)或先复制成普通 Set;要可变的表请复制成新 `Set`。四端源码普查零处 `.add` / `.delete` / `.clear`。
|
|
63
|
+
|
|
64
|
+
### Added
|
|
65
|
+
|
|
66
|
+
- **后台代理的登记读数(缺席行的下半场)**。回收判据本身(接入文档 §102)一个字没改,变的是宿主终于有东西可以填进 `registry` 那一位:
|
|
67
|
+
- 读口 `readEngineAgentRegistry(client, sessionId, caps, opts?)`:只在 `capabilities.background.listFace` 是自有严格 `true` 时读本会话的后台登记;能力位不成立 ⇒ `no_list_face`,不发请求;路由缺席(404 `not_found.route`)⇒ `route_absent`;部署没接会话属主面(501 `capability.session_ownership_required`)⇒ `ownership_required`;其余一切(含 404 `not_found.session`、传输层失败、200 体读不出、**任一行**读不出、能力位 / 体 / 行 / 错误对象上的取值器抛、客户端形坏)⇒ `read_failed`。读不到一律回原因,**绝不**回空表,promise **绝不**拒绝;有一行读不出就整张拒收(少一行在归类口那里就是「没列」)。`read_failed` 在错误对象上读得到时带上 HTTP 状态 `status` 与错误码 `errorCode`(诊断用,不改归类;读不到就不带,不编),并在宿主日志口留一行 debug。不缓存、不轮询。读到的读数带它读的会话 `sessionId` 与序号 `seq`(按客户端 × 会话在发出时单调递增);读数对象、行表与每一行都冻结。🔴 读到的读数要**原样**交给下面几只口:能授权删行或补行的,只有读口产出的那一个对象(见归类口)。
|
|
68
|
+
- 取代判定 `isEngineAgentRegistryListingSuperseded(listing)`:同一客户端 × 同一会话上,发出更晚的一次读已带着行表到货 ⇒ `true`;被取代的读数在下面两只口里不作数。读口没经手的读数(拷贝 / 自建)判不出取代(答 `false`)—— 这类读数在归类口与补行谓词里一律不作数。
|
|
69
|
+
- 归类口 `engineAgentRegistryReadingOf(listing, key, opts?)` 把读数变成 `reapEngineAgentAbsentRows` 已经在收的逐行读数:登记状态 `pending` / `running` / `parked`(停在一张待决审批上)⇒ `running`;`completed` / `failed` / `killed` / `cancelled` ⇒ `ended`;认不出的词 ⇒ `unknown`。行**不在**清单里时,**只有**键是登记句柄形、宿主给了 `{ singleReplica: true }` 与这一行所属会话 `sessionId`(且与读数读的会话对得上)、**且**读的那一刻上游那一页不满服务端的 500 行上限,才读 `not_listed` —— 登记只列答这次请求的那个服务端进程上的行,按登记时刻新 → 旧截取,多副本部署上或满额的清单里「没列」不等于「离场」;不核会话就不许删。否则读 `unknown`,缺席行照留(多一行,绝不错删)。`singleReplica` 只在确知引擎是单进程时给(例:宿主自己拉起、只连这一台的本机引擎);给了 `sessionId` 而读数不是那条会话 ⇒ `unknown`;别的身份空间的键(子代会话 id、工作流子代)一律 `unknown`。🔴 **只认读口原样产出的那一个读数对象**:展开复制、`structuredClone`、JSON 往返、过滤或重组过的读数、手搭的读数一律读 `unknown`(截断判据按读到时的原始行数记在包内,过滤成短表骗不过它;读数上的 `seq` 改写也骗不过取代判定)。
|
|
70
|
+
- 登记键桥 `engineAgentRegistryKeyOf(event)`:缺席 / 行 / 进度事件(或宿主按事件合并的行)在登记域里的键 —— 两位 id 里是登记句柄形的那一位,同一只代理的进度事件与行事件因此取到同一把键。
|
|
71
|
+
- 补行谓词 `engineAgentRegistryRowsMissingFromHost(listing, hostKeys, opts?)`:登记仍算活着、宿主却没有这一行的代理(补行事实 `{ id, status, description?, createdAt }`)。逐行不补:本客户端的读口**首次读到**这一 id 不到 30 秒(与缺席侧同一个稳定窗 —— 进度事件可能先于把它的 id 桥到登记句柄的行事件到达;窗自本地首见起算、用包内同一只单调钟,与服务端的钟无关:两台机器的钟差、重连后首读一只老代理都跳不过这个窗)/ 登记时刻读不出(补行事实要带它)/ 宿主自这次读**发出**以来见到结束或撤掉的(`goneSinceRead`);整张空:读不到 / 读数不是读口原样产出的那一个对象 / 读数被取代 / 会话对不上 / 宿主键集或 `goneSinceRead` 读不了 / `nowMs` 给了却不是有限数。`nowMs` 与首见同一只钟(epoch 锚定的单调毫秒,量纲同 `Date.now()`),缺席时读调用那一刻的这只钟(要可复现就显式传)。首见按客户端 × 会话 × id 记:宿主每次读都新包一层 `sessions` ⇒ 每次都是新账、窗永远起不来 —— 复用同一只客户端。
|
|
72
|
+
- 读口、取代判定、归类口、键桥、补行谓词都不抛。新型 `EngineAgentRegistryClient` / `EngineAgentRegistryListing` / `EngineAgentRegistryRowView` / `EngineAgentRegistryUnavailableWhy` / `EngineAgentRegistryReadingOptions` / `EngineAgentRegistryFillOptions` / `EngineAgentRegistryMissingRow`。接入文档 **§107a B-1–B-8**。
|
|
73
|
+
- **读目录授权的结论通告读口**。`readReadRootGrantNotice(notice)` 读引擎对一次读目录授权的两枚结论(`approval.read_root_granted` / `approval.read_root_grant_rejected`):结局、与卡对上的工具调用 id,以及引擎写下的目录(卡上候选 `dir` 与引擎现在持有的规范拼写 `root`)或拒绝原因。只有 granted 通告说明加了目录,**没有 granted 通告 = 什么都没加**。granted 通告缺 `dir` 或 `root`、或 `covers` 在场却不是 `exact` ⇒ 整只读不了(`undefined`),读不出的范围绝不放宽成「整个目录」。不抛。新型 `ReadRootGrantNoticeFactsView`。本版**不发**授权本身:「放行并加目录」这一选项候服务端宣告能受理的能力位(老服务端对未知请求键静默忽略并照回 200,按版本号开闸会让人以为加了目录、其实只得到一次普通放行)。接入文档 **§107a G-1 / G-2**。
|
|
74
|
+
- **接线回执的 `hands` 段**。`wiring_manifest.hands`(`{ mounted, reason? }`:引擎正面声明这条腿装配时有没有挂上它自带的文件与 shell 工具)投影为超集键 `_sema_hands`,并以 `hands` 出现在 chrome 臂与提交回执视图上(`WiringManifestView` / `WiringManifestChromeEvent` 各 +1 可选键);`mounted` 须是自有布尔,`reason` 按引擎原字节透传,段缺席 = 没报(不折成「有」,也不折成「没有」)。读口 `handsSeamReadingOf(view)` 三态(`not_reported` / `mounted` / `not_mounted`),措辞 `handsSeamDetail(reading)` 每态一句;「没有自带工具」那一句与名册派生的 `handsMountedDetail` 同句首句尾(那一只的字节不变)。服务端 7.103.0 起带这一段,老服务端缺席。`handsSeamDetail` 对认不出的入参(包括误传的名册派生读数)答「没报」那一句,不抛。名册派生读口 `handsMountedFromManifest` 不变 —— 两者是不同的事实:`hands.mounted` 是装配事实,名册派生是这条腿裁剪后的名册里有没有这类工具;`mounted: true` 也不等于 shell 可达(接入文档 §107a-4)。新型 `WiringManifestHandsView` / `HandsSeamReading`。
|
|
75
|
+
|
|
76
|
+
### Changed
|
|
77
|
+
|
|
78
|
+
- **`handsMountedDetail` 对认不出的入参不再抛**。此前 `undefined` / `null` 抛 `TypeError`,其余认不出的值(包括误传的 `hands` 段读数)回 `undefined`,`mounted` 读数没有行数时渲出「undefined of them」;现在一律答与 `handsSeamDetail` 同一句「没报」(同一个铸点)。它认得的六种读数的句子逐字节不变。按「会抛」写的 try/catch 从此不再触发;按「回 `undefined` 就不渲」写的判空分支从此拿到一句串。
|
|
79
|
+
- **审批词表追平 sdk 12**:sdk 12 在审批帧的键锚上声明了 `mandate`,在停泊行上声明了 `requiresRealApproval` / `mandated` / `mandate` / `origin` / `ruleOffersAbsence`,并在运行期导出已知出身词与强制位词;本包这些读法早已就位,读法不变,词表 `ASK_ORIGIN_WORDS` / `APPROVAL_MANDATE_WORDS` 的内容与次序也不变(改为对上游运行期值逐词逐序对账)。
|
|
80
|
+
- **云控制面子路径的声明缺口关闭**:`./registry` 子路径的声明不再引用一个没有安装的包,用 `skipLibCheck: false` 做类型检查的下游不再从它那里见到「找不到模块」。
|
|
81
|
+
- **十五张判定数组运行期冻结**(型面不变):`REWIND_ERROR_CODE_PREFIXES`、`STOP_CONFLICT_CODES`、`CLAIM_RELEASED_STATES`、`CLAIM_HELD_STATES`、`RUNNING_STATES`、`ASK_PARK_GATE_KINDS`、`ASK_PARK_STATES`、`PLAN_REVIEW_GATE_KINDS`、`PLAN_REVIEW_STATES`、`TOOL_PERMISSION_REQUEST_ID_DOMAINS`、`ATTACHMENTS_SPEC_KEYS`、`LIVE_DEFAULT_FIELDS`、`TIER_ORDER`、`DEFAULT_CATALOG_SOURCES`、`CATALOG_DEFAULT_HOSTS`。与上一条同一病形:本包在调用期拿这同一个数组查成员 / 前缀 / 次序 / 白名单(例:往 `REWIND_ERROR_CODE_PREFIXES` 里 `push` 一个前缀,`isRewindFamilyCode` 就对它答真)。现在 `push` / `splice` 抛 `TypeError`,下标写 / 截断在严格模式下抛(非严格模式下静默无效),表都不变;读、迭代、复制与深比较都照旧。四端源码普查零处就地改写。
|
|
82
|
+
- **工作流终态词集**(用来判一条轮询中的工作流是否已报过)现在恰是引擎两处会发的终态词:`completed` / `failed`(工作流运行记录)与 `cancelled`(进程内任务登记的工作流句柄 —— 没有持久存储时停掉一条工作流,轮询回的就是它);引擎从不发的五个词(`done` / `stopped` / `interrupted` / `error` / `canceled`)删掉。
|
|
83
|
+
- **不变**:`CLAIM_RELEASED_STATES` 与无头重连的 run 行终态表照旧认 `timeout`。它不是现役写词,是 core 5.8.0 之前落库的历史行上的词 —— 服务端读 run 行的状态列原样返回,这类行今天仍可能被读回,照旧当已释放、照旧合成终帧(不让重连重试到预算耗尽后报传输失败)。
|
|
84
|
+
- **`cancel_lost_race` 的消息**改为「…check what that decision did to the run before retrying the cancel」(原为「…re-read the run state and retry cancel if it is still active」)。码不变。
|
|
85
|
+
|
|
86
|
+
### Removed
|
|
87
|
+
|
|
88
|
+
- **`isResumeAtRejection` 不再读错误文本**:只按机读码判(`errorCode`,或政策折叠码携带的原码)。给 7.50 及更早服务端的文本兜底腿(那些服务端的预检拒绝只在消息里带 `resume_at.*` 码)删除;这些服务端自 0.60.0 起已在支持窗外,7.51 起每一条这类拒绝都带码。
|
|
89
|
+
|
|
90
|
+
### Gates
|
|
91
|
+
|
|
92
|
+
- **后台登记读数**(`run-engine-agent-absence-projection-test.mjs` 新增 B / EB / BF 段):读口只在能力位严格为真时发请求,四种读不到按码分,半张表整只拒收,七种取值器抛 / 客户端坏形一律 resolve 为 `read_failed`;归类口的终态划分逐词对装着的引擎运行期谓词对账,句柄形镜像对引擎真值逐字对账,500 行上限对引擎运行期夹限(给了服务端发布物时再对它的路由)双向对账;单副本位只认自有严格 `true`;取代判定(发出序 / 到货序 / 更新那次读失败不取代)、会话核对(`not_listed` 不给会话即 `unknown`)、`read_failed` 诊断位(有才带、值形不对不带)与一行 debug、`goneSinceRead`、宿主键集七种读不了的形各有格;读数出身 —— 被取代读数的展开复制 / `structuredClone` / JSON 往返 / Proxy 不补行、归类 `unknown`,没被取代的读数的拷贝同样认不出(原对象照读作正控),被取代空表的拷贝不复活 `not_listed`,500 行截断清单过滤成新对象仍 `unknown`,读数 / 行表 / 每行冻结,原读数上改写 `seq` 骗不过取代判定;稳定窗(把包内单调钟钉在固定读数上:首见窗内 / 恰在窗上 / 窗外一毫秒、登记时刻早已过窗也按首见算、本机钟快 60 秒的新代理与重连首读的老代理都不补、同一客户端沿用首见、中途不在最新读数里的 id 再出现时重新起算、坏钟、登记时刻缺席);行事件晚于稳定窗才到时照补一行是取舍格(KL-164)。归类 / 补行格一律经读口取读数(手搭形只作「认不出」的反例)。
|
|
93
|
+
- **`hands` 段**:`run-wiring-manifest-projection-test.mjs` N 段(投影、坏形逐形、原型链上的段不算、chrome 臂与提交回执视图同值);`run-tool-roster-projection-test.mjs` H12 段(名册派生措辞口坏入参不抛、认得的六句逐字节不变、与新读口共用铸点)。
|
|
94
|
+
- **读目录授权结论通告**:`run-engine-notice-catalog-test.mjs` H 段(只认两码、`toolCallId` 必在、拒绝原因原样、granted 缺 `dir` / `root` 或 `covers` 坏形整只读不了、`covers` 词与引擎型编译期双向钉)。
|
|
95
|
+
- **地板见证**:`run-sdk-floor-test.mjs` ②e 段 —— 地板 12.0.1,见证本包新读的三处声明(列表方法、`background.listFace`、七词状态型);门看得见的地板以上每个已装 sdk 也须带这三处,地板降到它们不存在的线即红。负控锚同批抬(篡改形 12.0.2)。
|
|
96
|
+
- **sdk 12 追平的对账**:审批帧键锚对账回到逐元素相等(锚 32 项,`mandate` 的领先登记按退出条件删);停泊行五键从「按结构读、无账」转为有账(读法零改);能力位台账对 sdk 12 新增的四键逐一显式处置(`background` / `fileHistoryCapture` 读,`taskApproverPosture` / `taskWriteFaceOpen` 不读);出身词表改对 sdk 运行期 `ASK_ORIGINS` 逐词逐序对账,另钉 sdk 的型确由它派生;`run-ask-survives-posture-test.mjs` 另核产品表里每个词都是上游已知词。
|
|
97
|
+
- **子路径入口**:`run-sdk-registry-transit-test.mjs` —— 上游子路径已自带声明,「找不到模块」实测归零,这一段留作回潜守卫(再出现缺失的声明包当场红);隔离证明改用只 import 诱饵的合成探针;另核上游子路径声明不再引用那个缺失的包。
|
|
98
|
+
- **请求装配**:退役的 `rewind.rewindFiles` 子键被拒收并点名;拒收消息是退役专句(点名退役版本、要删掉、代之以 `rewind.restoreFiles` 配 `rewind.resumeAt` 或 `rewind.rewindFilesTo`;`true` / `false` 同拒;混着别的表外子键时专句先报);值为 `undefined` / `null` 不拒;`rewind` 那一行恰四键;发布的 `SeamRewindSpec` 声明里没有 `rewindFiles`(`run-task-request-omission-receipt-test.mjs` F16b – F16e)。
|
|
99
|
+
- **释放与重连**:历史行上的 `timeout` 照旧算已释放、照旧合成终帧(`run-selfheal-reopen-test.mjs` G8④a / G8④a′,`run-client-core-pure-test.mjs` B5 重连格);终态表出身门把它登记为历史落库行读兼容的遗留扩员(C / F 段);工作流终态词集对引擎**两处**上游(工作流运行记录的状态型 ∪ 进程内任务登记工作流句柄的状态字面量,后者从引擎源码按语法读出)去掉 `running` 双向钉死,另加一格真任务登记的取消回归:停一条没有持久存储的工作流、读回终态卡,结构化卡与模型面两条投影路径都须记为已通知、从等待计数里摘掉(`run-terminal-table-provenance-test.mjs` E / E7 段)。
|
|
100
|
+
- **判定表**:九张 Set 逐一钉成员与次序,`add` / `delete` / `clear` 抛且表不变,实例冻结;十五张判定数组逐一钉冻结、`push` / `splice` / 下标写 / 截断抛且表不变,改写企图之后包内判定照旧(`run-client-core-pure-test.mjs` ㉚′ / ㉚″)。**`unrefTimer`**:在根公面上钉跨宿主行为(没有 `unref` 的句柄原样返回不抛;有 `unref` 的调恰一次)(`run-client-core-pure-test.mjs` TIMER④)。
|
|
101
|
+
- **`resume_at` 拒绝**:包装的纯文本形答 `false`;机读码形(含带码的 sdk 类型化错误)答 `true`。**工作流视图**:没有腿数组时 `agentCount` 缺席,空数组时为 `0`。**`cancel_lost_race`**:消息说先去看那次决断对 run 做了什么再重试取消(`run-wire-refusal-copy-test.mjs` S3c)。
|
|
102
|
+
- **公面收回**:`run-export-liveness-test.mjs` G 段 `removed` 账新增 0.84.0 一版 26 名;此前经包根读这些值的门改从属主模块取。
|
|
103
|
+
- 驱动层门的终帧夹具换成活流上真会出现的形(CC-224;只改门,出包面零变化)。`run-hitl-gate-honesty-test.mjs`(95 帧)、`run-shell-gate-durable-allow-test.mjs`(17 帧)、`run-park-hop-progress-test.mjs`(2 帧,park 那一帧经共用的 `syncLeg` 喂进 14 格)、`run-wire-refusal-copy-test.mjs`(2 帧,其中一帧经共用的 `askLeg` 喂进 4 格),以及 `run-client-core-pure-test.mjs` 里审批桥的问答闭环 / 无浮层两格与 `runStream` 的四帧 park(B3 park 终态、复核卡片主格、卡片出口缺席 / 抛错两格)—— `done` 帧从退役的平面形(`status` + `checkpointGate` / `errorCode`)换成带标因由 `result.terminal`(park = `{ kind: 'paused', gate }`,completed / failed 同理),读 `.result.status` 的断言改读 `terminal.kind`;`run-hitl-gate-honesty-test.mjs` F3 经常量喂给 `isPlanReviewPark` / `armPlanReviewApproval` 的复核 park 同批换形。门对象里缺 `kind` 的问答门按引擎真实铸形补 `kind: 'human'`。只换形不改判据:五套换形前后检查数逐一相同且全绿(556 / 34 / 59 / 109 / 4011)。换形之前,把审批桥的 park 读数、`runStream` 的复核卡片臂、终帧投影的 park 臂与 `isPlanReviewPark` 四处同时改成只认平面形,这五套照样全绿;换形之后同一组改动每套都红。每套另做一次夹具负控:把换进来的 `kind: 'paused'` 改成表外词,五套都红(`run-client-core-pure-test.mjs` 的 B3 park、复核卡片主格、审批桥三组各单独做一次)。
|
|
104
|
+
- 平面形不再是这几套的默认输入,但它在两条路上仍是真实字节,各留一格同果对照(三格,检查数 556 → 561 / 34 → 39 / 4011 → 4015):
|
|
105
|
+
- `run-hitl-gate-honesty-test.mjs` F1-a′ 与 `run-shell-gate-durable-allow-test.mjs` ①′:跨 7.64.0 升级之前落盘的 park 行,经 durable 事件流原样回放进审批桥(续跑那一段由现役引擎跑,终帧是因由形)。断言与因由形那一格逐条同果:毒化帧不上屏、续跑后的真结果上屏、决断真发出、终帧照出。
|
|
106
|
+
- `run-client-core-pure-test.mjs` 切边②′:同步提交当场停在复核门时,服务端自己组的 200 体 `{ taskId, sessionId, status: 'needs_review' }` —— 既没有 `terminal` 也没有门,`needs_review` 这个词是它唯一的复核信号。复核卡片照弹、先于终帧。选这一形而不选「带平面门的历史行」,是因为后者靠门种就能认出,证不到只读状态词的那一半。
|
|
107
|
+
- 三格各带一条自证(喂进去的确实是平面形),自证按**结构**判形(没有 `terminal` 键、状态词对得上、门在 / 不在),不拿产品读口当尺子;产品读口把它读成平面暂停另成一格 —— 读口坏了是产品红,不是判据坏。把审批桥或复核卡片臂改成只认因由形,对应那一格当场红。
|
|
108
|
+
- 不在换形之列(本来就是服务端的现役形):409 拒绝信封、`/decide` 200 体、durable `suspended` 事件、run 行的 `status` 列。
|
|
109
|
+
- 登记物:根公面基线 1278 → 1260(+8 −26);`export-liveness` 登记 73 → 46 行(contract 40 / internal 2 / retire 4),棘轮 `maxRows` 同批 73 → 46(两条 contract 行因判定数组冻结格按名引用而按退出条件删除);超集键台账 88 → 89(`_sema_hands`);单例清单 524 → 526(登记读口的两本读数账与句柄形镜像入册,`resume_at` 文本兼容表随腿删);可移植闭包 kernel 16 → 18 / adapt 34 → 35 / index 210 → 213;门数 144 不变。
|
|
110
|
+
|
|
111
|
+
### Known limits(本版新增)
|
|
112
|
+
|
|
113
|
+
- 登记只列答这次请求的那个服务端进程上的行:没有上游「这份清单覆盖本会话全部副本」的信号之前,「行不在 ⇒ 可回收」只靠宿主的单副本位 —— 不给位时到期缺席行不回收(多留一行),给错位时多副本部署上可能删掉一只还在跑的子代(行上留回收句,不静默)(KL-159)。
|
|
114
|
+
- 补行谓词按宿主给的登记域键判「有没有这一行」;宿主只记面板键、没用 `engineAgentRegistryKeyOf` 取登记键时会多补一行(KL-160)。
|
|
115
|
+
- 登记状态词的终态划分没有上游运行期值可取,本包按 sdk 七词型面编译期穷举(上游加词当天编译红;运行期先到的新词读 `unknown`:不删、不补)(KL-161)。
|
|
116
|
+
- 「放行并加目录」本版不发,候服务端宣告能受理的能力位;结论通告读口已在(KL-162)。
|
|
117
|
+
- `hands` 段在服务端 7.103.0 发布前的真部署上恒缺席;`{ mounted: true, reason }` 这种矛盾形原样透传、读口按 `mounted` 判(KL-163)。
|
|
118
|
+
- 补行稳定窗 30 秒,自本客户端首次读到这一 id 起算:行事件晚于窗才到时宿主会多一行,直到行事件到来、宿主按登记键并行;宿主每次读都新包一层 `sessions` 时每次都是新账,首见恒是「刚才」,补行永远不发生(KL-164)。
|
|
119
|
+
- 读数「被取代」按同一客户端 × 会话上的发出序近似判(服务端对并发请求的处理序本包看不见);读数在路上时结束的行,只有宿主给了 `goneSinceRead` 才不会被补回;读数只认读口产出的那一个对象 —— 宿主想过滤就得在归类 / 补行的结果上过滤,不能先过滤读数(KL-165)。
|
|
120
|
+
- 登记列表最多 500 行:行数到上限时本会话所有「行不在」读 `unknown`,这类会话里缺席行只等真终态(KL-166)。
|
|
121
|
+
- 审批出身词 / 强制位词两张词表仍是本包的一份(对 sdk 12 运行期值逐词逐序对账,今天两边相同);sdk 在两次提货之间加词时,本包到下次提货才跟(KL-167)。
|
|
122
|
+
- 九张判定表挡不住刻意绕开实例方法的写法(`Set.prototype.add.call(table, x)`)(KL-168);九张 Set 与十五张判定数组之外,公面上其余数组形导出(展示 / 拼装用,或每行是对象)运行期没冻结(KL-169)。
|
|
123
|
+
- `mcpReconnect` 的注入面(`sessions.mcpReconnect(sessionId, { server }, opts?)`)与 sdk 12 声明的 `sessions.mcpReconnect(sessionId, server, opts?)` 第二参形不同:宿主不能把 sdk 客户端直接传进来,要自己包一层(KL-171)。
|
|
124
|
+
- 对 7.50 及更早的服务端,预检阶段不带码的 `resume_at` 拒绝判 `false`,去锚自动重发不触发(这些服务端早在支持窗外)(KL-172)。
|
|
125
|
+
- 读目录授权的 granted 通告形坏(缺 `dir` / `root`、`covers` 是表外值)时读口答 `undefined`,宿主按「没有 granted = 没加」会说没加、而引擎其实加了;今天不可达(引擎恒铸这两位,`covers` 只有 `exact` 或缺席)(KL-173)。
|
|
126
|
+
- 补偿登记里一条「待上游确认后退役」的说明串仍是未兑现形:本包早已按上游的终态进度事件落终态,防御性清扫事实上已是兜底;串要等上游确认那一条承诺已满足再改(KL-174)。
|
|
127
|
+
- 上一版登记的「联网搜索原因句只对到源码判官」一条本版销(对服务端 7.102.0 发布产物逐字重跑零差异;KL-65);「登记读口到货前缺席行永不回收」一条改写为「只在登记能证明离场时回收」(KL-118)。
|
|
128
|
+
- 完整台账见接入文档 §107 末行「包侧缺口」。
|
|
129
|
+
|
|
52
130
|
## 0.83.6(2026-09-27)
|
|
53
131
|
|
|
54
132
|
> 主题:patch —— 会话规则记录的无损判定(`sessionPolicyDeliverable`,0.83.1 起)补上一个判定缺口:两类名字此前被判「能写」,写进会话规则记录后这个会话此后的每一跑都会在启动时失败,而且会话属主自己撤不掉。本版把它们扣下,成因闭集加一个词 `legacy_tool_name`。🔴 对成因词做穷尽分支的端要加一臂(类型联合 +1 员)。
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.
|
|
38
|
+
**Version:** 0.84.0
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -67,7 +67,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
67
67
|
against — the tables live upstream precisely so this package does not keep a second copy that can
|
|
68
68
|
fall behind. The browser bundle really bundles the SDK through (the portability guard would
|
|
69
69
|
exit 3 rather than quietly mark it external).
|
|
70
|
-
- The declared floor is `>=11.3.0` (raised from `>=11.2.1` in 0.80.0: a deny decision may now name its settler — `settledBy: "policy"` is typed on both the durable decide body and the live respond body from 11.3.0 on, and the gate record's settlement vocabulary carries its thirteenth word `policy_refused`, which this package reads to place a refusal in the policy bucket rather than the person's; the package now compiles against 11.3.0 and no consumer ships 11.2.x any more, so the older floor lost its witness; before that raised from `>=11.0.1` in 0.79.0: `capabilities.mcpProbe`, the MCP probe face (`mcpCapabilities` / `probeMcp` with `McpProbeFace`) and the write receipt's third liveness arm (`stillLive: "unknown"`) are typed from 11.2.x on, 11.2.1 adds `TaskRequest.approverPosture` and the `mandated` approval-frame key to the types and the runtime key anchor, the package now compiles against 11.2.1, and no consumer ships 11.0.x / 11.1.x any more, so the older floor lost its witness; before that raised from `>=9.8.1` in 0.78.0: `DeniedBy` carries its tenth word `read_boundary`, `rules.write` answers a `stillLive`-discriminated body, `RemovalLiveness` / `RuleWriteRequest` / `RuleWriteResult` / `RuleWriteBehavior` are exported from the SDK root and `TaskRequest.excludeAllTools` is typed from 11.x on, the package now compiles against 11.0.1, and no consumer ships 9.8.x any more, so the older floor lost its witness; before that raised from `>=9.7.1` in 0.75.0: `Capabilities.deviceExecutor.management` is typed from 9.8.x on, the package now compiles against 9.8.1, and no consumer ships 9.7.x any more, so the older floor lost its witness; before that raised from `>=9.6.0` in 0.74.0: `Capabilities.approvalsStreamLive` / `.executionLane`, `LivePendingRow.frame`, the `live_*` approval-stream events and `gates[].toolCallId` are typed from 9.7.x on, and no consumer ships 9.6.0 any more, so the older floor lost its witness; before that raised from `>=9.4.0` in 0.71.0: the `tool_disclosure` / `tool_progress` frames and `ToolApprovalFrame.readRootCandidate` are typed there; earlier: raised from `>=8.8.0` in 0.69.0: the `reasoning_end` frame and `McpStatusPanel.lastLegMcp` are typed from 9.4.0 on), and it is *witnessed*: the guard checks that an actually
|
|
70
|
+
- The declared floor is `>=12.0.1` (raised from `>=11.3.0` in 0.84.0: the per-session background listing (`sessions.background`), its capability bit `capabilities.background.listFace` and the closed background-status vocabulary are typed from 12.x on, and the package reads all three, so it no longer compiles against 11.x; before that raised from `>=11.2.1` in 0.80.0: a deny decision may now name its settler — `settledBy: "policy"` is typed on both the durable decide body and the live respond body from 11.3.0 on, and the gate record's settlement vocabulary carries its thirteenth word `policy_refused`, which this package reads to place a refusal in the policy bucket rather than the person's; the package now compiles against 11.3.0 and no consumer ships 11.2.x any more, so the older floor lost its witness; before that raised from `>=11.0.1` in 0.79.0: `capabilities.mcpProbe`, the MCP probe face (`mcpCapabilities` / `probeMcp` with `McpProbeFace`) and the write receipt's third liveness arm (`stillLive: "unknown"`) are typed from 11.2.x on, 11.2.1 adds `TaskRequest.approverPosture` and the `mandated` approval-frame key to the types and the runtime key anchor, the package now compiles against 11.2.1, and no consumer ships 11.0.x / 11.1.x any more, so the older floor lost its witness; before that raised from `>=9.8.1` in 0.78.0: `DeniedBy` carries its tenth word `read_boundary`, `rules.write` answers a `stillLive`-discriminated body, `RemovalLiveness` / `RuleWriteRequest` / `RuleWriteResult` / `RuleWriteBehavior` are exported from the SDK root and `TaskRequest.excludeAllTools` is typed from 11.x on, the package now compiles against 11.0.1, and no consumer ships 9.8.x any more, so the older floor lost its witness; before that raised from `>=9.7.1` in 0.75.0: `Capabilities.deviceExecutor.management` is typed from 9.8.x on, the package now compiles against 9.8.1, and no consumer ships 9.7.x any more, so the older floor lost its witness; before that raised from `>=9.6.0` in 0.74.0: `Capabilities.approvalsStreamLive` / `.executionLane`, `LivePendingRow.frame`, the `live_*` approval-stream events and `gates[].toolCallId` are typed from 9.7.x on, and no consumer ships 9.6.0 any more, so the older floor lost its witness; before that raised from `>=9.4.0` in 0.71.0: the `tool_disclosure` / `tool_progress` frames and `ToolApprovalFrame.readRootCandidate` are typed there; earlier: raised from `>=8.8.0` in 0.69.0: the `reasoning_end` frame and `McpStatusPanel.lastLegMcp` are typed from 9.4.0 on), and it is *witnessed*: the guard checks that an actually
|
|
71
71
|
installed SDK at that line still exports every value-level symbol this package imports and still
|
|
72
72
|
declares `TaskStats.costMicroUsd` (the key `costOrNull` reads). A floor nobody ever ran is a
|
|
73
73
|
promise, not a contract.
|
|
@@ -297,14 +297,14 @@ guard still cross-checks the table by name).
|
|
|
297
297
|
| `scripts/run-segment-authority-single-source-test.mjs` | The authoritative-segment replacement verdict, single-sourced. `text_end.content` and the `text_delta` stream stopped being byte-identical the day the engine started redacting the former through the same filter as the result, so every consumer now has to decide six ways what to do with the segment it has half-emitted — and until this release that decision existed **twice**: once here for the transcript lane, once in the shell for the print lane, hot-fixed a version apart. The verdict is now one pure function both lanes call, and the guard pins it on the quantity that actually decides the outcome: whether the authoritative text still *starts with* the bytes that already left, not whether a flush has happened — the latter is a precondition, and anchoring on it withholds a perfectly ordinary answer. Each of the six forms is checked with its counter-case, the prefix length is pinned to UTF-16 code units against a non-ASCII sample whose UTF-8 byte count differs (slicing by bytes leaves the very thing being redacted on screen), and the withheld-segment ledger is compared by normalised equality rather than substring, because a short redaction marker quoted in an unrelated later answer would otherwise suppress that answer entirely. The same file pins the session-level memory-capture declaration to one mint point — the wire value is a single-member closed set, and a consumer that spells it wrong gets a loud refusal rather than a silently dropped privacy request — and pins the SDK URL/health transit to be the **same function reference**, since wrapping it would discard the one guarantee the transit exists for. A last section strips comments with the TypeScript parser and asserts the second expression has not grown back |
|
|
298
298
|
| `scripts/run-print-bash-iserror-test.mjs` | The print lane's Bash `is_error` authority (structured over regex). A second section pins where the denial classification word lands on this lane: on the message envelope, never inside the tool-result block, because that block is forwarded verbatim to the provider on compaction and a self-minted key there is the shape of an old, real defect. A word outside the upstream table — or an empty string, a non-string, or nothing at all — mints no key rather than a guess, and the word never moves the error flag, because attribution does not decide anything |
|
|
299
299
|
| `scripts/run-bash-benign-exit-interpretation-test.mjs` | Benign non-zero Bash exits (`returnCodeInterpretation`) stay non-errors across all three derivation arms, and the annotation transits to the card |
|
|
300
|
-
| `scripts/run-sdk-floor-test.mjs` | The SDK version floor — and, more to the point, that the *installed* type declarations still carry the keys this package reads — including, from 0.80.0, the three declarations that justify the floor itself: the key naming who settled a refusal on both decision legs, and the thirteenth word in the settlement vocabulary. They are found through the syntax tree rather than by searching text, because this guard's own comment stripper blanks string contents and would have made that check permanently, silently green. |
|
|
300
|
+
| `scripts/run-sdk-floor-test.mjs` | The SDK version floor — and, more to the point, that the *installed* type declarations still carry the keys this package reads — including, from 0.80.0, the three declarations that justify the floor itself: the key naming who settled a refusal on both decision legs, and the thirteenth word in the settlement vocabulary. They are found through the syntax tree rather than by searching text, because this guard's own comment stripper blanks string contents and would have made that check permanently, silently green. From 0.84.0 the floor is 12.0.1 and the guard witnesses the declarations the package now reads: the per-session background listing method, the `background.listFace` capability bit, and the closed background-status vocabulary that the registry classification switches over exhaustively. Every installed SDK the guard can see at or above the floor must carry those declarations too, so lowering the floor to a line where they do not exist fails. |
|
|
301
301
|
| `scripts/run-engine-caps-ledger-test.mjs` | A per-key disposition ledger for `GET /v1/capabilities`. The SDK's `Capabilities` grew from 74 keys to 93 in one release and nothing on the board could see it: this package consumes that table through four synchronous readers, and *nineteen new positions arriving while the package does not move* is exactly the disease shape this repo keeps logging on other axes — the fact is already on the wire, the package boundary is the cell that swallows it, and no client can read it however they write their side. So the ledger is reconciled **element-wise against the SDK interface in both directions**: a key the SDK added with no ledger row is red (someone must classify it), and a row for a key the SDK removed is red too (a registration that no longer does anything). Each row then has to survive its own claim — a `read` row names the source file, and the **code** there (comments stripped) must really mention the key, because prose asserting an alignment is the classic way these guards go hollow; a `not_read` row must have **zero** read sites in the tree, so wiring one up while the ledger still says the package ignores it is red rather than invisible. The census behind those two directions recognises five call shapes, each of which really occurs here — a reader whose base argument carries its own parentheses, a direct `caps.<key>`, a narrowing cast, an own-property read helper, and a `*_CAP` constant — and proves it on fabricated samples first, since a census that recognises one shape reports "nothing here" for the other four. What the guard deliberately does **not** judge is whether a position *ought* to be read: that is a design call, and the ledger only pins that every capability was looked at once by a person and that what they wrote down does not contradict the code |
|
|
302
302
|
| `scripts/run-sql-engine-capability-test.mjs` | The SQL-posture read face and the four-state capability reader underneath it. One capability cell here carries **four different things**, and each one points an operator somewhere else: nothing has been observed yet in this process (a one-shot doctor run is always in that state), the response arrived but carries no such key (an older engine), the engine explicitly answered `null` — *this deployment has no SQL backend*, which is a **positive fact** rather than an absence — and a full reading. Fold any two together and the screen states something flatly, confidently, and wrongly, so every positive control here is paired with a control pointing the opposite way, and the four sentences the doctor row can print are checked to be pairwise distinct and non-implying. The reading itself is narrowed no tighter than the mint: `txnMode: null` is a **legal value** — two of the three engines always report it that way, and the upstream type note names reading it as "optimistic" as the error — so treating it as malformed would throw away the entire reading for ordinary deployments, which is the same disease this repo logged when a consumer's domain was narrower than the producer's. A response that cannot be parsed **clears** the cell rather than leaving the previous engine's answer in place, and a separate invalidation port exists for the case the generation latch cannot catch — a same-port respawn whose new probe never succeeded, where the stale reading would otherwise be answered as current fact. Untrusted values (the isolation string is read back from a database server variable) are sanitised and bounded before display, and the bound is applied **before** escaping so a visible escape never gets cut in half. Finally the export names are themselves a guard: the shell still carries a copy that is meant to go red on the package's same-named export and be swapped out, so renaming anything here would silently disarm that lock |
|
|
303
303
|
| `scripts/run-web-search-backend-capability-test.mjs` | The deployment-default WebSearch backend read face (`capabilities.webSearch.backend`, engine ≥7.82.1). Same four-state discipline as the SQL and write-protection cells, with two things that are specific here and therefore guarded: a **missing key** (an older engine) and an explicit **`"none"`** (the engine says this deployment has no default search backend) point an operator in opposite directions — "cannot tell" versus "not configured" — and must never be folded; and the `none` sentence has to say both halves of the contract at once: the default scenario mounts no WebSearch tool, **and** a caller-supplied `webSearch` setting can still mount it on a single-user lane, because the capability advertises the deployment default, not whether this request has search. The backend word is read as an **open set** — the engine's closed set is typed from its own provider tuple and grows with it, so hand-copying three words here would turn a newly configured backend into "unreadable" (the narrower-than-the-mint disease this repo already logged once). `webSearch: null` is malformed rather than `none` (the mint never emits `null`), extra members never cross, an unparseable response clears the cell, a stale probe generation is dropped, the invalidation port clears to "not observed", and the open-set word is sanitised and bounded before display |
|
|
304
304
|
| `scripts/run-terminal-cause-projection-test.mjs` | The `7.64.0` wire reshape, projected. A run's ending stopped being eight parallel flat keys and became **one tagged cause** (`completed \| failed \| blocked \| paused`), and a tool call's gate stopped being four orthogonal words and became **one record** (`disposition` / `settlement?` / `origin?`). Both are read in exactly one place in this package, and this guard pins them at **two levels**, because the dangerous seam is "the reader was updated, the consumer was not": each terminal arm is checked on the reader *and* on the `subtype` / `is_error` / `errors[]` the projector actually emits. Two properties carry most of the weight. First, a terminal word this reader does not know is **never** laundered into an empty success — it lands on an `unknown` arm carrying the word verbatim, while a payload with no terminal word at all (the mock lane) keeps the success arm exactly as before, which is the one and only case the reader answers `null`. Second, the three window words (`approval_window_expired`, `denial_limit_window_expired`, `park_sla_expired`) must each be told apart by a different predicate: the previous generation collapsed all three onto one `timeout`, and re-merging them would throw away the discrimination this reshape just restored. Two byte generations are read by one reader, keyed on the discriminator upstream nailed (`"terminal" in result`): the current cause form, and the **flat** form that a current engine still emits on two lanes — replayed persisted bytes, which the service passes through verbatim rather than back-filling, and the service's own rejection envelope. A cause-form payload that also carries stale flat keys must ignore them entirely: keeping one compatibility read is what gives a single fact two sources. The same file also pins the MCP delivery verdict and HTTP status riding the wiring manifest, the four-state write-protection reading (where three of the four states mean *cannot tell*, and none of them may be printed as "there is no table"), and the park-reopen fetch identity: that predicate is asserted through the **real entry point**, since the defect being fixed was precisely a call site wired to a different predicate than the one that routed the row there. From 0.80.0 one of those three boundaries flips: the key naming **who settled a refusal** stopped being a dead byte and became part of the wire, so the check stopped scanning the build output for the word and started reading the request bodies the two decision legs actually send. A refusal attributed to the deployment's own policy carries the word; one attributed to a person, one with no attribution at all, and one carrying a word the vocabulary does not hold carry nothing — the wire has no slot for “a person decided this” other than the key's absence, so inventing one would be minting a word upstream does not have. The allow family never carries it on any of its routes, because that combination is refused before the approval is judged while the side effects of allowing have already landed, and the three refusals nobody was asked about (a card that failed, a user who walked away, an interruption) carry nothing either. A deployment that signs the bodies it accepts does not sign that word, and there is no capability bit to ask beforehand, so a refusal on exactly that ground is answered by re-sending the same decision once with that one key removed — byte-for-byte the same otherwise — rather than letting an optional note take the whole denial down with it. The guard measures that along three axes: the decision still lands and is reported as decided with the attribution handed back and a separate flag saying it never reached the wire; a caller who aborted in between gets no second request; every other refusal code, and every decision that never carried the key, send exactly once. The classification of a second failure is made from what the second body actually carried, not from what the card asked for. |
|
|
305
305
|
| `scripts/run-auto-mode-unavailable-test.mjs` | The fact behind "you are being asked because the auto-mode classifier could not run", and the one place its sentence is minted. The cause table is a **copy**, reconciled word for word in both directions against the installed engine's own bytes — it narrowed upstream, and the guard follows rather than keeping the old shape: a table checked against something nobody ships any more is the oldest way for a guard to be green and wrong. The retirement is held from both sides — the removed table must really be gone upstream, and the removed reader and word must really be gone here — while the word that left keeps arriving cleanly from an older engine, because the reader takes the cause as an **open set**: the vocabulary belongs upstream, so a copied list here would discard a legal value the day one is added, and the value discarded is precisely "this outage is a NEW kind". The reader's one exclusion is the word the engine says it never stamps here — the classifier did run and did answer, just outside its contract, so reading it as a failure would invent an event the engine denies. That exclusion used to be derived from a second table which no longer exists; the reason for it never lived in that table, so it is now stated where it actually comes from, pinned as a **named** set (a magic literal scattered through the reader reds) and cross-checked against the engine's own verdict declaration and against the reader having exactly one such comparison. One reader serves both the live ask and its durable parked twin, since the two carry the same key path and a second copy is how two ledgers drift apart. Absence is pinned as absence — most asks never consulted a classifier at all — and the sentences are checked mutually distinct, prototype-safe, and walked end to end: an unknown word reaches the sentence a person reads (the fallback that names it verbatim) and the status reading (unavailable for this round, never a fallback to "available"), with counter-controls proving neither assertion is vacuous |
|
|
306
|
-
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails |
|
|
307
|
-
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source |
|
|
306
|
+
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails From 0.84.0 it also covers the reader for the two read-directory grant notices: it recognises only those two codes, needs the tool call id to match a card, passes the rejection reason through as written, and treats only the granted notice as evidence that a directory was added; a granted notice without both the directory and the spelling the engine now holds, or with a scope other than `exact`, is not read at all, and the scope word is pinned to the engine's type at compile time. |
|
|
307
|
+
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source. From 0.84.0 the roster-derived sentence source no longer throws on a value it does not recognise, including a reading of the manifest's `hands` section passed by mistake: it answers the same "not stated" sentence as the `hands` reader, from one shared source, and its six known sentences do not change. |
|
|
308
308
|
| `scripts/run-permission-rule-issue-codes-test.mjs` | The rule-lint refusal codes an engine reports when it will not compile a permission rule. The SDK publishes neither a schema nor a type for them, so the package mints the table from the engine's own bytes and the guard pays the cost of that copy instead of leaving it to somebody remembering: it parses the codes the engine actually mints and reconciles them against the table in both directions, so a code added upstream (the user would see a bare code) and a code only the package believes in (a branch that can never fire) both fail. It also reconciles the table plus a small retired ledger against the engine's declared union, which is deliberately not the same set — one member was renamed and its old name is still declared — so reviving a code the engine will never mint again is impossible and a future stale member shows up immediately. Sentences are pinned one per code, mutually distinct, and split by family: a rule that is wrong and a rule that is legal but unsupported on this lane are different next steps and may not share a sentence. The engine's own message rides along as prose — sanitized and capped after escaping, never matched on |
|
|
309
309
|
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, ten words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because an unknown word means different things in each: a denial layer this build does not know may have been added by a newer engine or may come from a damaged record, so its sentence says it cannot tell which instead of asserting damage; the asker vocabulary is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine). Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer; it must not say the question is asked every time — an answer for this one call may come from the person, a hook or an automatic check the deployment runs — and its wording is checked against the engine package's own description of the mandate. A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. Since 0.83.2 a sixth list covers the word a mandated question stands on (`APPROVAL_MANDATE_WORDS`, six words): it must equal the engine's own list word for word and in order, membership is exact, the card reader `readApprovalMandate` answers only for an own key holding one of the six words, and each word has one fixed sentence explaining why the question must be confirmed — six distinct sentences that never point the reader at writing a rule, never promise a question every time, never claim only a person may answer, and repeat no other sentence the package mints. The list is also pinned against the engine's type at compile time in both directions, while the published build references no engine package at all: every `.js` and `.d.ts` file in the build is scanned, and the same scan is first shown to fire on references planted in a scratch directory. |
|
|
310
310
|
| `scripts/run-engine-identity-test.mjs` | The engine generation anchors on `/health` (`pid`, `instanceId`, `startedAt`; engine >=7.67.0). `/health` is the one unauthenticated door and its heartbeat is always green, so "another host restarted the shared engine" used to be discoverable only by having some authenticated request hit a 401 first — a path that misreads a restart as a network fault. The reader narrows each anchor independently (one malformed field never hides the other two) and always hands back a reading object rather than an absence, because the caller is asking which anchors answered, not whether there was a response. The comparison is a three-word verdict, not a boolean: `unknown` when the two readings share no comparable anchor at all — an empty intersection means nothing could be compared, never that nothing changed — and the boolean convenience is pinned so that only `true` is an assertion. Any comparable anchor differing decides `changed`, so a reading whose `startedAt` matches while its `instanceId` does not cannot be waved through as the same life; precedence only decides which anchor gets named in the diagnosis |
|
|
@@ -376,7 +376,7 @@ guard still cross-checks the table by name).
|
|
|
376
376
|
| `scripts/run-additive-key-passthrough-test.mjs` | The one disease shape behind two legs: a **closed whitelist / flattening arm** dropping a fact that is already on the wire, while both sides of the seam look correct. (1) The `task_progress` projection carries a registered **key ledger** — a frame populated with every key the service really projects is pushed through the shipped `eventToSdkMessage`, and the set of wire keys that survive must equal the registered pass-through list **name for name in both directions**, so quietly forwarding one more key is as red as quietly dropping one. `model` (the child run's model id, minted by core as `prepared.model.id` and projected by the server since 7.52.1) is the key this batch adds, with the same conditional the server itself applies: a non-empty string or no key at all — an empty string is neither a model id nor "unknown". The ledger is also checked against the fenced list in `docs/INTEGRATION-CLIENTS.md` §3d, so a doc that still says seven keys while the code forwards eight is red rather than merely stale. (2) The decide-failure arms carry the server's S-02 `currentPending` pointer key from a 409 `approval_stale` refusal onto the outcome the host reads. The reader is structural rather than `instanceof`, because the client is host-injected and the class identity is not this package's to assume; a half triple never mints (half a pointer cannot relocate anything), an empty string is not presence, and `checkpointToken` never transits. Both the allow and the deny leg are driven end to end through the real durable approval path — as is the accept-session leg, where a refusal carrying the pointer key must now re-raise instead of silently re-sending the human's answer for the **old** card as a plain approve (one decide call, pointer preserved), while a legacy 400 still falls back exactly as before — and all three flattening points must call the one shared reader — the same-shape residue check that makes "fixed one arm and left the twin" red instead of invisible. (3) The same disease growing on the REQUEST side: the `.mcp.json` → server-spec projection rebuilds each server key by key, and the settings schema deliberately leaves some keys parse-transparent — whatever JSON the file carries reaches the engine untouched, because validating them where the whole domain parses all-or-nothing would let one bad declaration take every server down silently. The whitelist had no row for the newest of them, so an operator's per-tool declarations — the ones the write fence reads — were stripped at the package boundary while both sides looked correct. The criterion is not "is that key handled" but the transparent-key table read out of the INSTALLED schema at runtime, reconciled name-for-name against this leg's ledger, so the day upstream adds a third one this turns red and forces an explicit decision. Behaviour is pinned on both transports, by object identity rather than deep equality (a rebuild would be a second judge), and malformed values must transit UNCHANGED rather than be refused here — the engine refuses them loudly and names the server, whereas a package-side judge can only swallow a declared protection quietly. Absence still mints no key, unknown keys still never reach the wire (the fix is the dropped key, not the gate), and the one transparent key this leg deliberately does not forward is a ledger entry with its own exit condition: it belongs to the deployment plane, and the day the request-plane type declares it the entry's premise is gone and the gate says so |
|
|
377
377
|
| `scripts/run-esc-halt-plan-test.mjs` | The Esc stop decision every client shares: fire the **turn-level** halt first, and escalate to a **run-level** cancel in exactly two cases — the engine itself answered with a 409 from the closed code set (it is saying "there is no in-flight turn here; use cancel for a run-level stop"), or that shot came back with no verdict at all *and* the shell can independently prove a permission card was on screen. Everything else does not escalate. The asymmetry is the whole point and every negative control guards the same direction — deciding *not* to escalate costs the user one more choice on a busy-session card (recoverable), deciding to escalate wrongly tears down a run that was alive and takes every in-flight tool with it (not). So: the closed code set is a **frozen** value, not a `ReadonlySet` — type-level immutability does not stop a consumer's `.add()`, and the guard proves it by really trying to mutate the exported value and then checking the verdict did not drift; the escalation gate is the **conjunction** of that closed set and the 409 status, since honouring the code alone lets a 500 that merely quotes it drive a destructive call; `interrupt.not_held` and `steering.not_running` are deliberately outside the set (the first means *this replica* has no live face — the run may be perfectly alive on another); an unreadable code falls to the no-escalation side; a `parked` flag never overrides a verdict the engine did give, and only strict `true` counts when it did not. The first shot is unconditional by construction — it does not consult `parked`, because the 409 it earns is exactly the verdict the gate wants — and the verdict itself is a closed machine-readable reason word, not display copy. A third escalating case was added once tearing the stream stopped reaping the run: with detach armed, a shot that never lands leaves the run going all the way to the end of the turn, so the Esc the user pressed has no effect at all and nothing on screen says so — the old behaviour had a silent backstop (tearing the stream ended the run) and that backstop is gone. The new fact is held to the same three disciplines as `parked`: it is read only where the engine gave no verdict, it is judged **after** `parked` so an existing host's reason word does not change under it, and only strict `true` counts. Absence is proven to be a no-op rather than asserted — the guard carries its own reference implementation of the previous version's table, runs the full grid through both, requires zero divergence when the new field is omitted, and first shows the comparison really does report a difference on the one cell where the two versions are meant to differ |
|
|
378
378
|
| `scripts/run-peer-frame-projection-test.mjs` | The three engine-injected lanes design/385 puts on the **one** `task_notification` carrier, which are not the same kind of thing at all: a delegated child's uplink (`agentMessage`), another session's message drained from this session's own box (`crossSessionMessage`), and a receipt about one of *this* session's own outbound messages (`crossSessionNotice`). The engine renders none of them inside a `<task-notification>` shell, so a client that projects them as the generic completion card shows "background task finished" while the model read a colleague's sentence — two faces describing different events. The discriminator is pinned to the **typed carrier being present**, never to the `summary` text: those carriers can only be minted by the engine's injection legs (the external `notify()` input is a strict subset of the payload and can wear none of them), while `summary` is filled by every notification there is — so anchoring on text would let any background task impersonate a colleague's message by writing `<agent-message from="…">` into its own summary, and a positive control asserts exactly that payload still projects as the generic card. Fail-closed has two tiers rather than one: a broken **required** field (empty `from`, a non-string `body`, a notice `kind` outside the closed set) returns absence so the caller falls back to the generic card — an honest downgrade where the user still sees the notification — while a broken **optional** field drops only itself, because losing an attribution note and losing a colleague's whole message are not the same magnitude. The provenance side record is **required and must agree on four points** (`kind` matches the lane; `from`/`taskId`/`seq` are present and equal the carrier/payload — each equality is anchored on a core mint site and pinned by the cli wire-anchor A-K24), so a carrier signed with a trusted name but a disagreeing provenance falls back to the generic card; peer bodies pass the same authority-envelope neutralization core applies (`<task-notification>` etc. are defused) so a colleague's text can never seed the resume dedup ledger. Lane precedence copies the engine renderer's own order, because the model already read the frame in that order and a client ordering of its own would put a card on screen that disagrees with the frame the model saw. Rendering and parsing of the transcript line live in the same module and are round-tripped in both directions, including a body carrying a forged closing tag (a parser fooled there hands half a message to the next row) and a quote inside the sender label (which must not forge a second attribute); the notice lane is deliberately kept **out** of the parser, since recognising it would mean anchoring the `[Cross-session …]` prefix and a user typing that same line would be rendered as engine speech. Hostile carriers are read as own **data** descriptors only and accessors are never invoked at all — `catch` catches throwing, not never returning — proven by a counting getter that must stay at zero calls, alongside a revoked proxy and a prototype-only carrier; and four legacy payload shapes assert the no-carrier path is byte-identical to before, which is the executable form of "zero difference for an older host". A re-supplied cross-session message — same task id, status and sequence as the first delivery, handed to the model again after compaction — is not rendered a second time, because the first delivery is still on the user's screen; a record with the next sequence number still renders |
|
|
379
|
-
| `scripts/run-wiring-manifest-projection-test.mjs` | The two end-user facts carried on the engine's `wiring_manifest` frame (`modelGate`: which tools this run's model gate removed and the verbatim restore hint; `autoMode`: whether auto mode is actually armed and the engine's own reason word). Projection: both sections ride as `_sema_`-prefixed superset keys, verbatim, and no SDK-named key is minted; a frame where neither section is well-formed projects to `none/not_in_slice` (no empty arm); `modelGate` needs all three keys and treats `removed: []` as a bad value rather than a reading; `autoMode` needs a boolean plus a non-empty reason that agrees with it, and the reason word is never mapped onto the capabilities vocabulary; the frame is flat (a nested `manifest:{}` wrapper is not a supply); `eventId` rides like every other arm. Adapter: exactly one chrome event on the main lane, a sub-flow frame (any `parentToolCallId`, `null` included) yields nothing, and an absent `eventId` leaves the key absent. Added at receiving time because the shell-side gate could not see this package's behaviour: two mutations (empty `removed` accepted, sub-flow gate removed) had passed the package suite untouched 0.71.0 adds sections F–I: the fourth/fifth/sixth manifest sections (`tools` via the roster reader, `hooks[]` rows dropped one by one when malformed, `lsp` absent unless `mounted` is a boolean), the `tool_roster_delta` arm (narrowed `delta`, `malformed` when `fromDigest`/`roster` cannot be read, host applies it against its own digest), the `context_usage` arm (finite-gated scalars plus `sections[]` rows dropped one by one), and the `WiringManifestMcpEntryView` rename with `MAX_AGENT_SKILLS` gone from the surface |
|
|
379
|
+
| `scripts/run-wiring-manifest-projection-test.mjs` | The two end-user facts carried on the engine's `wiring_manifest` frame (`modelGate`: which tools this run's model gate removed and the verbatim restore hint; `autoMode`: whether auto mode is actually armed and the engine's own reason word). Projection: both sections ride as `_sema_`-prefixed superset keys, verbatim, and no SDK-named key is minted; a frame where neither section is well-formed projects to `none/not_in_slice` (no empty arm); `modelGate` needs all three keys and treats `removed: []` as a bad value rather than a reading; `autoMode` needs a boolean plus a non-empty reason that agrees with it, and the reason word is never mapped onto the capabilities vocabulary; the frame is flat (a nested `manifest:{}` wrapper is not a supply); `eventId` rides like every other arm. Adapter: exactly one chrome event on the main lane, a sub-flow frame (any `parentToolCallId`, `null` included) yields nothing, and an absent `eventId` leaves the key absent. Added at receiving time because the shell-side gate could not see this package's behaviour: two mutations (empty `removed` accepted, sub-flow gate removed) had passed the package suite untouched 0.71.0 adds sections F–I: the fourth/fifth/sixth manifest sections (`tools` via the roster reader, `hooks[]` rows dropped one by one when malformed, `lsp` absent unless `mounted` is a boolean), the `tool_roster_delta` arm (narrowed `delta`, `malformed` when `fromDigest`/`roster` cannot be read, host applies it against its own digest), the `context_usage` arm (finite-gated scalars plus `sections[]` rows dropped one by one), and the `WiringManifestMcpEntryView` rename with `MAX_AGENT_SKILLS` gone from the surface From 0.84.0 the `hands` section (whether this leg was assembled with the engine's built-in file and shell tools) is projected as well: a strict boolean `mounted` plus the engine's reason word passed through as written, absence kept as "not reported" rather than folded to either answer; a three-state reader and one sentence source share their opening and closing with the roster-derived sentence, whose bytes do not change. The sentence source answers the not-reported sentence for anything it does not recognise, including a roster-derived reading, and never throws. |
|
|
380
380
|
| `scripts/run-submit-wiring-manifest-test.mjs` | The non-streaming submit receipt can carry the run's opening wiring manifest (`TaskResult.wiringManifest`, additive on newer servers). `readSubmitWiringManifest` answers one of three: the key is absent on the receipt itself (older server, or a deployment whose engine never produced that frame) — not the same as unreadable; the key is present but cannot be read (not an object, or none of the nine sections survive); or a manifest view. The view is the same shape the streaming lane's chrome event carries (minus its two envelope keys) and is assembled by the same code path, so both lanes agree byte for byte on the same object. Liveness fields ride through untouched — this reader never mints a liveness verdict — and an operator-shaped receipt with extra governance sections reads to the same view as a tenant-shaped one. A zero-tool roster is a real reading, not an absence. |
|
|
381
381
|
| `scripts/run-rule-offers-reader-test.mjs` | The narrowing reader behind the "don't ask again" options, now a public entry point rather than a card-port-only one. Hosts that render the frame themselves (a browser has no three-way terminal card) previously had to rebuild this reader on their side, and what it carries is a **redemption-safety** judgement, not a convenience: the batch arm is redeemed by **index**, so a reader that compacts the array after dropping a malformed entry makes the k-th option a person clicked and the k-th rule the server writes two different rules. So: a bad entry is dropped **on its own** (one bad option must not make a real one disappear) while every surviving entry keeps its **original wire index** — pinned from both ends, with the bad entries leading and trailing. A batch's *members* are the opposite: any malformed member drops the whole batch, because a conjunctive batch is one "yes" to all of them and a batch missing a member is a different grant; its honest-remainder count is a reading, not decoration, so a non-integer or negative value drops the batch rather than rendering a fabricated zero. An empty array, a non-array, an over-cap array and an all-bad array all read as **absence** rather than an empty list, because an empty list renders as "there is an option lane with nothing in it". The two wire generations are ordered by a rule, not a preference: the newer key wins outright, a newer key that is **present but unreadable** does not fall back to the retired key (borrowing the older material would pass someone else's options off as this request's), and a `null` newer key reads as absence so a relaying layer that serialises "missing" as null cannot delete the whole lane on older engines. The public entry is finally reconciled against **both** card-port legs on the same material, byte for byte, so the exported reader and the one the card sees can never become two. Two upstream vocabularies used to be **hand-copied** here, and both had fallen behind: a match word outside the copied pair dropped an otherwise valid option outright, and a batch carrying a directory-read member — a member kind the copy did not know — dropped the whole batch. Both tables now come from one place upstream and are re-exported verbatim, pinned in both directions: every word in the table must be accepted (a narrower copy reds on the words it never learned) and a word constructed to be outside it must still be refused (a reader widened to "any string" reds too), with the retired-key normalising leg sharing the same narrowing so the fix cannot land on one leg only. A member whose kind is genuinely unknown still drops **the whole batch and only that batch** — never one member, because a conjunctive batch one member short renders "yes to N" as "yes to N−1", and never the card, because the honest single beside it is intact — while a member from before the discriminant existed normalises to the historical kind rather than being refused. The additive per-segment reasons ride through verbatim, drop only the row that is malformed, and stay **absent rather than empty** when nothing survives, since an empty list would read as "confirmed nothing uncovered" while the count remains the only source of truth |
|
|
382
382
|
| `scripts/run-resume-refusal-copy-test.mjs` | The **words** a client says when a resume is refused, minted once here instead of three times. The facts behind them already lived in this package; the sentences did not, so each client wrote its own — and those sentences answer a safety question (was my decision consumed, can this token still be redeemed), which is exactly the kind of answer that must not vary by client. Two closed sets meet here and the guard pins their relationship in both directions, because it is a premise rather than a coincidence: one set answers *can waiting help* (the codes the server mints a wait on), the other answers *what should a person be told*, they **intersect in exactly one code**, and each keeps a member the other must not have — a placement mismatch is never waitable no matter what arrives on the response, since its remedy is a changed argument rather than elapsed time, and a full governance window needs no prose because "you can wait" is the whole message. The overlapping code delegates its wait and its disposition to the existing reading rather than judging again: nine shapes of input drive both entry points and the two readings must agree byte for byte, the absent case included, because two judges always diverge somewhere. The wait is narrowed to the domain the server mints it in, which is **stricter than the shell's own copy was** — a zero now reads as no window rather than as "retry now", and the wake-up it would retry is an at-most-once action with real side effects. The third sentence is chosen by the disposition, never by the engine's prose: rewriting the message to either upstream branch's exact wording, with the window untouched, must leave all three sentences unchanged, while adding a window must change the third one and only the third one. The engine's folded resume refusal (`resume_blocked_by_policy`) gets its own reading — the original code as named, absent or unreadable — and a wording that neither claims nothing was consumed nor predicts whether a retry would pass. |
|
|
@@ -386,7 +386,7 @@ guard still cross-checks the table by name).
|
|
|
386
386
|
| `scripts/run-approval-frame-chrome-arms-test.mjs` | The two in-stream approval frames finally reaching every host through the shared pipeline instead of one shell's private branch — the shape of a layering defect: hosts that only consume the package could not rebuild their pending cards after a reconnect, and did not clear a card the engine had withdrawn. The payload is deliberately carried as the **envelope** the upstream types declare rather than the first-version card: the stream parser applies no predicate, so narrowing here would let a legitimately newer frame pass as the older shape and invite consumers to read keys a newer card never promised. The guard therefore pins that every open key survives untouched, that an unknown version still passes through, and that narrowing is left to the host's own predicates — with the fallback being a generic card and a person, **never** an automatic denial. A frame whose version cannot be read at all is reported as malformed rather than dropped in silence, because both frames carry user-visible decisions and state changes. Both arms are registered as **required** host duties, and their duty text names the load-bearing rules a host would otherwise have to rediscover: which predicate to narrow with, that the reconnect preamble — not a replayed historical frame — is the authority on which cards exist, and that a withdrawal frame can be lost entirely. Unlike the sibling arms, these carry **no** sub-stream cutoff: an approval raised under a delegated call still has to reach a person, and filtering it by ownership is the host's job, not a reason to discard it. Finally the upstream bytes that justify the envelope discipline are checked to still be there, since the whole design rests on them |
|
|
387
387
|
| `scripts/run-terminal-status-vocabulary-test.mjs` | One place that decides whether a run has **ended** and whether it ended badly — written because that judgement had already been hand-copied three times, so the day the engine added a word for *the agent itself reported it cannot continue*, every copy missed it and a panel settled a self-reported failure as a success. The distinction the table exists for is pinned from both sides: that word belongs in it, while the two words meaning *waiting for a person to decide* deliberately do **not** — reading those as endings would bury a run that is actively waiting on the reader. A word this client does not know answers *no*, and the guard states plainly that *no* is not evidence of success: proving success means reading the positive side, so negating this predicate is the very mistake that caused two earlier incidents. The fleet lane gets the same treatment from the other direction: a workflow parked on a durable approval used to fall through to *running*, leaving the person with no hint that a card was waiting, and it now lands on the same rendered word the task lane already used — same fact, same word, checked end to end on a real row. Why the word was added directly rather than carried as a private superset key is checked mechanically against the upstream declaration being open, so the day it closes this reds and the decision gets revisited. The residue sweep is the point: the source tree must contain **no** further inlined copy of the judgement, each of the three former sites is checked to really read the single predicate, and the one reviewed exemption carries its reason **and** a liveness assertion, so an exemption whose justification expires cannot quietly keep standing |
|
|
388
388
|
| `scripts/run-terminal-word-source-test.mjs` | Two tables of ending words, kept apart by **who owns them** — because they used to be one. The engine's own closed set of reasons a run ended, and the server's set of row states a run can finish in, overlap in three words but not in all of them: one word for *something outside stopped it* exists only on the server side, and one for *it paused and can be resumed* exists only on the engine side and means very nearly the opposite of an ending. Merged into a single list, those two sources became indistinguishable, so a new word on either side looked the same as a new word on the other, and the safest-looking move — folding the unknown word into a known one — is the exact mistake that has caused incidents here before. The engine-owned table is checked as a **copy, not an opinion**: it is reconciled word-for-word and in order against the installed engine package, read from both its declaration and its runtime bytes with the two required to agree, so the day upstream adds a fifth reason this reds before anything ships. The two dividing words are each pinned from both sides, including against the upstream declaration directly rather than only against this package's own list. Why the table is copied rather than re-exported is itself an assertion with an expiry: the day upstream publishes the set as a value, this guard reds and the decision gets revisited. The renamed tables leave **no alias** behind, since an alias would let a reader keep consuming the merged list and the split would have bought nothing |
|
|
389
|
-
| `scripts/run-terminal-table-provenance-test.mjs` | Several tables answering *has this ended*, which until now only asserted their own current wording rather than that the wording was right — a snapshot equality passes forever even the day upstream adds a word this package never learns about. Each is reconciled against a named upstream source instead, one comparator shared across all of them rather than one copy per table: a notification's terminal words are the engine's own closed set minus its one live word; a sub-agent tick's terminal words are the engine's own inline status literal minus *running*; a run row's terminal words must **cover every** engine reason a run can end — missing one is the exact failure mode that once let a client retry a connection until its budget ran out while the ending sat unread in the row the whole time — plus one explicitly named legacy word the engine's current declaration no longer carries; a fleet row's terminal words are pinned to **exact equality** with the transport's own status set minus its known non-terminal words (an adversarial pass found the earlier one-directional form let a real terminal word be quietly deleted from this side and still pass), and separately pinned against that set's current member count so the day it changes a person has to look. A sixth table has
|
|
389
|
+
| `scripts/run-terminal-table-provenance-test.mjs` | Several tables answering *has this ended*, which until now only asserted their own current wording rather than that the wording was right — a snapshot equality passes forever even the day upstream adds a word this package never learns about. Each is reconciled against a named upstream source instead, one comparator shared across all of them rather than one copy per table: a notification's terminal words are the engine's own closed set minus its one live word; a sub-agent tick's terminal words are the engine's own inline status literal minus *running*; a run row's terminal words must **cover every** engine reason a run can end — missing one is the exact failure mode that once let a client retry a connection until its budget ran out while the ending sat unread in the row the whole time — plus one explicitly named legacy word the engine's current declaration no longer carries — kept on purpose to read rows stored before that word was retired, since retiring a word the engine writes does not remove rows already stored with it; a fleet row's terminal words are pinned to **exact equality** with the transport's own status set minus its known non-terminal words (an adversarial pass found the earlier one-directional form let a real terminal word be quietly deleted from this side and still pass), and separately pinned against that set's current member count so the day it changes a person has to look. A sixth table, a workflow's terminal words, has two upstream producers and is pinned to **exact equality** with their union minus the one live word: the stored run record's status set, and the statuses the in-process task registry gives a workflow handle (read from the engine's own source; it emits *cancelled* when a workflow without a store is stopped — a version of this table reconciled against the first producer alone dropped that word, so a stopped workflow was never marked delivered and its completion could be delivered twice). The table is separately checked against every non-terminal word gathered — including one meaning *durably paused*, sourced from the engine's own outcome vocabulary rather than any of the other unions, after the same adversarial pass found a caller-documented non-terminal word this boundary had missed — and a behavioral regression drives the installed engine's real task registry through *stop* then *read the output*, feeds that result through both projection paths, and requires each to mark the run delivered and clear it from the waiting count. A seventh pair, found by the same pass sweeping the whole tree for the same shape of hand-copied table, answers a related but distinct question — whether a session's claim on a run has been released or is still held — and is pinned as two complementary halves of one upstream set: released-minus-one-named-legacy-word and held must partition the transport's status set exactly, so a real state going missing from either side is caught the same way a fabricated one would be. Every extraction in this guard parses real syntax rather than pattern-matching quoted text, so a comment mentioning a word never counts as that word being present, and single- and double-quoted members are read identically |
|
|
390
390
|
| `scripts/run-workflow-park-truth-projection-test.mjs` | The read face for *which approvals a workflow run left parked* — and the credential that must never ride along with it. Upstream strips the redemption token from that response, and this package's reader is built so the token **cannot** come back: each row is assembled field by field from the three identity keys, never copied wholesale, so an extra key appearing upstream is structurally unable to reach anything this package hands a UI. The guard proves that rather than asserting it — a poisoned row carrying a secret is read, and the secret is searched for across the **entire** serialized result, with the same search proven to find it in the input so a blind search cannot pass; renaming the credential key does not help it through, because the rule is *only these three*, not a blocklist; and the reader's own source is checked to contain no object spread, since one such line would quietly void all of it. The other half is an absence distinction with opposite consequences: a record with **no** parks field at all was written by an older engine and proves nothing about whether approvals are waiting, while an empty list is a positive statement that none are — collapsing those two would let a run whose parked approvals cannot be proven be resumed anyway, so they are kept literally distinguishable, and a payload whose rows are all unreadable answers *unknown* rather than *none*. The four refusal codes for this family are checked code by code against the engine's real bytes, never matched by name prefix, and the older umbrella code they were split out of is asserted to still be **alive** — treating the whole code as retired would make a family of real refusals vanish silently **0.68.3 (core 7.18.0):** two more keys ride the same projection duty as `parks` itself: `originUnconfirmed: true` on a row (never `false`; absence is the confirmed state) and `resumeAdmissionIncomplete: true` on the run (presence means "not a resume base"). Dropping either would turn a refused record back into an admissible one, so the guard pins both, including that neither folds into the other **0.69.1 (CC-12):** both keys now also ride the projected `WorkflowRunState`, so a host that only sees the projection can render them |
|
|
391
391
|
| `scripts/run-retired-vocabulary-census-test.mjs` | Whether a retirement really happened. When upstream removes a family, a downstream package can cut it out or keep a courteous alias — and the alias is the worse outcome: three clients keep writing branches for something nobody emits, and a status line advertises a state it can never reach. Choosing the clean cut only means something if a guard holds it, since a comment saying *retired* is not an exit code. Each registered entry is held two ways: the name must be gone from **code positions** in this package (comments stripped first, because the explanation is supposed to stay) and off the published surface, and — the half that keeps this from being self-congratulation — it must really be gone **upstream**, since that is the entire reason it was removed here; if it comes back, the disposition deserves reconsideration rather than silence. The scanner proves it can speak by finding a symbol that is genuinely present before any absence is believed, and distinguishes a mention inside a comment from one in a string literal, which is exactly the form being cleared. A closing check runs the other way: the retirement **story** must remain in the comments, including a promise this package made earlier and has now had to withdraw — deleting the history alongside the code is a bad way to satisfy *zero hits*, and leaves the next reader with code that has no reason |
|
|
392
392
|
| `scripts/run-classifier-status-test.mjs` | What state the auto-mode classifier is in **on this session** — the question a doctor line, a model settings page and a permission card’s status row all ask, and a different question from the one the approval card asks (*why am I being asked right now*), so the sentences are pinned mutually distinct from that face’s as well as from each other. The session-level half of this reading — a breaker record the engine used to keep — was **retired upstream**, and the guard now holds that retirement from **both** sides: the engine's own declarations must really no longer carry it (a fact coming back would mean the removal here was the wrong disposition, and that deserves a conversation rather than silence), and this package must carry no alias, no state word and no leftover narrowing for it — a reading kept alive for something nobody emits any more is a promise the interface cannot keep, and it left the doctor line advertising a state it can never reach. What remains is ordered by the quantity that actually decides whether the classifier is running: the fact from **this round** first, then whether this leg is armed — a decider is minted per run, so a later leg can be armed again. Not armed, and a section that never arrived, both answer **undefined** rather than *available*; that arming question has its own field and answering it twice grows a second ledger. Arming and availability are also **two words, not one**: the engine says a decider was minted *for this leg*, which is an assembly-time fact, while whether that decider answers any given round is a **per-call** one — so an armed leg reads `armed` and only a positive per-call fact (an ask whose origin is the classifier's own denial-bound fallback, which by construction stands *after* the classifier ran) reads `available`. Every other ask origin is refused as evidence and for a stated reason rather than out of caution: several are ones the classifier is structurally forbidden to answer, and for the rest a surviving ask is precisely the case where it did **not** resolve one — so reading availability off them would be a guess. The projection is a **whitelist**, so an older engine still sending the retired member loses it at the boundary while the two live facts beside it ride through untouched. Rendering never throws and never impersonates: a state word this client does not know — including the retired one, which a restored view can still carry — reaches an honest fallback that names it verbatim, carries no invented explanation of a mechanism that no longer exists, and is proven distinct from all three real sentences; prototype keys reach that same fallback rather than a function body, checked against a real out-of-table word so the comparison cannot hold vacuously |
|
|
@@ -412,7 +412,7 @@ guard still cross-checks the table by name).
|
|
|
412
412
|
| `scripts/run-registrar-tables-test.mjs` | The four registrar table bodies — this Guards table and the three census tables in the repository's negative-control record — are **generated** from `scripts/gates-manifest.json`, the one file that describes a suite. Every row's text must equal what the generator emits, the manifest's suite set must equal the suites on disk, each entry must declare how it is negative-controlled (rehearsed, blind, or behavioural, with the census taker itself declared as such since it does not appear in its own tables), and each row's outward prose is scanned against the published-surface word list — the same list the packaging-hygiene guard uses, shared rather than copied — before the generator may write it into this file. Adding a guard is therefore one manifest entry plus one generator run instead of six hand edits across three files, and a description that drifts in one place and not the others stops being expressible. The row count is no longer what is compared: the earlier arrangement checked the census tables by **length**, so rows naming the wrong suites reconciled green. Positive controls run entirely on in-memory copies — a changed description, a dropped entry, an added entry, a changed class and a hand-edited row on disk each have to make the same judgement speak — and the quieter halves are pinned too: nothing outside a table body may move, a line inside one that is not a recognisable row makes the generator refuse rather than drop it, byte equality is backed by a column-count check (a cell holding a bare pipe splits a row into extra columns, and a code span does not protect it), and the malformed rows kept byte-for-byte as they are found are registered individually, so the registration turns red the day it stops being needed rather than outliving its reason — audited in both directions, since a registration pointing at a row that is no longer malformed and one pointing at a guard that was reclassified or deleted are both exemptions nobody reads |
|
|
413
413
|
| `scripts/run-mcp-probe-face-test.mjs` | The engine's **run-free MCP status face** (server ≥7.93.0, sdk 11.2.0): the `capabilities.mcpProbe` bit read the same four-state way as its eleven sibling capability readers, and two call ports on top of it — read the deployment's own declared servers, or probe a caller-supplied list. Presence of the bit is carried by the engine version, so an **absent key means an older engine** (that route answers a coded 404) and is read as *not reported*, never as *no*; an explicit `false` is the engine's own no and is reported without inventing a reason (whether caller-supplied declarations are accepted is a different bit's question); a non-boolean is malformed and is dropped rather than folded into a no. The availability verdict answers only *should this call go on the wire*: an explicit no means zero requests, and both kinds of *cannot tell* are sent anyway, so an older engine answers with its own coded refusal instead of being silently skipped. One failure judge serves both ports and asks **provenance before status**: a 4xx, 501 or 503 that carries no machine code proves nothing about who answered and is reported as *no verdict*; twelve coded refusals each get their own arm (three identity codes, one of which sits outside the `auth.` family so a prefix fallback would miss it; three different treatments ride the same 400), and a coded answer this version does not recognise lands in *cannot tell*, never in *this engine has no such face*. Retry-after seconds ride 429 and 503 and are absent rather than 0 when the engine gave none. The 200 body is narrowed through the **same per-row narrower** as the streaming `wiring_manifest.mcp[]` leg and the session panel's replay leg, liveness cell included; on this face rows are paired with the submitted declarations **by index**, so a dropped row makes the whole answer unreadable rather than a half table, a row count that differs from the submitted count is reported as misaligned, and an honest empty `servers: []` is kept apart from *non-empty but nothing readable*. `probedAt` and `ttlSec` pass through untouched — this package mints no freshness verdict — the declaration list is handed over as-is with its length snapshotted once, a list that serialises itself differently from what was counted is refused locally with zero requests, and no port ever retries a dial. |
|
|
414
414
|
| `scripts/run-sdk-wire-transit-test.mjs` | The package's pass-through of a few SDK names (`sdkWireTransit`), pinned: every value re-export is the **same reference** as the SDK's own (a client class the host recognises with `instanceof`, the two approval-frame predicates, the three session-bundle calls and the SDK error class), not a look-alike wrapper — wrapping would discard the one anti-drift guarantee a pass-through has; every type re-export is present by name in the emitted declaration file; the gate's list and the source file's export lists are compared in both directions so a name added to one without the other turns red; and a name the SDK does not export must fail the same test, so the gate is not vacuously green. |
|
|
415
|
-
| `scripts/run-sdk-registry-transit-test.mjs` | The package's cloud control-plane surface (`sdkRegistryTransit`), which lives behind its own `./registry` subpath entry point rather than on the root barrel, and this guard holds both halves of that decision. Upstream publishes the same surface behind a subpath of its own, because the subject differs: the engine-wire surface speaks for one engine's service credential, this one for a person's rotating token, and their refresh and error semantics were deliberately never merged. Keeping it behind a second entry point means a client that never touches the control plane neither resolves nor type-checks it. Every value re-export is therefore read from the file the subpath entry actually resolves to, and must be the **same reference** as upstream's own (the control-plane client class, the three config reads, the health probe, the feedback call, the auth-path constant, the content-address helper, and the two typed error classes a host recognises with `instanceof`); every type re-export is compared with the emitted declaration file in both directions; the gate's list and the source file's export lists are likewise compared both ways; and the root entry is checked to carry none of these names, with the root barrel's own source checked to reference the entry file nowhere — a name leaking onto the root would put the cost of this surface back on clients that never asked for it. The subpath is then verified end to end: the installed SDK must really publish its own `./registry` entry and declare every transited name inside it, and this package's own `exports` must point that subpath at exactly the files the gate just judged. Portability is two checks rather than one, done with a parser rather than a text search: the upstream subpath's emitted JavaScript, walked recursively, must contain neither a `node:` specifier (static, side-effect, dynamic, `require` and re-export forms all exercised) nor a Node **global** — because the same package's third entry point is a Node-only surface that imports nothing at all and reaches for the `Buffer` global, so a specifier check alone would pass it as isomorphic, while a byte-level search of it reports two `node:` hits that live entirely inside a documentation example. A text scanner that merely strips comments first gets both directions wrong on ordinary JavaScript — a regular-expression literal containing a slash pair swallows the rest of its line, and the word in a string reads as a reference — so both scanners run off the syntax tree and are checked against fixtures for each failure direction as well as against that real material — including a dynamic import written with a template literal, which a check that accepts only quoted strings misses entirely, and a dynamic import whose target cannot be determined statically, which is refused rather than read as no edge at all. Zero-processing is likewise enforced with a syntax-tree allowlist rather than a keyword search: every top-level statement must be a named re-export carrying that one specifier, so an import followed by an in-place edit of the upstream prototype is refused with a file and line — that shape leaves the name lists untouched and even keeps the same-reference check green, since both sides are then the one object that was damaged. The last section measures, rather than merely notes, one declaration-level gap: two of the upstream subpath's declaration files reference a package the SDK lists only among its own dev dependencies. Moving such a check to a scratch directory is not isolation, because package resolution walks the ancestor directories, so the gate builds a sandbox served by a restricted compiler host and proves the isolation both ways — a decoy copy of the missing package placed one level above the sandbox must silence the errors for an unrestricted host and must not silence them for the restricted one. Then it installs this package into that same sandbox as a real consumer would, and pins the two readings that justify the entry-point split: a consumer that imports only from the root sees no unresolved-module errors at all, while a consumer that imports the subpath sees exactly the two, reported honestly rather than swallowed by this layer. The day upstream ships those declarations, that section turns red and the
|
|
415
|
+
| `scripts/run-sdk-registry-transit-test.mjs` | The package's cloud control-plane surface (`sdkRegistryTransit`), which lives behind its own `./registry` subpath entry point rather than on the root barrel, and this guard holds both halves of that decision. Upstream publishes the same surface behind a subpath of its own, because the subject differs: the engine-wire surface speaks for one engine's service credential, this one for a person's rotating token, and their refresh and error semantics were deliberately never merged. Keeping it behind a second entry point means a client that never touches the control plane neither resolves nor type-checks it. Every value re-export is therefore read from the file the subpath entry actually resolves to, and must be the **same reference** as upstream's own (the control-plane client class, the three config reads, the health probe, the feedback call, the auth-path constant, the content-address helper, and the two typed error classes a host recognises with `instanceof`); every type re-export is compared with the emitted declaration file in both directions; the gate's list and the source file's export lists are likewise compared both ways; and the root entry is checked to carry none of these names, with the root barrel's own source checked to reference the entry file nowhere — a name leaking onto the root would put the cost of this surface back on clients that never asked for it. The subpath is then verified end to end: the installed SDK must really publish its own `./registry` entry and declare every transited name inside it, and this package's own `exports` must point that subpath at exactly the files the gate just judged. Portability is two checks rather than one, done with a parser rather than a text search: the upstream subpath's emitted JavaScript, walked recursively, must contain neither a `node:` specifier (static, side-effect, dynamic, `require` and re-export forms all exercised) nor a Node **global** — because the same package's third entry point is a Node-only surface that imports nothing at all and reaches for the `Buffer` global, so a specifier check alone would pass it as isomorphic, while a byte-level search of it reports two `node:` hits that live entirely inside a documentation example. A text scanner that merely strips comments first gets both directions wrong on ordinary JavaScript — a regular-expression literal containing a slash pair swallows the rest of its line, and the word in a string reads as a reference — so both scanners run off the syntax tree and are checked against fixtures for each failure direction as well as against that real material — including a dynamic import written with a template literal, which a check that accepts only quoted strings misses entirely, and a dynamic import whose target cannot be determined statically, which is refused rather than read as no edge at all. Zero-processing is likewise enforced with a syntax-tree allowlist rather than a keyword search: every top-level statement must be a named re-export carrying that one specifier, so an import followed by an in-place edit of the upstream prototype is refused with a file and line — that shape leaves the name lists untouched and even keeps the same-reference check green, since both sides are then the one object that was damaged. The last section measures, rather than merely notes, one declaration-level gap: two of the upstream subpath's declaration files reference a package the SDK lists only among its own dev dependencies. Moving such a check to a scratch directory is not isolation, because package resolution walks the ancestor directories, so the gate builds a sandbox served by a restricted compiler host and proves the isolation both ways — a decoy copy of the missing package placed one level above the sandbox must silence the errors for an unrestricted host and must not silence them for the restricted one. Then it installs this package into that same sandbox as a real consumer would, and pins the two readings that justify the entry-point split: a consumer that imports only from the root sees no unresolved-module errors at all, while a consumer that imports the subpath sees exactly the two, reported honestly rather than swallowed by this layer. The day upstream ships those declarations, that section turns red; the registered count is then updated and the section stays as a guard against a regression. The separation itself rests on the root closure being computed correctly, so the portability guard that computes it was extended in the same change: a template-literal dynamic import is followed like any other edge, and an edge whose target cannot be resolved statically is refused on every one of the four graphs — without that, a single line in a third file already reachable from the root would put this surface back into the root runtime while every guard stayed green. Since 0.84.0 the upstream subpath ships those declarations itself: the measured count is now zero, the section stays as a guard (a new missing declaration package turns it red again), the isolation proof uses a synthetic probe that imports only the decoy, and the gate also checks that no upstream subpath declaration references the missing package any more. |
|
|
416
416
|
| `scripts/run-rule-removal-consequence-test.mjs` | The one sentence a rule-removal confirmation surface shows for what removing the rule will actually do: one line per behaviour the rule could have been enforcing, plus a neutral line for when that behaviour cannot be read back, so a caller that hits an out-of-set or missing value never falls back to a specific claim it cannot support. The guard compares the actual output against frozen text rather than merely checking that some string came back, so a dropped word or a swapped clause is caught the day it lands, and the four sentences are pinned pairwise distinct. The line for a rule that was denying something is pinned to say the removal widens what can run rather than echoing the wording used for a rule that asks again — the two are opposite directions, and sharing a sentence between them would tell the person confirming the removal the opposite of what is about to happen. The neutral line is checked from the other side for the same reason: it must not contain a word that belongs to only one of the three behaviours, because that would answer on behalf of a state the caller was unable to determine. The lookup that turns a raw stored or transmitted value into one of the three behaviours reads by strict equality only, proven with a fully trapped proxy and a counting getter to show it never touches a property on whatever it is handed, so a value with a legitimate-looking word sitting on its prototype chain is rejected exactly like any other out-of-set value rather than being unwrapped. A closing self-check mutates one character out of each frozen sentence and asserts the exact-match comparison actually fails on it, so the guard cannot pass by checking only that a string of some kind came back |
|
|
417
417
|
| `scripts/run-mcp-engine-leg-test.mjs` | The two legs behind one row on the MCP detail card. In a two-process setup the servers are hosted by the **engine**, while the Status cell on the card reports **this client's own** connection to them, and the two are independent truths: this client failing to connect does not mean the server's tools are unavailable, and the engine leg on the same screen may be saying it completed an exchange moments ago. Two mints share the work. The first reads one server's liveness off the engine leg as six readings, and its load-bearing distinctions are three. **No roster that could be read end to end** (`null`, `undefined`, anything that is not an array, a length that is not a non-negative integer, a length beyond the scan bound, or rows that could not be read at all with nothing matched) is **not** the same as an **empty** roster, which is the engine leg's positive statement that it declared no servers; the same narrower that feeds this reader answers `undefined` when a non-empty section yields no readable row, so *I could not read it* never turns into *I know it is zero*, and a roster that was not read to the end never produces *this server is not on it*. **Could not be read is not the same as absent**: absence means only that the row carries no liveness key of its own, while a record that is present but unintelligible — a cell that is not an object, a word that is not a non-empty string, or a read that fails outright — is reported as unreadable and is decided before the observation beside it, since a record that may well have said the opposite is no evidence of reachability. **An ambiguous name is answered as ambiguous**: when a roster that was read end to end matches a name more than once, the reader reports the match count instead of picking one, because the upstream contract for the sibling MCP face states that names are not guaranteed unique, and picking optimistically would contradict the worst-fact-first verdict shown on the same screen. When a row's identity could not be read at all the reader makes **no claim about that name whatsoever**, not even a count, since the row it could not read may well be a second one carrying the same name — a row that is plainly not a row, such as a hole in a sparse array, is a different matter and leaves the roster complete. The observed word is passed through **verbatim as an open set**, so a fourth word one day arrives at consumers untouched. The second mint is the one sentence that goes under Status, and it appears **only** when this client's own connection really did fail — the other Status values already tell the truth, and a sentence on top of them would only muddy the verbatim health vocabulary. **The liveness observation outranks the tool roster, and having heard from the engine includes hearing that it could not tell, and hearing something unreadable**: a word meaning *looked and could not tell*, a word this version does not recognise, and a record that arrived but could not be read each get their own sentence rather than falling back to the roster, because falling back would say this client holds nothing at all while the diagnostics page shows that very record. A value that is not a reading at all is a different case and does fall back, since nothing then establishes that the engine sent anything. Only when liveness genuinely cannot speak does the roster get a turn, and the roster itself has four answers — tools were listed, nothing was said, the engine reported zero, and the count could not be read — because **unreadable, absent and zero are three different facts**. No sentence claims that nothing at all has been seen about the server, because on two of the readings that reach the roster the engine has plainly listed it; the sentence for a silent roster states only what this client holds. The nine sentences are pairwise distinct, each one names the engine leg, and **none of them renders the liveness word itself**. Membership of the word list is decided by the one shared predicate rather than a second copy, and the branch over the known words is pinned so that a new word upstream fails the build instead of silently taking the *not recognised* sentence. Neither mint ever throws, whatever a host hands it: every value is taken through one reader that accepts **own properties only** — an inherited key is not a wire fact, and one on a shared prototype could otherwise manufacture an engine observation or suppress a real one — reads each key exactly once, and keeps a read that fails apart from a value that is absent. The roster is walked by index rather than through the array's own `find`, the array test is guarded because it can throw on its own, and a length that overstates itself would otherwise spin forever |
|
|
418
418
|
| `scripts/run-cc-message-key-census-test.mjs` | The one rule behind CC-shaped messages this package emits: every top-level key on a `user` / `assistant` / `result` / `system` message must be a member the CC SDK mirror (`@sema-agent/agent-types`) declares for that same arm, or one of this package's `_sema_`-prefixed additive keys, or a row in a dated transition table that goes red the day its retire version arrives. Mint sites are found syntactically (identifier, quoted and computed `type` names alike) and their key sets are resolved syntactically too — inline literals, both branches of a conditional, the nullish-coalescing and logical or/and operators, `const` initializers and every `return` of a helper — so a type assertion, a `Partial<Pick<…>>` narrowing or a computed name cannot launder a key past the check, while a spread the tool cannot follow (a parameter, a `let`, a member access) fails the tool, never the product. Arms that carry a `subtype` discriminator are checked against that subtype's own member set, so a key declared only for another subtype does not pass on the strength of the arm-wide union. The runtime section drives the real adapter pipeline and pins the tool-result record's SDK-spelled `tool_use_result` (the transitional camelCase twin rides along by reference until 0.83.0). A second runtime section feeds the same tool-result frame to both the interactive adapter and the non-interactive frame builder and requires the structured-result key and the no-output marker to be present together, absent together and equal on the two outputs, while the transitional camelCase name stays on the transcript record only; a static section checks a per-key routing table for the internal tool-result frame against the key sets resolved at the three mint sites, in both directions, and requires the non-interactive frame's structured-result value to trace back to the very same syntax node the transcript record uses — so recomputing or copying that logic on the non-interactive path fails even when the values happen to agree. |
|
|
@@ -429,8 +429,8 @@ guard still cross-checks the table by name).
|
|
|
429
429
|
| `scripts/run-session-policy-deliverable-test.mjs` | Which of a batch of user-written permission rules can be written into a session’s own rule record without changing their meaning, and why each of the others cannot. The record holds whole tool names and command names only, so exactly one class maps across losslessly: a deny rule that names one tool with no qualifier. Everything else is withheld with one word from a closed six-word list — an ask rule (the record has no ask tier), a deny rule with a parenthesised qualifier (recording just the name could block more), a rule that names a server or agent peer without naming one of its tools (for every protocol namespace the engine knows, checked against the engine package's own table) or contains a wildcard (*) anywhere (an engine that compares exact names would block nothing), an entry that is not a tool name, and a name the engine refuses at the start of every run — a retired tool name, or one containing "__" without a protocol prefix, where the prefix check is case-sensitive (once such a name is in the record, every later run of the session fails at startup until that entry is removed; the retired-name list is checked entry by entry against the engine package's own list, and a withheld retired name carries its current name when the engine says it was renamed) — and each word has one sentence, which never echoes the rule itself; asking for the sentence never throws, even with a value that throws when turned into a string. A name with leading or trailing whitespace counts as not a tool name: the record compares exact bytes, so it would block nothing. The guard pins the batch semantics: the deliverable part is either the whole batch or empty, never a subset, so a caller cannot send half a change and report it as saved. An end-to-end check runs the engine package itself: every batch this function would deliver — the recorded vectors and a fixed-seed sample of generated names — is written into an in-memory session rule store and the next run must get past its start-up checks and reach the model, while every name withheld as refused — every retired name included — must indeed make that run fail at start-up, and every string literal in the judgement source that it withholds as refused must be one the engine package's own tables refuse. It also checks that malformed input never throws and never delivers anything (non-arrays, non-string entries, holes, a polluted array prototype, a length or index that throws, a changing index read once), that a batch which cannot be read at all is marked `unreadable: true` while an empty batch is not, so the two stay tellable apart, that each word is produced by some vector and nothing outside the list is produced, and — when a checkout of the previous in-client implementation is present — that this function gives the same answer on every recorded vector and on tens of thousands of generated rules and pairs, except for four deliberately stricter classes (whitespace-padded names; rules with a wildcard anywhere, which the previous implementation sent as exact names unless the wildcard was the whole tool part of a server rule; peer-wide rules outside the MCP namespace, which it did not recognise; and names the engine refuses at start-up, which it sent as ordinary names), whose disagreements are counted per class and must match an independent count exactly. |
|
|
430
430
|
| `scripts/run-plugin-hooks-projection-test.mjs` | Plugin hooks: each command hook an enabled plugin declares is decided one by one as running in the engine, running in this client, or not running at all, and the page of hooks sent with a request is built from the same per-turn plan the client uses to skip its own copies, so one hook never runs in two places. Governance is judged first and always wins — a managed hooks switch-off, an untrusted workspace, safe or bare mode, or a governance read that fails sends no plugin hook and does not list it as a gap; managed-hooks-only (set directly, or through a merged non-managed hooks switch-off) keeps only managed plugins; the plugin-only customization lock does not touch plugin hooks. A hook reaches the engine only when this client started the engine on this machine, the engine reports plugin-hook support, the entry is a command, the plugin declares no sensitive option, and the event still fits the engine's per-event limits; the gate walks that matrix cell by cell, including the limit boundaries and a session goal hook counting toward them. A fact that was never read is reported as not known rather than as a fact: a host that does not say where the engine runs gets a "not known whether this client started the engine" reason, an engine whose capabilities have not been read yet gets a "not known yet whether it supports plugin hooks" reason, and the plan's two engine facts are null in those cases, not false. Events the engine never fires run only if the client says it fires them itself, and hooks the upstream behaviour itself refuses (option references in a shell-form command, an unset option in exec form, malformed entries) run nowhere. Exec-form arguments are passed element by element with only saved non-sensitive option references filled in; path placeholders are left for the executor. Sensitive option values never reach the request: with a host that wrongly supplies one, every string in the plan, the request body, the notice, the labels and the log is searched for it across eight cells. A host without the plugin reader keeps the previous request body and gets exactly one warning per settings port; plugin data that throws while it is being read (a throwing getter, a revoked proxy) is treated like a failing reader — no plugin hooks this turn, settings hooks still sent, nothing thrown; the not-running notice names the plugin and events, never a command or an option value, and escapes control characters in names. Command hooks from settings that carry arguments (a non-empty `args` array, which is the exec form, or any other non-null value) are removed from the request until the engine reports support for arguments, because the engine would otherwise drop the arguments and run the bare command through a shell; an empty `args` array is not treated as carrying arguments when the command is made only of letters, digits and `_ . / : + -` (the shell runs the same executable), so such a guard still reaches the engine, while an empty array on a command with spaces or shell characters is removed; `args` on a prompt or http entry, a null `args`, or an entry with no type is left alone, and those go out unchanged. MCP tool hooks, which the engine cannot parse, are removed only from a request built from a plan, whose not-running notice the host shows; a request built without a plan still carries them on engine-fired events, so the engine rejects the whole request loudly instead of a guard hook silently not running — the gate checks both request bodies against the engine's own hooks schema. Without a plan, every removed hook of that kind on an engine-fired event produces one warning per settings port, event and reason. Malformed entries still pass through for the engine to reject loudly, and passing null where the options object goes behaves like passing nothing; a `plan` option that is not a plan is ignored rather than turning the whole page into nothing, and a plan passed directly in place of the options object is recognised and used. A `plugin` key written by hand on a settings hook is stripped before sending (even when its value is undefined), because only hooks that come from the plugin reader may carry plugin context; the settings document itself is left untouched and a debug line records the count. The two hand-copied tables, the engine-fired event list and the engine limits, are checked against their owners. |
|
|
431
431
|
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. Since 0.83.5 the outlet has two opt-in switches and a position read-out. `escapeBackslashes` (escape form only) writes every literal backslash as a pair, so each output decodes back to exactly one input (a real invisible character and its literal six-character spelling no longer look alike); a reference decoder round-trips thousands of random strings, the output is byte-identical to the default when the input has no backslash, a length cap is measured on the paired output and keeps the longest fitting prefix, credentials are washed exactly as in the default, and the switch is not idempotent by design (use it only at the final render). Zero-width joiners and non-joiners are kept only inside emoji sequences drawn as emoji (so not between symbols such as © or ™ that display as text) and between letters of scripts where they change the shaping (joining scripts such as Arabic, and the Brahmic family), each listed script checked both ways; the one other place a joiner is kept is right after a virama at the end of a word, the older spelling still found in Malayalam and Bengali text. Next to Latin, Cyrillic, CJK and other letters, next to modifier letters shared across scripts, at the start of a word, at the end of a word without a virama before it, or on their own they are now marked. `blanks` marks characters that look like a space but are not an ASCII space (no-break and other width spaces, the ideographic space, Hangul fillers, the blank Braille pattern) before whitespace folding, for names that must never look alike. `displayUntrustedMarks` returns the same text plus the position of every character mark; its text is compared with the outlet over thousands of inputs. With both credential nets off, the character face stays byte-identical to the previous release for input without joiners. The credential nets read escape sequences the way a terminal would when one is cut short: an unfinished colouring or character-set sequence interrupted by another one is dropped as a whole, a sequence never takes the `@` of an address as its final character, and a final character that starts a well-known secret literal (`sk-`, `ghp_`, `AKIA`, a JWT) is also read as the start of that literal; escape tokens this outlet writes are read as one unit, while look-alike text it never writes (an upper-case `\U`, or a code point it never marks) is read as plain text. A URL is cut before a run of non-ASCII blanks followed by a credential label or scheme word, Hangul fillers and the blank Braille pattern count as spaces around a label's separator, and a value that itself starts with a quoted label (`token= "password":"..."`) is left to that inner label. With `blanks` on, a blank written as an escape token right after a label is read as a blank when the value is judged, so `password:` followed by a no-break space and `missing` stays unmasked and an empty value gets no marker; the credential nets also read the text the way it looks after default whitespace folding and combine what each reading masks, so whatever the default form masks stays masked with `blanks` on (checked over a seeded corpus for the escape form, with paired backslashes and without folding; the exceptions are text that itself contains a literal six-character blank escape, which cannot be told apart from one the outlet wrote, and the dot and space marks, which cannot tell a marked blank from a real dot or space). A lone surrogate wedged between a label, its separator and its value no longer hides the value: the credential nets treat it exactly like a format character, in the machine-readable wash, in a single pass of the display outlet, and on the end-of-run rows and the result frame's `errors[]` on both lanes, while lone surrogates anywhere else are left byte-for-byte. |
|
|
432
|
-
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. |
|
|
433
|
-
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. |
|
|
432
|
+
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. Since 0.84.0 the product table itself is also checked against the SDK's runtime list of known origin words, so a word the upstream does not know cannot sit in the table. |
|
|
433
|
+
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. From 0.84.0 the package also reads the registry itself: a reader lists the session's background registry only when the server advertises the listing, classifies its failures as unavailable (a failed read keeps the HTTP status and error code when the error carries them, and never invents either), and refuses a partly readable listing as a whole; a classifier turns the listing into the per-row reading, treating a registry status it cannot place as unknown and reading "not listed" only for keys shaped like registry handles and only when the host states the engine is a single process and names the row's session, because a key from another identity space proves nothing by being absent and, on a multi-replica deployment, a listing answered by another replica does not contain this session's agents at all (without that statement a missing row reads unknown and the row stays); and a predicate returns the agents the registry still counts as live that the host has no row for. A listing at the server's 500-row cap cannot show that an agent is gone either, so a missing row there reads unknown; the cap is checked against the engine's own list clamp and, when a server build is supplied, against the server's route. Each listing carries its session and a per-client sequence number, and a listing superseded by a later delivered read, or read for another session than the one the host names, counts for nothing. Only the exact listing object the reader returned can authorize removing a row or adding one: a spread copy, a structured clone, a filtered or a hand-built listing reads unknown and adds nothing, the listing, its row array and every row are frozen, the cap check uses the row count recorded when the page was read, and rewriting a listing's sequence number fools nothing. The fill predicate holds back agents first seen within the absence settle window — timed on one monotonic local clock from when this client's reader first saw the id, so a skew between the server's clock and this one, or reconnecting to a long-running agent, cannot skip the window — agents without a readable registration time and agents the host saw end since the read went out; the registry key of a row, progress or absence event is whichever of its two ids is shaped like a registry handle; and none of the reader, the classifier or the predicate throws. |
|
|
434
434
|
| `scripts/run-plan-review-dismissal-test.mjs` | An automatic reopen does not put back a plan-review card the user closed (the first-presentation path neither checks nor clears that record, so a replayed park frame still presents its card). A plan-review card the user dismissed (Esc, abort, or any answer that is not approve or reject) is recorded per session and run at the moment of dismissal, synchronously, before anything queued behind the card can be released; a reopen marked `trigger: 'automatic'` then refuses with `{ reopened: false, dismissedByUser: true }` instead of minting a new card the user's next keystroke would land on, while the user's own next action (`trigger: 'user'`) reopens it and clears the record. The record is keyed by gate instance when the host supplies an instance reader: a new plan gate on the same run is still surfaced, and anything that cannot prove the gate is new (no reader, a failed, empty, thrown or timed-out read) refuses on the conservative side. The asynchronous form re-checks after its reads and before minting — a decision handed over meanwhile (seen by the package, or reported by the host's optional hand-over predicate) answers as "your answer is on its way"; a record that changed meanwhile makes the stale evaluation mint nothing and answer from the current record: another close refuses as the user's close and keeps the newer record, a card already back on screen (the user's own action or a concurrent automatic reopen, waited for within the receipt window and re-read once the wait is over) answers `reopened: true`, and a record that is gone (session change, ledger overflow) answers a plain refusal without `dismissedByUser`; a hand-over predicate that throws refuses with a plain `{ reopened: false }`. Each run has at most one reopen on its way: a reopen that arrives while an earlier one's card is published but not yet settled joins it instead of minting a second card and retiring the first card's answer path. Instance readers are snapshotted when they resolve, so a host that hands over its own live set still gets a new gate recognised; and a decisive-looking host answer note does not clear the record for the very card the package's own responder already judged non-decisive (the label was not on that card). A successful reopen replaces only the record taken before the card was minted (with an on-screen marker, not a deletion), a decisive answer clears it, the per-session ledger is bounded, and a session change clears its bucket. Without `trigger` the reopen answers exactly as before, apart from joining a reopen already on its way. |
|
|
435
435
|
| `scripts/run-gate-interrupt-safety-test.mjs` | The two safety properties of running the suites themselves. The collecting runner no longer kills a suite outright when its time limit expires: it sends a termination signal first, waits a grace period for the suite to clean up, and only then kills it — and it records the suite as timed out however it exits, so a suite that exits cleanly after the signal is still not counted as passing. The limit and the grace period can be widened for a single suite in `gates-manifest.json` (an optional entry with a written reason; a malformed entry, including one with a misspelled field, stops the runner before any suite starts instead of silently falling back to the default, and an entry filed under a misspelled name is reported by this guard rather than ignored), and the negative-control suite is widened there. Every property is exercised on byte-for-byte copies of the real runner and the real negative-control suite in a throwaway directory: a suite that honours the signal finishes within the grace period, one that ignores it is killed when the grace period ends, and the summary lines are unchanged; once a suite has exited the runner waits at most a short drain window for its output pipes, so a child process that inherited them and outlives the suite neither turns a passing suite into a timeout nor holds the runner past the limit, the grace period and that window; the negative-control suite, interrupted in the middle of a rehearsal by any of the three signals or by the runner's own time limit, restores the file byte-for-byte, leaves no backup behind, stops the rehearsed guard together with anything it started, regenerates the build output (checked on a small project in the throwaway directory), and exits with 128 plus the signal number; a rebuild that would overrun the grace period is cut short and its compiler stopped; a backup left behind by an earlier run — next to a later target, loose in the source tree, or inside a symlinked dependency directory — makes it refuse to start without touching anything, naming the file and how to restore it. |
|
|
436
436
|
|
package/dist/adapt/wireShapes.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
|
+
import { frozenSet } from '../frozenSet.js';
|
|
1
2
|
export const FLUSH_INTERVAL_MS = 100;
|
|
2
3
|
export const IDLE_FLUSH_MS = 1500;
|
|
3
|
-
export const SUBAGENT_TOOL_NAMES =
|
|
4
|
-
export const WORKFLOW_TOOL_NAMES =
|
|
4
|
+
export const SUBAGENT_TOOL_NAMES = frozenSet(['Task', 'Agent', 'Fork']);
|
|
5
|
+
export const WORKFLOW_TOOL_NAMES = frozenSet(['Workflow', 'RunWorkflow', 'run_workflow']);
|
|
5
6
|
export const TASK_TOOL_NAMES = new Set(['TodoWrite', 'TaskCreate', 'TaskUpdate']);
|
|
6
7
|
export const REJECT_MESSAGE = "The user doesn't want to proceed with this tool use. The tool use was rejected (eg. if it was a file edit, the new_string was NOT written to the file). STOP what you are doing and wait for the user to tell you how to proceed.";
|
|
7
8
|
export const CANCEL_MESSAGE = "The user doesn't want to take this action right now. STOP what you are doing and wait for the user to tell you how to proceed.";
|
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
import { abortableSleep } from '../abortableSleep.js';
|
|
2
2
|
import { corruptStoredRowContent, corruptStoredRowWhere } from '../wireErrorTriage.js';
|
|
3
3
|
export const PLAN_REVIEW_GATE_KIND = 'plan_review';
|
|
4
|
-
export const PLAN_REVIEW_GATE_KINDS = [PLAN_REVIEW_GATE_KIND, 'dry_run_review'];
|
|
5
|
-
export const ASK_PARK_GATE_KINDS = ['human', 'irreversible_ask', 'policy_ask', 'tool_approval'];
|
|
4
|
+
export const PLAN_REVIEW_GATE_KINDS = Object.freeze([PLAN_REVIEW_GATE_KIND, 'dry_run_review']);
|
|
5
|
+
export const ASK_PARK_GATE_KINDS = Object.freeze(['human', 'irreversible_ask', 'policy_ask', 'tool_approval']);
|
|
6
6
|
export function askParkForeignGateKind(row) {
|
|
7
7
|
const kind = typeof row.gateKind === 'string' && row.gateKind.length > 0 ? row.gateKind : null;
|
|
8
8
|
return kind !== null && !ASK_PARK_GATE_KINDS.includes(kind) ? kind : null;
|
|
9
9
|
}
|
|
10
|
-
export const PLAN_REVIEW_STATES = ['needs_review'];
|
|
11
|
-
export const ASK_PARK_STATES = ['suspended'];
|
|
12
|
-
export const RUNNING_STATES = ['running'];
|
|
13
|
-
export const CLAIM_RELEASED_STATES = ['completed', 'failed', 'blocked', 'timeout'];
|
|
14
|
-
export const CLAIM_HELD_STATES = ['running', 'suspended', 'needs_review'];
|
|
10
|
+
export const PLAN_REVIEW_STATES = Object.freeze(['needs_review']);
|
|
11
|
+
export const ASK_PARK_STATES = Object.freeze(['suspended']);
|
|
12
|
+
export const RUNNING_STATES = Object.freeze(['running']);
|
|
13
|
+
export const CLAIM_RELEASED_STATES = Object.freeze(['completed', 'failed', 'blocked', 'timeout']);
|
|
14
|
+
export const CLAIM_HELD_STATES = Object.freeze(['running', 'suspended', 'needs_review']);
|
|
15
15
|
export function selfHealSubmissionDisposition(outcome) {
|
|
16
16
|
switch (outcome.kind) {
|
|
17
17
|
case 'decision-pending':
|
|
@@ -3,14 +3,16 @@ import { stamp, snapshotSegmentIdentity, } from '../types.js';
|
|
|
3
3
|
import { turnUsageToModelUsage } from './turnUsageToModelUsage.js';
|
|
4
4
|
import { ccToolDenialKindForToolEnd, gateOutcomeOf } from '../../gateOutcome.js';
|
|
5
5
|
import { projectToolRoster, projectToolRosterDelta } from '../../toolRoster.js';
|
|
6
|
+
import { projectHandsSection } from '../../handsSeam.js';
|
|
6
7
|
import { readMcpLiveness } from '../../mcpLiveness.js';
|
|
7
8
|
import { SEGMENT_END_IDENTITY_KEYS } from '../types.js';
|
|
9
|
+
import { frozenSet } from '../../frozenSet.js';
|
|
8
10
|
function armBody(body) {
|
|
9
11
|
return body;
|
|
10
12
|
}
|
|
11
13
|
const SUGGESTION_BATCH_CAP = 8;
|
|
12
14
|
const SUGGESTION_CHARS_CAP = 200;
|
|
13
|
-
export const INTERNAL_SDK_ARM_TYPES =
|
|
15
|
+
export const INTERNAL_SDK_ARM_TYPES = frozenSet([
|
|
14
16
|
'tool_disclosure',
|
|
15
17
|
'tool_progress',
|
|
16
18
|
'tool_roster_delta',
|
|
@@ -546,7 +548,8 @@ function wiringManifestSupersetBody(ev) {
|
|
|
546
548
|
const writeProtection = projectWriteProtectionSection(ev.writeProtection);
|
|
547
549
|
const autoConsolidation = projectAutoConsolidationSection(ev.autoConsolidation);
|
|
548
550
|
const readDeny = projectReadDenySection(ev.readDeny);
|
|
549
|
-
|
|
551
|
+
const hands = projectHandsSection(ev);
|
|
552
|
+
if (modelGate === undefined && autoMode === undefined && mcp === undefined && tools === undefined && hooks === undefined && lsp === undefined && writeProtection === undefined && autoConsolidation === undefined && readDeny === undefined && hands === undefined)
|
|
550
553
|
return undefined;
|
|
551
554
|
return {
|
|
552
555
|
...(modelGate !== undefined ? { _sema_modelGate: modelGate } : {}),
|
|
@@ -558,6 +561,7 @@ function wiringManifestSupersetBody(ev) {
|
|
|
558
561
|
...(writeProtection !== undefined ? { _sema_writeProtection: writeProtection } : {}),
|
|
559
562
|
...(autoConsolidation !== undefined ? { _sema_autoConsolidation: autoConsolidation } : {}),
|
|
560
563
|
...(readDeny !== undefined ? { _sema_readDeny: readDeny } : {}),
|
|
564
|
+
...(hands !== undefined ? { _sema_hands: hands } : {}),
|
|
561
565
|
};
|
|
562
566
|
}
|
|
563
567
|
function projectWriteProtectionSection(raw) {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { WiringManifestModelGate, WiringManifestAutoMode, WiringManifestMcpEntryView, WiringManifestHookEntryView, WiringManifestLspSeamView, WiringManifestWriteProtectionView, WiringManifestAutoConsolidationView, WiringManifestReadDenyView } from './eventToSdkMessage.js';
|
|
2
|
-
import type { ToolRosterView } from '../../toolRoster.js';
|
|
2
|
+
import type { ToolRosterView, WiringManifestHandsView } from '../../toolRoster.js';
|
|
3
3
|
export type WiringManifestSections = {
|
|
4
4
|
_sema_modelGate?: WiringManifestModelGate;
|
|
5
5
|
_sema_autoMode?: WiringManifestAutoMode;
|
|
@@ -10,6 +10,7 @@ export type WiringManifestSections = {
|
|
|
10
10
|
_sema_writeProtection?: WiringManifestWriteProtectionView;
|
|
11
11
|
_sema_autoConsolidation?: WiringManifestAutoConsolidationView;
|
|
12
12
|
_sema_readDeny?: WiringManifestReadDenyView;
|
|
13
|
+
_sema_hands?: WiringManifestHandsView;
|
|
13
14
|
eventId?: string;
|
|
14
15
|
};
|
|
15
16
|
export interface WiringManifestView {
|
|
@@ -22,6 +23,7 @@ export interface WiringManifestView {
|
|
|
22
23
|
writeProtection?: WiringManifestWriteProtectionView;
|
|
23
24
|
autoConsolidation?: WiringManifestAutoConsolidationView;
|
|
24
25
|
readDeny?: WiringManifestReadDenyView;
|
|
26
|
+
hands?: WiringManifestHandsView;
|
|
25
27
|
eventId?: string;
|
|
26
28
|
}
|
|
27
29
|
export declare function wiringManifestViewOf(s: WiringManifestSections): WiringManifestView | undefined;
|