@sema-agent/client-core 0.86.0 → 0.87.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/README.md +6 -4
- package/dist/adapt/arms.js +9 -0
- package/dist/adapter/downstream/terminalToSdkResult.js +4 -1
- package/dist/adapter/runStream.js +75 -2
- package/dist/adapter/types.d.ts +3 -0
- package/dist/decideReceipt.d.ts +1 -1
- package/dist/decideReceipt.js +1 -1
- package/dist/detachWire.d.ts +6 -1
- package/dist/detachWire.js +22 -1
- package/dist/hitl/crashConverged.d.ts +1 -0
- package/dist/hitl/crashConverged.js +17 -2
- package/dist/hitl/hitlBridge.d.ts +4 -0
- package/dist/hitl/hitlBridge.js +6 -0
- package/dist/hitl/parkResolver.js +17 -12
- package/dist/hitl/suspendedReopen.d.ts +7 -0
- package/dist/hitl/suspendedReopen.js +38 -7
- package/dist/hitl/toolApprovalWire.js +4 -5
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/resumeRefusalCopy.d.ts +2 -7
- package/dist/resumeRefusalCopy.js +2 -24
- package/dist/rewindArchiveCapability.js +5 -0
- package/dist/seam.d.ts +11 -1
- package/dist/seam.js +7 -0
- package/dist/toolHistoryMismatch.d.ts +63 -0
- package/dist/toolHistoryMismatch.js +425 -0
- package/dist/workflowClient.d.ts +3 -2
- package/dist/workflowClient.js +3 -3
- package/docs/INTEGRATION-CLIENTS.md +232 -8
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -49,6 +49,61 @@
|
|
|
49
49
|
> 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
|
|
50
50
|
> commit 漏转在发布前就红,不再靠人记。
|
|
51
51
|
|
|
52
|
+
## 0.87.0(2026-10-01)
|
|
53
|
+
|
|
54
|
+
> 主题:**minor** —— 六票同发(按票序)。① **会话历史 tool_use 错配的自救出路**(CC-268;终端的请托):provider 因为会话历史里一只工具调用对不上结果而拒掉这一会话的每一次请求时,合成终局错误行带机读超集键 `_sema_tool_history_mismatch`;本面真有 `/rewind` 命令的宿主钉 `EmitContext.offersRewind`(`true`,或对象形 `{ engineBaseUrl }` —— 那台引擎明说没有对话锚时不加句)后,行尾另起一行出路句(与 CC 同句),同一会话连续第二次起再补一句次数;另出两只根入口纯函数 —— 判型 `classifyToolHistoryMismatch` 与回退节点解析 `resolveToolHistoryRewindTarget`。② **崩溃收敛 needsHuman 桶的因由位**(CC-269):`projectCrashConverged` 的 needsHuman 行带闭集四词 `needsHumanReason`,分桶判据一字未改。③ **停车 fail-soft 真因超集键**(CC-273):HITL 桥在同步腿终帧在场时 fail-soft 收场,那只终帧照旧原样回吐、顶层多 `_sema_park_failed_reason`(与 durable 腿合成终帧的 `errorMessage` 逐字相等),结果帧透传同名键。④ **workflow 活动台账与监视器源的凭证位收取值函数**(CC-274):连线身份改按凭证来源认,宿主持同一只取值函数时令牌轮换不再清账。⑤ **detach 取消兜底的 arm 口收取值函数**(CC-275):取件口每调一次现读。⑥ **chrome 臂 `suspended_reopened`**(CC-276):流上 `suspended{reopened}` 投成一条 chrome 事件(读数 + reopen 族单源句 + 幂等键)。**为什么是 minor**:两只新顶层导出(`classifyToolHistoryMismatch` / `resolveToolHistoryRewindTarget`)+ `ChromeEvent` 新臂 `suspended_reopened`(对 `ChromeEvent['kind']` / `ChromeArmKind` 做穷尽 `Record` / `switch` 的 TypeScript 端编译红,与 0.86.0 同口径)+ 四处可选位 / 入参放宽(`EmitContext.offersRewind?`、`CrashConvergedRow.needsHumanReason?`、`LiveWorkflowConfig.authToken` / `WorkflowActivityLedgerConfig.authToken` 收取值函数、`armDetachCancel` 入参 `authToken` 收取值函数;对构造方均 additive)。型面 BREAKING 零处(对构造方;读那两只配置位 `authToken` 并按「串 / 中继」二分的宿主代码会编译红,见 Changed 第 5 条)。根公面运行期导出 1354 → 1356(+2);测试钩 62 不变;公面类型名 922 不变(两只新口的入参 / 出参型名不上入口,宿主按结构传;chrome 新臂是 `ChromeEvent` 联合的内联成员);超集键 +2(`_sema_tool_history_mismatch` / `_sema_park_failed_reason`);投影臂 +1(chrome `suspended_reopened`;`CHROME_ARMS` +1 行,`required:false`);另有**已发导出的可观察变化**在 Changed 逐条单列(第 1–6 条,写明谁要跟;第 7 条是公面零变化的内部搬家)。peer sdk 地板 `>=12.0.1` 不动;开发依赖不变。
|
|
55
|
+
|
|
56
|
+
### Added
|
|
57
|
+
|
|
58
|
+
- **会话历史对不上时的自救出路**(CC-268;接入文档 **§113 TH-1 / TH-5 / TH-6**):provider 因为会话历史里一只工具调用没有对应的结果(或一只结果没有对应的调用、或一只调用 id 重复)而拒掉这一会话的每一次请求时,合成终局错误行(`API Error: …`)多一只机读键 `_sema_tool_history_mismatch: { kind, toolUseId?, consecutive? }`。`kind` 取 `tool_use_without_result` / `orphan_tool_result` / `duplicate_tool_use_id` 之一;`toolUseId` 是 provider 点名的第一只 id(provider 那句话没点名、或读不出时缺席);`consecutive`(只从第二次起在场)= 同一会话连续几次 run 撞上同一错配。这只键只铸在那一行上,且只在失败是三句 provider 模板之一时铸;别的终局行与 0.86.0 逐字节同。`adapt()` 重建这一行时接力这只键;原样转发这一行的宿主(例如 print 车道的 JSON 流)也带着它。
|
|
59
|
+
- 判型从不读 assistant 正文。别的状态(401 / 404 / 429)、别的 400(例如模型名无效)与引用这句话的形都不认。provider 回体是 JSON 时只认错误信封上的那句话(顶层 `message` 或 `error.message`;回体须从开头就是这个对象,宿主直接交 SDK 错误对象时的 `400 {…}` 形同认),请求回显、数组成员、更深一层对象里的同名键都不认。宿主直接交 SDK 错误对象、回体不是 JSON 时的 `400 <provider 原话>` 形同认(只剥一个状态词)。
|
|
60
|
+
- `toolUseId` 按整只读:带字符集外字符(例如 `/`)、后接截断标记 `…`、或顶在失败文本末尾(被截断)时整只缺席,不取前缀。token 后面紧跟列表分隔符(`,` / `;`)时整只就是 id、一个字符都不剥(带尾冒号的 `call_1:` 照收);后面是空白或结尾时至多剥一个句末 `.`,剥完仍以 `.` / `:` 收尾 ⇒ 分不清 id 后缀与句末标点 ⇒ 缺席。孤儿结果那一句在 `` `tool_result` `` 之后带 `blocks`、`block` 或直接接冒号都认。
|
|
61
|
+
- 连续次数按进程、按会话记(键是引擎回显的会话 id),只数不同的 run(同一次 run 被重看不加码)。回放的终局、不带 run id 的终局(含失败文本为空的失败事件帧)与结局读不出的终局不记次数,但与这一次错配不同时照样清零;这一会话的任何别的结局清零。先开后收的老流晚到的终局**只许清零、不许加码**:与行上指纹不同(别的结局 / 别的错配)⇒ 本段到此为止、不答次数;同指纹 ⇒ 不改账;较新的流收到的任何终局(包括不改次数的那几形)都推水位。终局上没有引擎回显的会话 id 时(失败事件帧、会话占锁的 409 拒绝、结局读不出的 done)按宿主开流时钉的 id 记;那只 id 从没被引擎回显过(宿主每轮现铸会话 id)时这一终局认不出属于哪一只会话,本进程里每一只会话正在数的连续都按被打断处理(下一次从 1 起)。拿不准时不报次数,不多说。
|
|
62
|
+
- **`EmitContext.offersRewind?: boolean | { engineBaseUrl: string }`**(CC-268;**§113 TH-2 / TH-3**):本面有 `/rewind` 命令的宿主钉它;错配那一行随之在行尾另起一行 `Run /rewind to recover the conversation.`(与 CC 同句),从连续第二次起再多一句,给出次数并说明 `/rewind` 是恢复这一会话的唯一出路(` The same error has now happened N times in a row; /rewind is the only way to recover this conversation.`)。缺席、`false`、`print` 车道与 `utility` 车道都不加句。结果帧 `errors[]` 恒是引擎原文。
|
|
63
|
+
- 对象形 `{ engineBaseUrl }`(推荐):本包按那台引擎查自己的「只回退对话」判据(与 `/rewind` 菜单同一只 `rewindConversationAvailability`)。引擎明说没有对话锚时两句都不加,键照铸;引擎说有、或判不出(没报、读不出、没观测过)时照加。`true` 不查这只判据,由宿主自己担保。
|
|
64
|
+
- **`classifyToolHistoryMismatch({ code, status, message })`** ⇒ `{ kind, toolUseId? } | null`(CC-268;**§113 TH-7**):与那一行同一只判型,给手里握着失败本身(run 的失败码、provider 状态、失败文本)的宿主。只认 HTTP 400(状态 `400`,或失败文本以 `<标签> HTTP 400:` 起头);给了码时只认 `invalid_request`;且 provider 那句话须在 provider 消息的开头(JSON 回体里是顶层 `message` 或 `error.message` 的开头)。永不抛。
|
|
65
|
+
- **`resolveToolHistoryRewindTarget(rows, mismatch)`** ⇒ `{ target } | { rootReopen: true } | { undecidable: reason }`(CC-268;**§113 TH-8**):交进宿主自己的转录行(每行可带 `toolUseIds` / `toolResultIds` / `prompt` / `engineHandle` / `local` / `sessionRoot`),找最早提到那只 id 的一行,再回溯到那一轮的 prompt:答那只 prompt 行本身(回退到它之前);那只 prompt 是会话第一轮时答从头开始;判不了时答六个理由之一(`no-mismatch` / `no-tool-use-id` / `id-not-found` / `prompt-not-found` / `prompt-without-handle` / `unreadable-rows`)。跳过宿主本地行,绝不越过没有引擎句柄的 prompt,不改入参,永不抛(`mismatch` 的位读不出 ⇒ `no-mismatch`;行读不出 ⇒ `unreadable-rows`)。
|
|
66
|
+
- **崩溃收敛 needsHuman 桶的因由位**(CC-269):`projectCrashConverged` 产出的 `needsHuman[]` 每一行多一位 `needsHumanReason`(`CrashConvergedRow` 上的可选位,闭集四词):`approved_then_interrupted`(`decided` ∧ `resumeSafe:false`:人当时批了、执行窗内引擎断了)/ `pending_unsafe`(`pending` ∧ 无已批证据 ∧ `resumeSafe:false`)/ `unstable_row`(载体带取值器或自带原型)/ `contradictory`(`pending` 却带已批证据,或 `decided` 却 `resumeSafe:true`)。优先序:载体不可信 > 判据位冲突 > 人已批 > pending 却不安全。与分桶出自**同一次读**;`resumeSafe[]` 的行不带;供给行上碰巧有同名位 ⇒ needsHuman 行被覆盖、resumeSafe 行被摘掉。分桶判据一字未改。接入文档 **§113 N-1 / 113a′ 第 4 条**。
|
|
67
|
+
- **停车 fail-soft 真因超集键 `_sema_park_failed_reason`**(CC-273):HITL 桥(`bridgeAskUserQuestionGates`)在一张卡决不成、fail-soft 收场时,若这张 park 来自同步腿终帧 `done{suspended}`,那只终帧**照旧原样**回吐(不造第二只终帧,`type` / `id` / `result` 同一只对象),但顶层多一位 `_sema_park_failed_reason` = 真因 + 出路那一句,与 durable 腿同一失败合成的 `failed{errorCode:'hitl_unanswered'}` 的 `errorMessage` 逐字相等;结果帧投影(`terminalToSdkResult` / `runStream`)把它透传到 CC 结果信封同名键,出口与 `errors[]` 过同一只凭据洗消(句中 URL 的 userinfo / 查询串换成记号;帧上那一位是原句),`errors[]` / `subtype` / `is_error` 一字不动。触顶收场两形(连续非进展轮 / 决了又停回来 `hitl_stalled`)与宿主撤卡收场同样带各自那一句。缺席 = 没有能给用户的真因:用户 Esc、普通终帧,以及这一端没有决断面(宿主没装审批卡口、卡口装在别的会话键上、问答腿没有 overlay —— 真因是装配诊断,「卡再出现时在卡上决」在这一端不成立;durable 腿合成终帧照旧带那一句)。接入文档 **§113 P-1 / P-2**。
|
|
68
|
+
- **chrome 臂 `suspended_reopened`**(CC-276):`runStream` 对流上一帧 `suspended{reopened:<码>}` 发一条 chrome 事件(与该帧同一拍、按流序;转录面仍零行):`{ kind, laneProof, reopen, content?, eventId?, taskId? }` —— `reopen` = `suspendedReopenOf` 读数(恒 `reopened` 臂,原码开集);`content` = reopen 族单源句 `resumeReopenContent`(与决断回执那一路逐字同句;码不在 reopen 族闭集内 ⇒ 缺席);`eventId` = 帧的 durable id;`taskId` 只在帧上自带时在场(帧上 `taskId` 键原样 —— 服务端用这个键、且值按 run 唯一时它才等于 run id)。明说没重开 / 老引擎缺键 / 坏形 / 帧上的位读不出(取值器抛)⇒ 一条不发、流照走。同一条流上的重放按既有 durable 幂等只报一次。`CHROME_ARMS` +1 行(`required:false`)。接入文档 **§113 R-1**。
|
|
69
|
+
- 公面合计:根入口运行期导出 +2(CC-268:`classifyToolHistoryMismatch` / `resolveToolHistoryRewindTarget`);测试钩不变;公面类型名不变;`EmitContext` +1 可选成员(`offersRewind`)、`CrashConvergedRow` +1 可选成员(`needsHumanReason`)、`ChromeEvent` +1 臂(`suspended_reopened`)。
|
|
70
|
+
|
|
71
|
+
### Changed
|
|
72
|
+
|
|
73
|
+
🔴 **已发导出的可观察变化(逐条;按旧行为断言过的测试请按接入文档 §113a′ 改锚)**:
|
|
74
|
+
|
|
75
|
+
1. **合成终局错误行**(CC-268):错配形那一行多 `_sema_tool_history_mismatch`(宿主不改代码也会看到);宿主钉 `offersRewind: true`(或对象形,且那台引擎没明说没有对话锚)时行尾另起一行出路句、连续时再加次数句;`adapt()` 重建的行同带这只键;终端 `-p` stream-json 输出的合成行照出这只键、不加句。非错配行与结果帧 `errors[]` 与 0.86.0 逐字节同。**谁要跟**:按「行恰等于 `API Error: <原句>`」或「合成行键集恰为 …」断言的格改锚(§113a′ 第 1 / 2 条)。
|
|
76
|
+
2. **`projectCrashConverged`**(CC-269):needsHuman 行多一位 `needsHumanReason`;供给行上碰巧有同名位 ⇒ needsHuman 行被覆盖、resumeSafe 行被摘掉;分桶不变。**谁要跟**:按「needsHuman 行键集恰等于供给行键集」断言的格改锚(§113a′ 第 4 条)。
|
|
77
|
+
3. **`bridgeAskUserQuestionGates`**(CC-273):在上述收场形上吐出的同步终帧是一只**新信封**(多一位,`result` 仍是同一只对象;按对象同一性比对终帧的宿主注意);结果帧在上述收场形上多一位超集键。**谁要跟**:按 `===` 比对回吐终帧、或按「帧键集恰等于引擎那一帧」断言的格改锚(§113a′ 第 5 条)。
|
|
78
|
+
4. **`runStream`**(CC-276):有 `emitChrome` 时多发一种 chrome kind `suspended_reopened`;转录面不变。**谁要跟**:对 `ChromeEvent['kind']` / `ChromeArmKind` 做穷尽 `Record` / `switch` 的 TypeScript 端补一行(否则编译红);「`runStream` 发出的 chrome kind 恰为某集合」的判据加这一词(§113a′ 第 6 条)。按端:终端必改两处(臂登记表补一行;臂对账要求每条已知臂有在场的消费点 ⇒ `consumeChrome` 补一条零行为 `case 'suspended_reopened':`,或 `hostSink` 补一支判别);网页端必改三处(chrome 臂台账补一行;臂 parity 测试正控臂数 35 → 36;同一测试的 `SAMPLES` 补一条合成事件)。另:`suspendedReopenOf` 读 `reopened` 时取值器抛错(或入参是已撤销的 Proxy)由外抛改答 `unstated`(兑现「永不抛」;流驱动主循环也读它)。
|
|
79
|
+
5. **workflow 活动台账与监视器源的凭证位收取值函数**(CC-274):`LiveWorkflowConfig.authToken` 与 `WorkflowActivityLedgerConfig.authToken` 由 `string | { mode: 'same-origin-relay' }` 放宽为 `WireAuthTokenSource | { mode: 'same-origin-relay' }`(`WireAuthTokenSource` = 串或 `() => string | undefined`,与 `makeEngineWireClient` 同一形)。取值函数原样交给 wire client,**每发一次请求读一次**(活动流的每次开流 / 断流重连、监视器源的每次 list / get),构造期零读。放宽对**构造方** additive;**读**这两只配置位、按 `typeof c.authToken === 'string' ? … : c.authToken.mode` 二分的 TypeScript 宿主代码会编译红(TS2339)—— 要补函数那一支(运行期函数形落进「串」那一支会把函数源码当令牌拼进头里)。
|
|
80
|
+
- **连线身份改按凭证来源认**:活动台账判「同一 workflow 的新请求是不是同一条连线」时,串按值比、取值函数按**函数引用**比、同源中继声明形按 `mode` 比;形不同恒不等(串 → 取值函数算换了来源,顶掉一次,之后稳定)。比较时**从不调用**取值函数。结果:宿主持同一只取值函数时,令牌轮换不再把同一台引擎、同一 principal 的台账当成另一条连线顶掉 —— 修前宿主只能每次现读一个串塞进来,每一次轮换都让台账被换成空账、已收的运行期活动事实清零。
|
|
81
|
+
- 🔴 宿主要持**同一只**函数(建一次,各入口 spread 同一份连线对象):每次调用现造一只箭头函数 = 每次都是另一连线身份,台账会被顶掉,监视器源随之停手退回只读快照轮询(接入文档 §113a′ 第 7 条)。
|
|
82
|
+
- 串形 / 中继形:判法与 0.86.0 逐格同(串的值就是它的来源)。
|
|
83
|
+
- `authToken` 缺席(`undefined`,JS 宿主漏传):同一 workflow 第二次 `ensureWorkflowActivityLedger` 由抛 `TypeError` 变为复用同一台账(两次来源相同);`null` 两版都抛。
|
|
84
|
+
6. **detach 取消兜底的 arm 口收取值函数**(CC-275):`armDetachCancel` 入参的 `authToken` 另收 `() => string | undefined`。arm 时零读;取件口 `detachCancelArm()` **每调一次读一次**,读到的值过与 wire client 同一只三态解析(`resolveWireAuth`)后,仍以 `string | { mode: 'loopback-unauthed' }` 两形之一交出 —— 宿主发射腿(按「是不是串」分两支)的读法一字不变,发送前同步取件即拿到发送那一刻的凭证。`DetachCancelArm` 型本身不变(它是取件形)。
|
|
85
|
+
- 取不到(`undefined` / 空串 / 非串 / 抛错)⇒ 回环地址答 `{ mode: 'loopback-unauthed' }`(发射腿零 Authorization 头,绝不伪造),非回环答 fail-closed 匿名身份;取值函数抛错时取件口**不抛**(它跑在信号处理路径上)。
|
|
86
|
+
- 串 / 回环免鉴权声明形:取件口照旧给回 arm 进来的**同一个对象**,与 0.86.0 逐字节同。取值函数形每次取件给一个新对象:原型与 arm 相同(类实例 arm 原型上的 `getTaskId` 照样可达),自有位按属性描述符整份拷(含不可枚举的自有位;`getTaskId` 等同一引用);副本不是同一实例,方法读类的 `#private` 字段时在副本上仍抛(KL-316)。
|
|
87
|
+
7. 内部搬家(公面零变化;CC-276):`ResumeReopenDetail` / `resumeReopenFromError` / `resumeReopenContent` 的定义从 `resumeRefusalCopy` 模块逐字搬到 `suspendedReopen` 模块(为 chrome 臂与决断回执两路共用一句,且不把前者的依赖拖进流驱动);原模块留同一绑定的转口(已发文档写的坐标照旧找得到),包根经两条星号导出拿到的是同一个函数对象,公面名单与个数不变。只走包根入口的端零感知。
|
|
88
|
+
|
|
89
|
+
### Gates
|
|
90
|
+
|
|
91
|
+
- 新门 `scripts/run-tool-history-mismatch-projection-test.mjs`(212 格,带两代服务端夹具;不带时 204 格、夹具腿标为未跑;CC-268,包侧投影门):S0 两只口在根入口上、与实现模块同一函数对象、在公面基线里,账本 / 两句文案 / 锚表 / 测试钩不上入口;D1 判型 —— 三句 provider 模板单源锚表(冻结、三行三 kind),真端点 / 用户报案 / 兼容端点 / 无反引号 / 原始与截断 JSON 体都认,孤儿结果 / 重复 id(不带 id ⇒ 键缺席)也认;K5 401 / 404 / 429 / 模型名无效 / 无任何 HTTP 400 证据 / 治理码 ⇒ `null`;K6 句中引用 / JSON 消息以别的话开头 / 半句 ⇒ `null`;会抛的取值器不抛;id 窄读(控制字符 / 超 128 字 ⇒ 缺席);用开发依赖引擎包的真铸终局喂入。D2 `runStream` 终局行:错配形挂键、宿主声明 `offersRewind` 才在行尾另起一行补句(缺省 / `false` / `-p` / 工具车道不补)、`errors[0]` 原文;401 / 模型名无效行逐字节同旧、无键;治理拒绝理由复述 / 模型输出问题行 / 模型正文逐字复述 ⇒ 无键无句;`adapt()` 重建行接力同键(坏 kind 整键丢、坏成员逐个丢、模型行不接力)。D5 会话级连续同指纹:不同 run 第 2 / 3 次 ⇒ 次数句、`consecutive` 2 / 3;同一 run 重看不加码;回放帧与无 run id 帧不记账也不打断;成功一轮清零;换指纹从 1 起;别的失败打断;别的会话各记各的;键是引擎回显的会话 id;开流代际(先开后收的老流不改账);有界 64 只会话、挤出后从 1 起。D3 节点解析:坏轮 prompt 行本身(本地行跳过)/ 会话第一轮 ⇒ 从头开始 / 结果晚到取最早提到 id 的那一轮 / 判不了六理由闭集(prompt 无句柄不越级)/ 敌意行不抛。D1 另钉三件:JSON 回体只认错误信封(请求回显 / 兄弟对象 / 数组成员 / 更深一层 / 体是数组 / 顶层 `message` 非串 / 回体前有别的文字 ⇒ `null`;真信封 / 顶层 `message` / 转义引号在前 / 截在首只 id 里 / 闭合值里最后一只 / SDK `400 {…}` 形照认);id 按整只读(带 `/` / 后接 `…` / 服务端截断标记 / 顶在末尾 ⇒ 缺席,逗号 / 句点 / 分号 / 点号冒号形照读,带 `/` 的那一只交节点解析 ⇒ `no-tool-use-id`);孤儿结果 `blocks` / `block` / 直接冒号三形都认。D2x `offersRewind` 对象形:引擎明说没有对话锚 ⇒ 两句都不加、键照铸(含 `consecutive`);引擎说有 / 没观测过 / 读不出 / 读数失效后 / 地址不是串 / 取值器抛 ⇒ 照加;`true` 不查判据;print 车道不加;非错配行不变;类型面对象形可赋值、老形照过。D5x 连续账本:同一会话 17 次后重看第 1 次不报次数、第 17 次仍 17、下一次 18;两次错配之间夹不带 run id 的别的失败 / 结局读不出的 done / 不带 run id 的另一只错配 / 回放的成功 / 409 会话占锁拒绝 ⇒ 清零;同指纹的无 run id 帧与回放帧不打断;成功清零后重看上一段的 run 不开新段;换指纹再换回、重看第一段的 run 不加码。D1 另钉 SDK 错误对象的非 JSON 形 `400 <原话>`(认、id 取首只;重复 id 那一句认、不带 id;`400 ` 后面是别的话 / 句中才引用 / 两个状态词 / 状态 429 ⇒ `null`)。D3 另钉敌意入参不抛(`mismatch.kind` / `toolUseId` 取值器抛、`mismatch` 是已撤销的 Proxy ⇒ `no-mismatch`;`rows` 是已撤销的 Proxy ⇒ `unreadable-rows`)。D5y 代际水位:同指纹无 run id 的较新终局与较新流重看认得的 run 都把老流挡在外面(两条交错序列逐项读数 + 两条真开流的 `runStream` 交错);同 run 接管重放照答原次数、水位不回退。D5z 会话键无从归属:宿主每轮现铸会话 id、两次错配之间夹失败事件帧 / 体无会话 id 的 409 占锁拒绝 ⇒ 第三次无次数;宿主钉的 id 等于引擎回显 id 时夹 429 清零 / 同指纹无 run id 不打断 / 成功清零三形逐字同修前;取舍格(另一只从没被回显过的会话里的一帧失败 ⇒ 这一会话下一次从 1 起,少报);段被打断后重看最近那只 run 不报次数。H 出路句与 CC 同句逐字、次数句过卫生禁表且零机器码。X 夹具腿(服务端 7.106.0 / 7.104.0):夹具自带的引擎包铸终局 → 夹具服务端上 wire 前洗消 → 本包:键与出路句在、run id 原样上 wire ⇒ 第二次加码、401 无键。
|
|
92
|
+
- 新门 `scripts/run-park-failed-reason-projection-test.mjs`(55 格):P1 同步腿 + 409 `resume_blocked_by_policy` 折叠码 ⇒ 恰一只终帧、帧上带句;P2 含折叠句与出路、与 durable 腿合成终帧 `errorMessage` 逐字相等;P3 原样吐(`result` 同一只对象、调用方那一帧不被就地改写);P4 结果帧透传、`errors[]` / `subtype` / `is_error` 与未带键的同一帧投影逐字同;P5 缺席三形;P6 触顶收场两腿;P7 durable 腿不变;P8 宿主撤卡收场同样带句;P9 凭据洗消 —— 决断失败的原话里嵌着带 userinfo 与 `?token=` 的 URL,409 折叠码 / 422 续跑起不来 / 403 三形:帧上那一位与 durable 腿合成终帧同句(原句)、结果信封上这只键无明文且与 durable 腿结果信封 `errors[0]` 逐字相等、其余位与不带键的同一帧投影逐字同;P5′ 没有决断面的收场 —— 宿主没装审批卡口 / 卡口装在别的会话键上 / 问答腿没有 overlay ⇒ 同步终帧与结果帧都不铸、`errors[0]` 不变、durable 腿合成终帧照旧带句,宿主装了卡口而卡口自己报失败 ⇒ 照铸。
|
|
93
|
+
- 扩门 `scripts/run-crash-converged-projection-test.mjs`(118 → 138 格):G8 四因由各一形 + 优先序 + resumeSafe 行不带 + 供给同名位覆盖 / 摘除 + 32 格全组合分桶与 §12c 五项合取逐格一致 + 值形(自有可枚举数据位)+ 旧词 `crashed_before_park` 带因由 / 新词 `shutdown_before_park` 今天仍整行 dropped + 混合表账不混;G7a 编译语料加两行(因由位可赋闭集四词、不带它的老字面量照样可赋值),G7c 表外词赋值恰为 TS2322。
|
|
94
|
+
- 扩门 `scripts/run-suspended-reopen-projection-test.mjs`(15 → 55 格):C 段形(臂表登记、载荷键集、三码单源句、非重开零事件、开集码不带句、`taskId` 只取帧上、车道证明逐条新建);E 段端到端序列:重放(同 id 一次、第二次按既有幂等留 `duplicate_seq` 痕 / 不同 id 各一)· 接管(从第一帧重放:报在流序里自己的位置,晚于之前那一帧的转录、先于之后的正文与终局;前台腿与接管腿幂等键相同)· 迟订阅(游标在重开之后 ⇒ 零事件;停在重开上才接 ⇒ 照报)· 宿主没接 / sink 同步抛 / 返回被拒 Promise ⇒ 转录逐形不变、零未处理拒绝 · 从 HITL 桥一路到 chrome 回调;另钉帧上的位读不出(`reopened` / `taskId` 取值器抛、已撤销的 Proxy)⇒ 读口答 `unstated`、零事件、流不断、转录与不带 `reopened` 的同一条流逐形相同、零未处理拒绝。
|
|
95
|
+
- 扩门 `scripts/run-wire-auth-source-test.mjs`(30 → 83 格)。G8t / G9t:类型面 additive,真跑 tsc(取值函数形可赋值;老形 —— 串 / 中继 / 回环免鉴权 / 老宿主手里的 `DetachCancelArm` 变量 —— 照过;取件口仍只收窄成两形,`arm.authToken.mode` 那一支照编得过;非法形负控恰两条)。G8(CC-274):同一只取值函数两次预热之间轮换 ⇒ 不开第二条流、台账帧数与已收 beat 原样、未收口;断流重连与监视器源 GET 每发带当下的值;取值函数读取次数 = 出站请求数(身份比较零读);负控:换另一只函数(同值)⇒ 按新来源另开一条流、旧账被顶掉、持旧函数的监视器源停手不抢回;串按值 / 中继按 mode 与 0.86.0 逐格同、串 → 同值取值函数与中继 → 取值函数都算换来源。G9(CC-275):arm 期零读、台账读口(`isDetachArmed` / `detachedTaskId` / 400 分诊)零读;arm 后换值 ⇒ 取件口给取件那一刻的值且恰读一次;宿主发射腿冻结副本一字未改,发出去的取消请求头上是发送那一刻的值、凭证只在 Authorization 头;三态(回环 `undefined` / 空串 / 非串 ⇒ 免鉴权声明形且零 Authorization 头;非回环 ⇒ 匿名身份);取值函数抛错取件口不抛;串 / 回环声明形给回同一个对象;类实例 arm(`getTaskId` 在原型上)取值函数形取件后方法可达、宿主发射腿照发,串形类实例仍给回同一个对象。
|
|
96
|
+
- 登记物:根公面基线 1354 → 1356(+2);测试钩 62 不变;超集台账 `docs/type-superset.json` 95 → 97 条(`_sema_tool_history_mismatch` / `_sema_park_failed_reason`);门清单 `scripts/gates-manifest.json` 165 → 167 条(两道新门;崩溃收敛 / 重开挂起 / 凭证取值三道扩门的说明同批更新),README「Guards」与负控文档派生表同批重生;棘轮 `portability.kernel` 20 → 22、`portability.adapt` 36 → 37、`portability.index` 228 → 229(registry.json 追四条账:内核闭包新边 `src/hitl/suspendedReopen.ts`〔CC-276,值级 import 只有已在内核里的 `engineErrorCodes`〕与零 import 叶子 `src/toolHistoryMismatch.ts`〔CC-268,经流驱动与 `adapt()` 进内核 / A 层 / 根入口闭包〕);单例清单 561 → 573 条,高风险上限 133 不变。
|
|
97
|
+
|
|
98
|
+
### Known limits(本版新增)
|
|
99
|
+
|
|
100
|
+
- 会话历史错配(CC-268):判型只认三句 Anthropic 形 provider 模板,OpenAI 兼容车道的同病不出键、不出出路句;JSON 回体只认从开头起的错误信封(宿主直接交的 SDK 错误对象形 `400 {…}` / `400 <原话>` 同认;KL-308);`toolUseId` 读不出的几形(字符集外、超过 128 字、失败句被截在 id 中间)键里不带 id、节点解析随之判不了(KL-309);连续次数账本在进程内、按会话、上界 64 只会话,重启 / 挤出后从 1 重数;宿主开流时钉的会话 id 从没被引擎回显过时,没有回显会话 id 的终局(失败事件帧、409 拒绝等)认不出属于哪一只会话,本进程里每一只会话正在数的连续都按被打断处理 —— 少报,不多报(KL-310);没有 run id 的终态、回放终态与结局读不出的终态不记次数(与这一次错配不同时清零),这些形上只有出路句、没有次数句;段外只认得最近 16 只 run,更早的 run 被重看会被当成新的一次(KL-311);重复 id 那一句模板不带 id,节点解析判不了,次数的指纹只剩 kind(KL-312);节点解析信宿主交进来的行序(KL-313);孤儿结果 / 重复 id 两句的报文全形是照 provider 错误体形构造的,只有锚串有出处(KL-314);连续账本是可写的会话级模块单例,同一进程里有两份本包实例时各记各的 —— 各从 1 数起(少加一句次数),另一份经过的别的结局这一份看不见、不会清零(可能多报一次「连续」)(KL-315)。
|
|
101
|
+
- 崩溃收敛 / 重开挂起(CC-269 / CC-276):`suspended_reopened.taskId` 今天恒缺席 —— 服务端的重开挂起帧不带 `taskId` 键,本包不从同一条流的首帧代用;将来在场时是帧上 `taskId` 键原样,只有服务端用这个键、且值按 run 唯一时它才等于 run id(KL-305);接管 / 从头重放的腿上本臂按流序报每一次重开(含已成历史的那次),「此刻还成立吗 / 还属于这块屏」由宿主判(KL-306);`decided` ∧ `resumeSafe:true` 判 `contradictory`,不判 `approved_then_interrupted`(KL-307)。
|
|
102
|
+
- 取值函数形的连线身份按函数引用认:宿主每次现造一只函数会让活动台账每次都被顶掉、监视器源停手退回快照轮询。按值比会重新引入本版要去掉的形(令牌轮换即清账),按形比会让同址同 principal 的两只不同凭证来源共用一条台账 —— 两害取轻,口径成文(KL-303)。
|
|
103
|
+
- 监视器源在 401 之后的 60 s 长退避不变:取值函数形下,若某一发恰好落在引擎换代、新令牌还没写回的窗口里答了 401,即使取值函数随后已给出新值,下一次重试仍要等到长退避到点(KL-304)。
|
|
104
|
+
- 登记物按既定裁定改窗(本版对这些物零改动):同名影子对账门豁免表对终端 1.0.139 集成分支跟表后剩下的两行(重开口外层 / 回执成因词表),期限 0.87.0 → 0.88.0;装配 / 宿主扩展类十行对终端 1.0.139 集成分支重看判词不变,重看期 → 0.89.0;过渡物退役登记 RL-11 … RL-14 到期重看 —— 上游契约那一节与同名常量仍未出、本包支持的引擎 / 服务端底线仍未越过 —— 重看期 0.87.0 → 0.89.0(本版距上一版一天,重看期实质未过)。
|
|
105
|
+
- 完整台账见接入文档 §113 末行「包侧缺口」。
|
|
106
|
+
|
|
52
107
|
## 0.86.0(2026-10-01)
|
|
53
108
|
|
|
54
109
|
> 主题:**minor** —— 八件同发,另有开发依赖两步换钉。① **引擎 7.34.0 提货,两代读法**(CC-249 / CC-245 / CC-230 第二批;老引擎照旧、新引擎按新形,缺席读作老引擎、不折成否定):引擎包入口导出的六张事实表与本批碰到的三张闭集改为构建期生成;门处置「谁拒的」改为两代并集(新引擎的 `mode` 有了专句,老引擎仍发的 `plan_mode` 照认);门记录上权限模式两臂进视图并给具名读口 `gateModeRelease`;通告码册 +6 且各有事实读器;读站判不出读什么时的专属卡 `read_unestablished` 进结构化卡型全集;本人规则店读不出在新引擎上改为「跳过 + 一条通告」,老引擎那张卡照读;粗 shell 档那一句改成两代都成立的说法。② **服务端 7.105.0 提货**(CC-251 / CC-252):会话规则写口拒「写进去就跑不了」的工具名(`400 rules.legacy_tool_name`)投成三只写面的具名结局;配了服务凭据的部署上,门前只读路(workflow 读路、fleet 流)的 `401 auth.unauthorized` 投成具名态与一句话。③ **请求键 `approverPosture` 退役**(CC-253):引擎按名拒收、服务端 7.106.0 起整条 400,本包停发;宿主仍给 ⇒ 构造期按名响亮拒,不改发 `permissionMode`。④ **委派判不出通告带补救话**(CC-254):`delegation.ask_unresolvable` 视图 +可选 `remedy`,原样透传。⑤ **开发依赖换钉引擎 `~7.35.0` 再到 `~7.36.0`**(CC-255 / CC-261):生成物重生成、既有表一个值都没变;引擎工具声明门的九个拒码登记进配置拒绝识别表(`CONFIG_REFUSAL_CODES` 8 → 17,只登记不分派);通告码册再 +1 `route.tool_call_id_folded`(受众运维面,只登记)。⑥ **`adapt()` 在错误信封上不再回落读旧名 `result`**(CC-221):0.83.0 两名并读的收口;现役流零变化,只有宿主把 0.83.0 之前的原始输出落盘再喂回 `adapt()` 那一形看得见。⑦ **服务端 7.106.0 提货**(CC-250 / CC-256 / CC-257 / CC-262 / CC-263 / CC-265):caps `store` / `centerWiring` 两只窄读口、SQL 姿态两代读、计费入口 503 `fleet.lease_unavailable` 的读口与一句话;caps `executionLane.isolated` 与「本机无 OS 沙箱」判定;请求键 `inheritEnv` 的构造、版本闸片段口与两只拒按码出句;权限规则表超帽 / 坏项的机读读口与一句;工具声明拒按码认领「不是暂时故障」;请求体键形拒的两张键表读口;强制卡值变零改动实证。⑧ **`-p` 首帧带出工具清单的来源**(CC-266 / CC-267):`withPrintInitFrame` 放出的首帧多两只超集键;三张估计工具表改为常驻(零行为)。根公面运行期导出 1310 → 1354(+46 −2);测试钩 60 → 62;公面类型 891 → 922(+31);超集键 +2(`_sema_tools_source` / `_sema_tools_fallback_reason`);通告码册 73 → 80;零新投影臂(既有 `tool_end` 臂上的门记录视图多两组可选位);开发依赖引擎 `~7.33.1` → `~7.36.0`。peer sdk 地板 `>=12.0.1` 不动 —— sdk 14 的三处型面 BREAKING(`TaskRequest.approverPosture` 删、`Capabilities.taskApproverPosture` 删、顶层 `Capabilities.sql` 搬进 `store.sql`)本包已按结构读覆盖:本版全部 `.d.ts` 在 sdk 12.0.1 / 13.1.0 / 14.0.0 / 15.0.0 下严格编译(不跳过库检查)都零错误,消费方先升 sdk 14 或先升本包都不编译红;服务端 7.105.0 / 7.106.0 与 sdk 13 / 14 的新形一律按码 / 按结构读,不 import 只有新版 sdk 才有的型 / 值。🔴 **型面 BREAKING 三处**(`TaskRequestInput.approverPosture` 删;两个运行期名字退出根入口,见 Removed;`SemaPermissionDenial.deniedBy` 闭集 +`mode`:型由 SDK `DeniedBy` 换成 `GateDeniedByWord`,对它穷尽 `switch` 的端编译红,见 Changed 第 4 条);另有**已发导出的可观察变化**在 Changed 逐条单列(第 1–34 条与第 36 条,写明谁要跟;第 35 条是零行为的重分类);对下列联合做穷尽 `switch` 的端编译期红:`EngineNoticeFactsCode`(+6 员)、`SessionPolicyFailure` / `SessionPolicyTightenOutcome` / `SessionPolicyRemovalOutcome` / `WorkflowResolution`(各 +1 臂)、`BackgroundSourceHealth`(+1 词)、`SemaPermissionDenial` 的 `deniedBy`(+`mode`)。
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.
|
|
38
|
+
**Version:** 0.87.0
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -317,7 +317,7 @@ guard still cross-checks the table by name).
|
|
|
317
317
|
| `scripts/run-absence-fold-census-test.mjs` | A package-wide census of the "absence folded into a positive outcome" defect shape, so that fixing the six sites this release does not merely move the shape somewhere else. The defect is defined by position, not syntax: a fallback position (the unconditional tail return, the `default:` arm, the literal minted when there is nothing to pass on, the value returned from an error path) may only say `unknown` or stay absent, never a positive word. Detection walks the syntax tree of every source file, so comments, strings and multi-line spellings cannot hide or fake a hit, and covers five forms: the right arm of `??` / `\|\|`, the else arm of a ternary, the first return of an explicit `default:`, a `catch` block or `.catch(() => …)` arrow returning a healthy value, and a function whose last statement returns a positive word after other returns. Every remaining hit must be registered with a written reason, an unregistered hit fails the gate naming the file and line, the registered count must equal the real count so a cleared site cannot leave a spare allowance behind, and the gate proves its own teeth behind a fence (a failed self-proof refuses to report any count): each form injected into an in-memory copy must add exactly one hit, two correct spellings are pinned as non-hits, and samples inside comments or strings do not count. It also pins the headline site: the fleet panel projection no longer mints an `end` with `isError: false` on absence |
|
|
318
318
|
| `scripts/run-device-executor-management-capability-test.mjs` | The engine's device-management self-description (`capabilities.deviceExecutor.management`, engine ≥7.88.0), read the same four-state way as its five sibling capability readers: an absent `management` key is reported as not reported (never folded into `false`; an older engine really ships the lane object without it), an absent `deviceExecutor` key is likewise not reported, `deviceExecutor: false` is the lane being absent, presence is judged by own-property not truthiness, the value must be a strict boolean, the tee never throws and drops stale generations, and the package owns the verdict on whether the `/v1/devices` management verbs are usable (`yes` only when present and true, `no` when present-false or lane-absent, otherwise `unknown`) |
|
|
319
319
|
| `scripts/run-run-cancel-context-test.mjs` | The run record's `cancelContext` side-note (engine ≥7.87.3) read structurally, and the cause of a `turn_aborted{engine_error}` classified from machine-readable evidence only: `cancelled` (code `cancelled`, with the cancel-time context when present) / `engine_error` (any other failure code, passed through verbatim) / `run_still_live` (the record is not terminal — a dropped stream is a client-side fact, not the run's cause) / `unknown` (never guessed). An absent `cancelContext` reads as *not reported*, never as "not cancelled"; `elapsedMs` is never folded to 0. |
|
|
320
|
-
| `scripts/run-suspended-reopen-projection-test.mjs` | The durable `suspended` event's `reopened` key read as three distinct states — `reopened` (with the engine's code, verbatim), `not_reopened` (an explicit `null`), `unstated` (key absent or unreadable) — and carried on the HITL bridge's active gate (`currentGateReopen()`), re-read on every `suspended` and cleared with the gate. |
|
|
320
|
+
| `scripts/run-suspended-reopen-projection-test.mjs` | The durable `suspended` event's `reopened` key read as three distinct states — `reopened` (with the engine's code, verbatim), `not_reopened` (an explicit `null`), `unstated` (key absent or unreadable) — and carried on the HITL bridge's active gate (`currentGateReopen()`), re-read on every `suspended` and cleared with the gate. The stream driver also turns a `suspended` frame that reads `reopened` into the chrome arm `suspended_reopened` (the reading, the single-source reopen sentence when the code is in the reopen family, the frame's event id, and the run id only when the frame itself carries one), pinned end to end: one notice per event id on a stream, emitted in stream order on a replay from the first frame, nothing for a cursor that starts after it, and an unchanged transcript when the host has no chrome sink or the sink fails. |
|
|
321
321
|
| `scripts/run-panel-identity-normalization-test.mjs` | One background subagent has two ids on the wire — the fleet row id tail and the `task_progress` task id (its transcript id). Every panel event goes through one funnel that rewrites the `tick` / `end` task id onto the fleet row's id once a `fleet-row` has registered the key (`transcriptId` first, `parentToolCallId` as the fallback), carrying the original as `wireTaskId` and marking `taskIdOrigin`; an unbound tick whose row has not arrived yet waits one beat (bounded) and is released verbatim on the next tick / `end`, when the buffer is full, or after `MAX_HELD_WIRE_TICK_BEATS` other fleet-row / end / sweep events (a `sweep` itself leaves it alone: there is no row to settle yet); a normalized `end` that carries no cycle identity borrows the registering row's, so a late close of a revived task is recognized as stale; the key table is an LRU (a task that keeps ticking is never evicted by newer registrations); the residency mark migrates with the id and both keys are cleared on settle — except that a stale (previous-cycle) terminal never clears the revived row's mark — so the notification lane can clear it. Once a tick has been delivered verbatim under its UUID, that UUID is the subagent's key: later fleet rows and fleet-side ends are rewritten onto it (the tail kept in `wireTaskId`), so a consumer sees one row in every arrival order; a late tick from a previous cycle is dropped rather than folded into the revived row. |
|
|
322
322
|
| `scripts/run-prompt-assembled-projection-test.mjs` | The `prompt_assembled` frame (one prepare's prompt-assembly manifest) projected to an internal arm and then to the additive `prompt_assembled` chrome event — the per-section / per-block **character** counts, the mounted tool names and `totalChars`, each key present only when the engine really sent it (the frame's `constitution` is deliberately not carried: no consumer asks for it today, and every published key is a contract to keep). The manifest carries **no token counts** anywhere upstream, so this projection mints none: a token figure derived from characters would be an invented number, and the engine's own estimate lives on `context_usage.sections[].tokens` (same id wordlist, joinable). Bad rows are dropped one by one, and a face that loses every row reads as an absent key rather than an empty array — so an absent face means only "this event carries no readable view of it" (an absent upstream key, an empty array and a fully filtered list all land on the same shape) and is never reported as a diagnosis about the engine. `blocks[].id` and `sections[].id` are two different wordlists with a many-to-one relation, and the token join against `context_usage.sections[].tokens` only holds when both sides carry a section view. A frame with no readable composition key at all is malformed, ids and slots are read as an open set, one chrome event per frame with zero transcript rows, several prepares per task are all handed over (de-duplication — "take the last one" — is the host's move), and the lane is told honestly (`parentToolCallId` ⇒ subagent lane; a frame attributable only by `sourceTaskId` / `bgAgentId` is not surfaced on the main lane). Both entry points obey the same rule: the adapt layer rebuilds every row too, so a host pipeline (or a replayed transcript) that feeds the raw frame straight into `adapt()` cannot smuggle extra keys (`tokens`, digests, aliases), a negative `chars` or a `null` row into the chrome payload, an empty array does not count as a composition face, the identity keys are snapshotted once on both paths (read exactly once each, a throwing accessor rejects the whole frame — reading one twice is what lets an accessor frame land on a different lane on each path), and the two paths are compared verbatim so the two readers cannot drift. |
|
|
323
323
|
| `scripts/run-compaction-outcome-projection-test.mjs` | The `compaction_outcome` frame (a compaction that did **not** end as compacted: mooted by the task ending, failed, …) projected to an internal arm and then to the additive `compaction_outcome` chrome event — `outcome` required and verbatim (open set), `trigger` / `reason` present only when the engine sent a non-empty string, malformed frames dropped, zero transcript rows, the lane told honestly (`parentToolCallId` ⇒ subagent lane; a frame attributable only by `sourceTaskId` / `bgAgentId` is not surfaced on the main lane). |
|
|
@@ -338,7 +338,7 @@ guard still cross-checks the table by name).
|
|
|
338
338
|
| `scripts/run-display-body-test.mjs` | The engine wraps text it hands a model in a fence — an opening marker naming the payload, the payload itself, and a closing marker — so the model reads it as data and not as instructions. That fence is minted and read in one place here, which makes stripping it for a human reader this package's job rather than each shell's: a shell that renders the envelope verbatim is showing a person a defence that was written for a model. The reader answers with a discriminated union — fenced, with the label and the payload, or not fenced, with the text as it came in — and it reaches that answer through the **same** matcher the mint side registers, never a second copy of it; the guard proves that by walking the syntax tree of every source file and requiring exactly one literal carrying the marker text, and by requiring the reader's own body to contain no matcher of its own. Eighteen shapes are run through both entry points and required to agree line for line. Anything the package does not recognise — a near-miss in the wording, a hyphen where the marker has a dash, a different case, an opening marker with no close, a close before an open, a truncated close, or any non-whitespace byte outside the pair — comes back unfenced with the input returned **verbatim**: no guessing, no trimming, no repair, because a half-stripped envelope puts a sentence on screen that nobody wrote. Only the outermost layer is removed, so a nested fence, or one forged inside the payload, survives byte-for-byte in the body — those bytes are part of what the engine said, not part of this protocol. Nothing else is washed: control characters, leading and trailing whitespace and a twenty-thousand-character payload all pass through untouched, and so does the label, because sanitising and length-capping belong to the mint point that puts a string on a screen and a passage of text must not have two launderers. A value that is not text is answered with **nothing at all** rather than with an empty payload: the reader never stringifies it, never calls its `toString`, and never emits `[object Object]`, and it does not hand back a body of zero length either — an empty payload is a real reading (a fence can legitimately wrap nothing, and an empty string is an empty string), so folding "there was no readable text" into it would leave a caller unable to show a degraded line at all. Those three stay apart: no text yields nothing, an empty string yields an unfenced empty payload, and an empty fenced payload yields a fenced one with its label. The `fenced` discriminator is always present on a reading, and the label key exists only on the fenced arm, so a missing label is never rendered as an empty one. One shape needed more than the whole-string match this started with. When the engine reports back from a delegated run, the fence is only **one section** of the report: ahead of it sit a frame header, a handful of optional field lines and a section label, behind it a closing instruction addressed to the model, an optional internal identifier and a usage block. A matcher anchored to both ends of the input answers *not fenced* on that, and the whole scaffold - written for a model - goes on screen. The reader therefore also locates the fence **inside** a recognised report frame, using four anchors that are always present and always byte-for-byte fixed, and hands the located slice back to the same single matcher rather than a second one; the guard assembles its corpus from the engine package actually installed (the fence from that package's own constructor, the frame lines read structurally out of the minting file and then compared byte-for-byte with what this package registers), so a rewording or a reordering upstream turns the guard red the day it lands. Three readings are pinned one cell each: a fenced result section yields the payload the delegate actually wrote; a partial-findings section yields that text and says which of the two it is, with the prefix line kept out of the payload; and a section the engine filled with its own no-text sentinel yields an empty payload with a reason, which stays distinguishable from an input that was simply an empty string. Whatever surrounded the fence is returned alongside rather than dropped - the bytes before it, the slice itself and the bytes after it reassemble into the input exactly - and the payload never contains a line of the frame. The criterion deliberately does **not** enumerate the lines outside the fence: several of those field lines carry interpolated untrusted text and cannot be told apart from prose, so requiring every one of them to be recognised would mean that a single new field line upstream sends every failed report back to being unreadable, and a failed report is exactly when a person most needs to read what the delegate said. Fourteen negative shapes hold the line against the easy widening, strip anything that looks like a fence: a fence sitting in ordinary text, a missing frame header, a different sentence where the closing instruction belongs, a report with no closing instruction at all while the fence sits at the very end, a header and label in the wrong order, mismatched open and close tags, an open with no close, a bare unfenced section, a stray line between the label and the fence or between the fence and the closing instruction, an entirely absent section, a no-text sentinel with another line after it, a partial-findings prefix followed by something that is not a fence, and the whole report in carriage-return line endings all come back unfenced with the input verbatim and no frame at all. Above all, **provenance is not in the text**: locating a fence inside a frame happens only when the caller states where the bytes came from, because four anchors can only recognise a shape and never prove an origin. The default reading is byte-for-byte what it was before, so a passage of ordinary prose that happens to quote a report - with real warnings on either side of the quoted part - is returned untouched and those warnings stay on screen; a caller that does state the origin gets the located reading, and even then every surrounding byte comes back alongside. Twenty-three malformed origin values fall back to the narrower default without throwing: a near-miss in case, a camel-cased spelling, the right word padded with spaces or tabs or a newline or a zero-width character, a string wrapper object, an object whose `toString` or `valueOf` reports the right word, and an object whose converters both throw. The guard compares the caller’s argument for **exact equality** and nothing else — no trimming, no stringifying, no calling the value’s own converters, because that would let a value of unknown provenance choose its own lane — and a source-level cell requires that the argument reach the comparison unreassigned and unnormalised, with its own two-way check that those patterns speak. The two accepted words are read off the published type rather than copied into the guard; adding a third word later is a type-compatibility change for any caller that switches exhaustively on them. Each of the three anchors must also be **unique** in the text, and ambiguity means the reader declines. The reason is not hypothetical: the frame header interpolates the task's own description verbatim, and the engine only requires that description to be a string, so it can carry newlines and a complete set of protocol lines. Any rule that picks one candidate out of several can therefore be made to pick the planted one, hiding the real result among the surrounding bytes - a guard cell reproduces exactly that, with a planted section and a real one, and requires the reader to decline and the real text to stay on screen. Two reports back to back are the same ambiguity and are declined the same way, with both payloads left visible; a delegate that quotes any one of the three anchor lines inside its own answer also falls back to the input verbatim, which is the registered cost of the rule, and a discrimination cell shows the same corpus reads cleanly once the quoted line is gone. The two legs are not the same shape either: the forked one ends at its closing instruction with no trailing bytes at all, and that real shape has its own cell. The witness arm reads each leg only inside its own array of lines, decodes every extracted literal to its **runtime** value rather than trusting the spelling in the source, and fails loudly if it cannot - an escape rewrite upstream leaves the runtime label unchanged while the spelling diverges, and since that same extracted value builds the corpus and serves as the expectation, trusting the spelling would close a self-proving loop. The two payload labels are therefore also pinned in the guard and compared against what was extracted, so an equivalent rewrite stays green while a real rename turns red the day it lands. A delegate that quotes the frame lines inside its own answer does not move the location, and those quoted lines survive in the payload byte-for-byte |
|
|
339
339
|
| `scripts/run-display-cap-order-test.mjs` | The order in which untrusted text is sanitised and length-capped, across every mint point that puts an engine- or database-supplied string on a screen. The sanitiser rewrites each invisible character as a six-character escape, so capping the **raw** string first and escaping afterwards hands the screen six times the width that was budgeted — a forty-character allowance becomes two hundred and forty. The guard does not hardcode that allowance, because each mint point wraps its field in different fixed prose and the prose moves: it anchors on the deciding quantity instead, feeding one benign and one control-character input of the same length through the same mint and requiring the second not to come out longer. That criterion is immune to wording changes and stays sensitive to the expansion, and it is `<=` rather than `==` on purpose — a correct escape-then-cap backs the cut off a partially-consumed escape token, so the control-character line is legitimately the shorter of the two, and demanding equality would score that avoidance as a regression. Each mint is bracketed by two positive controls (the input really reaches the screen; the cap really engages) and the expansion predicate is shown to turn red against a deliberately cap-then-escape reference, so an all-green run cannot mean the guard simply measured nothing. The shared mint point is checked directly for the two avoidances it owes — never splitting an escape token in half, which would leave something on screen that looks like the beginning of a complete answer, and never splitting a legal surrogate pair, which would manufacture the very lone surrogate the sanitiser exists to catch |
|
|
340
340
|
| `scripts/run-seat-task-request-origin-test.mjs` | Where every field of the seat lane's send-message payload comes from, and whether it actually lands anywhere. The seat payload is a closed interface this package mints itself, and most of its fields are meant to ride verbatim onto the engine's request body — two facts nothing used to connect, so both directions could drift in silence. A seat field could be named after a request position that does not exist, in which case a client writes to it, the wire carries it, the engine ignores the whole key, and the screen shows a switch that does nothing; conversely a new request position could arrive with no seat to sit in, which is **structural** absence — the closed set *is* the carrier, so a decision missing from it has nowhere to be put at all, the same shape logged when the effort dial had no seat. The guard turns each field's origin into data: either it names the request position it forwards to, or it is declared seat-local with a written reason, and the two are mutually exclusive. Forwarding claims are then checked against the **installed** SDK's type declarations, parsed rather than restated — a hand-copied list of position names would only ever prove that two transcriptions agree. The parser is held to reading top-level positions only, since a nested option object's inner keys would otherwise be mistaken for positions of the request itself, and it proves that discrimination on synthetic input before any verdict is given. The two subagent fields carry a standing regression pin, and the retention window's inner keys are read from the declaration the same way, so a seat that offers a tunable window cannot offer one the wire has no room for |
|
|
341
|
-
| `scripts/run-wire-auth-source-test.mjs` | **When** the outbound credential is read. A literal string is consumed at construction — the transport captures it in a closure and every later request reuses that one copy — so once the engine is replaced by another session and the credential rotates, a long-lived client keeps presenting the old one and the only way out is to rebuild the client along with everything hanging off it. The credential position now also accepts a getter that is called **once per outbound request**. The guard anchors on the deciding quantity, which is not "was the getter called" — reading once at construction and reusing the result would satisfy that too, and is exactly the shape being removed — but *which read produced the value on the wire*: it changes the getter's answer between two requests through the same client and requires the second request to carry the new one, and it requires construction to read the getter **zero** times. The three-state credential semantics are replayed per request rather than assumed: on loopback an unavailable credential sends **no** authorization header at all rather than a fabricated one, off loopback it sends the fail-closed anonymous identity so the deployment answers with an honest 401, and the guard shows a single client moving between those states across successive requests. A getter that throws is fail-soft — the request still goes out under the no-credential branch, because a broken credential port should not take the whole wire down, and the exception may itself carry credential material. The same-origin relay form is checked to stay out of the getter path entirely, and every request is checked to keep the credential in the authorization header only — never in the URL, never in another header |
|
|
341
|
+
| `scripts/run-wire-auth-source-test.mjs` | **When** the outbound credential is read. A literal string is consumed at construction — the transport captures it in a closure and every later request reuses that one copy — so once the engine is replaced by another session and the credential rotates, a long-lived client keeps presenting the old one and the only way out is to rebuild the client along with everything hanging off it. The credential position now also accepts a getter that is called **once per outbound request**. The guard anchors on the deciding quantity, which is not "was the getter called" — reading once at construction and reusing the result would satisfy that too, and is exactly the shape being removed — but *which read produced the value on the wire*: it changes the getter's answer between two requests through the same client and requires the second request to carry the new one, and it requires construction to read the getter **zero** times. The three-state credential semantics are replayed per request rather than assumed: on loopback an unavailable credential sends **no** authorization header at all rather than a fabricated one, off loopback it sends the fail-closed anonymous identity so the deployment answers with an honest 401, and the guard shows a single client moving between those states across successive requests. A getter that throws is fail-soft — the request still goes out under the no-credential branch, because a broken credential port should not take the whole wire down, and the exception may itself carry credential material. The same-origin relay form is checked to stay out of the getter path entirely, and every request is checked to keep the credential in the authorization header only — never in the URL, never in another header. The same getter form is also accepted by two further entry points, and each is judged by its end result. For the workflow activity ledger, connection identity now follows the credential's *source* rather than the value read from it: a string is compared by value, a getter by reference, the same-origin relay declaration by its mode, and a change of form counts as a new source. The guard rotates the getter's answer between two warm-up calls and requires that no second stream opens and that the ledger keeps every frame it had already collected — before, a host could only pass a freshly read string, so each rotation looked like a different connection and the ledger was replaced by an empty one. A reconnect of that same ledger and the monitor's detail reads must then carry the rotated value, and the getter must be read exactly as many times as requests go out, so comparing identities never reads the credential. As the negative control, a different getter returning the same value opens its own stream, replaces the ledger, and the monitor holding the previous getter stands down instead of taking the slot back. For the detach cancel fallback, arming with a getter reads it zero times; the pick-up the host calls on its signal path reads it at that moment and still hands out only the two established forms, so an unchanged copy of the host's cancel leg sends the credential that is current at send time. The three-state semantics match the wire client (no authorization header on loopback when nothing is available, the anonymous identity elsewhere), a throwing getter never makes the pick-up throw, and string or loopback arms are handed back as the very same object. The widened inputs are checked with the compiler: the getter form is assignable, every previous form still is, the pick-up still narrows to the two forms, and illegal forms are rejected |
|
|
342
342
|
| `scripts/run-subagent-durable-divert-test.mjs` | The side-channel that keeps a **sub-agent's** content out of the leader's transcript, on the replay leg. A content frame stamped with a parent tool-call id belongs to a child, and rendering a child's tokens as the leader's own text is the pollution this divert exists to prevent — but the predicate only listed the four **live** frame shapes, while the durable leg replays the same segment in its **aggregated** form. Those frames fell straight through onto the main projection path, which is how a reconnect or a resumed session ended up with the child's answer printed as the leader's. The anchor is unchanged and shared: the parent tool-call id is what says whose frame this is, and whether the frame is an increment or a whole segment has nothing to do with whose it is — judging the two shapes separately is exactly how one of them got missed. Folding the aggregate into a synthetic increment would have been the smaller diff and the wrong one: an increment means *append*, so a segment that already streamed live and then replays whole would be counted **twice**. The two are kept distinct and the aggregate absorbs instead — a whole segment whose prefix is what the buffer already holds replaces it, which also makes a redelivery of the same frame idempotent, and a prefix that does not match falls back to appending both rather than deciding on the engine's behalf which version counts. Segment boundaries stay with the tool frames rather than moving into the aggregate arm, since closing there would turn a second replay of one segment into a second entry, and the increment arm is pinned to keep appending so a token run that happens to be a prefix of the next does not silently lose characters. When the host declares the non-interactive lane, a sub-agent's tool calls and results are also forwarded into the main output with their parent tool-use id (the sub-agent's text and thinking still stay out, as in the reference CLI); without that declaration the output is unchanged. |
|
|
343
343
|
| `scripts/run-subagent-content-budget-test.mjs` | The **byte** budget on the sub-agent transcript ledger. It used to be bounded only by *counts* — so many entries per child, so many children — and a count is not a budget when a single entry has no ceiling of its own: one tool result carrying an inlined attachment, or one long model answer, and a single slot sits on tens of megabytes. The guard anchors on how many bytes are **still held** after over-filling, not on whether truncation fired, because an implementation that flags the overflow without actually dropping anything satisfies the second and not the first. Dropping is required to leave a record — how much went and where the retained content now starts — and that record has to reach the render plan, because content that vanishes with no marker gives the reader a transcript shorter than what happened with nothing to say so; the record is one per child, updated in place, pinned to the front, and excluded from the budget it describes. Order matters and is checked: oldest entries go first and the live tail is trimmed only as a last resort, since taking the text the user is watching stream while older history survives is the wrong end. The total budget evicts a whole least-recently-used child rather than shaving every child, and the configuration surface is fail-loud on zero, negatives, non-finite and non-integer values — a silently ignored budget is the exact failure this exists to remove — with the rejection proven atomic so a bad second field cannot leave half a configuration behind. The defaults are checked to be a magnitude that can really be reached, since a number too large to hit is a field rather than a budget |
|
|
344
344
|
| `scripts/run-subagent-usage-projection-test.mjs` | Per-subagent usage, split by task. The engine's final accounting carries the delegated spend as **one total** — tokens, turns, task count — and no per-task breakdown, while every sub-flow turn on the stream carries its own usage. This package used to fold that away at the leader/sub-flow divide (a child's output tokens must never reconcile the leader's response length), so a client showing a subagent's detail pane had nothing to print. The split table can therefore only be accumulated from the stream, and this guard pins what that costs. The two existing leader-only arms stay **byte-for-byte unchanged** — the new arm is additive and always carries the sub-flow's own lane proof, so a host cannot mistake a child's numbers for the session window. Attribution is by the engine's own originating-task id — deliberately not a second `taskId`, which the event identity does not carry and whose absence would silently collapse every child under one parent call — falling back to the parent call id; a turn that answers neither is dropped rather than filed under an invented row, because merging two children's ledgers is worse than missing one. Cache-read tokens are read from the **engine's own shape** rather than the mirrored one, since the mirror fills that member with zero when the wire omits it and reading it there would erase the difference between *not reported* and *no cache hit*. A turn that reported no usage at all still counts as a turn and still adds its zeros — the numbers are a lower bound, and dropping the round would make the bound less true, so the honesty bit rides on the row instead and is never spelled `false`; such a round still emits its live arm, because the frame that says "this round has no account" is the one a real-time consumer most needs and the easiest one to drop. The same honesty bit also survives a terminal that carries no statistics at all: what the stream observed is unioned with what the final record says, so a run that already reported an unmeasured round cannot come out the other end looking like an exact zero. Finally the table says whether it is **partial**, and that verdict is anchored on the quantity that actually decides it: the engine's own totals. Turn count and row count must both reconcile before the table claims to cover the whole run; anything else — including totals that cannot be read — marks it partial, so the failure direction is always the safe one (a complete table called partial, never the reverse). The two accounts are kept separate and are never added together or used to correct each other. One more thing the totals cannot settle: the row key has **two namespaces** — the originating-task id and the parent call id it falls back to — and nothing upstream promises they are disjoint, so the same literal can name one child's identity and another child's parent call. Accumulation therefore keys on the origin as well as the id; the delivered table still keys on the bare id, and a cross-namespace clash is merged into one row that says so, with the partial verdict forced, because a row count and a turn count can both reconcile while the attribution behind them is wrong. The table itself is likewise a **per-stream snapshot** handed to the terminal projector by value rather than left on the caller's context: the three terminal projectors are public, so a host may drive one run through the stream and project another's terminal directly on the same context, and a table left behind would be attributed to whoever projects next — silently called complete whenever that run's own totals happen to match. Without a snapshot, both table keys are simply absent |
|
|
@@ -364,7 +364,7 @@ guard still cross-checks the table by name).
|
|
|
364
364
|
| `scripts/run-abortable-sleep-test.mjs` | The shared `abortableSleep(ms, signal)` leaf (consumed by `workflowClient.ts` and `agentSession/backgroundView.ts`'s poll backoff): normal timeout resolution, immediate wake-up on `abort` mid-wait, `clearTimeout` really firing on that path, and a post-resolve late abort staying a no-op |
|
|
365
365
|
| `scripts/run-durable-card-display-keys-test.mjs` | The durable approval row's two display keys survive the row→card recast in `surfaceFsApprovalAndDecide`: `governanceForced` stamps on strict `true` only (absence is "no evidence", never `false`), the row's rule offers (`ruleOffers`, or the older `ruleSuggestions` key that earlier servers send) pass through the same shape-narrowing reader as the live-frame leg and land on the **read-only** card key `ruleOffersReadOnly` — plus a standing pin that the durable leg never stamps the redeemable `ruleOffers` card position (the `/decide` body has no rule slot; offering a "don't ask again" option there would be an affordance nothing can honour), and a section for the parked twin of the classifier-unavailable fact: the upstream declares that key on the parked action itself, verbatim and under the same name as the synchronous ask, so this leg reads it rather than guessing a carrier name the way the deliberately unprojected keys must. The guard drives both legs with the same cause and asserts the card ends up byte-identical either way — the observable consequence of one reader serving two key paths, and the thing that silently diverges the day someone writes a second copy. Its own reach is printed rather than implied: what is proven is the package-boundary promise "on the row ⇒ on the card", not that today's engine flattens that key onto the pending row. A further section covers the two display facts the recast had been dropping for far longer. One of them the row has carried all along under a DIFFERENT NAME than the live frame uses — the frame puts it at the top level, the row nests it under the risk descriptor — and that difference in name is exactly why it went unnoticed; unlike the keys this leg deliberately refuses to project, its carrier is witnessed in the engine's own artefact rather than guessed. Neither is decoration: the shell's stand-aside arm reads them, so a call that matched a remembered allow rule which could NOT silence it looked like an ordinary ask on the durable path and was auto-approved with no card at all. Both land on the SAME card slot the live leg uses (one shape for the ends), verbatim bytes, present only when non-blank, never folded into an empty string — and the guard pins the discipline in both directions, including that a top-level key the upstream row does not actually have must still not grow this position. The security-class approval bit (`requiresRealApproval`) rides a parked row's card when the row carries it at top level, on strict `true` only, while look-alike nested carriers are ignored; this is pinned with a constructed row, because today's pending list does not carry the bit yet. A second, separate bit (`irreversibleParkGate`) marks a parked card whose row sits on the irreversible-ask gate kind — a gate-kind fact that covers asks the engine flagged for real approval at the first decision plus safety-tightened gates, with a known engine gap for approval demands raised only on a storage recheck — on the park path only, on the exact gate word only, and never in place of the real bit. Since 0.83.2 the recast also carries the row's ask origin (`origin`) exactly as the live-frame leg does — a non-empty string, verbatim, open vocabulary, never invented or defaulted — and the word a mandated question stands on (`mandate`), accepted only when it is one of the six known words and read back through `readApprovalMandate`; a word outside that set, or a malformed value, leaves the card without it. The word never adds the separate mandated key to the card, but a known word on its own makes `approvalIsMandated` answer true, because the engine only sends the word as the whole reason for that bit. An `origin` inherited through the row's prototype chain is not read, and since 0.83.4 neither is an inherited `mandated` on either leg — both legs stamp it from an own strict `true`, the same reading the seat crossing uses; the single judge `approvalIsMandated` reads `mandated` and `ruleOffersAbsence` the same way, so an inherited key no longer makes it answer true. |
|
|
366
366
|
| `scripts/run-session-memory-status-test.mjs` | The session **memory-status** read face (S-53): the two judgements three clients would otherwise each get wrong. First, *same status, different code* — this route's 404 carries two unrelated meanings (`not_found.session` = unknown or non-owned session; `not_found.route` = a pre-7.53 server that has no such route at all), so dispatching on the **status** would report "your deployment lacks this surface" as "your session does not exist". The verdict is anchored on `errorCode`, the two 404s are pinned to **different** verdicts, and — the load-bearing negative control — a 404 carrying **no** code falls to `failed` rather than guessing either way, since a wrong guess in either direction is a false statement a user would act on. 501 is allowed a codeless fallback because both of its arms mean the same thing here, and `capability.*` stays split from `feature.*` because those two share a status while their dispositions are opposite. Second, *absence means something different per key*: `optOutSource` and `lastCaptureAt` are legitimately absent on a **healthy** session (a zero-history session really is `{captureOptedOut:false, committedCount:0, foldedCount:0}` with no degradation at all), so reading absence as "off/none/0" asserts something unprovable. Two combined readers are pinned: capture opt-out is read from **both** its keys (a record-store fault yields `indeterminate`, never `active` — the difference between "your conversation is being remembered" and "nobody knows"), and last-capture is a **three-state** read whose discriminator is the *other* key, because `lastCaptureAt`'s absence alone covers both "ledger unreadable" and "genuinely no contributions" and therefore decides nothing; the two shapes are pinned to different verdicts so a single-key read turns red. The thin wrapper is the only IO: it never throws, drops malformed keys to absence rather than trusting them (an unreadable value must answer "don't know", never render as truth), refuses to spend a request on an empty `sessionId`, and passes `signal` through untouched |
|
|
367
|
-
| `scripts/run-crash-converged-projection-test.mjs` | The `crashConverged` read face on `GET /v1/approvals` (L-38): what the *previous life* of a crashed local engine left behind, projected for every client. Three judgements are pinned. First, **absence is not an empty list** — a missing key (an older server, deps not present, or a carrier that is not an array at all) returns `undefined`, and the client renders nothing; an empty array returns a present zero-count object, which is the server actually saying "none". Folding the first into `{total:0}` would have the client assert "nothing was left behind" on a surface a person uses to decide whether it is safe to re-run something — the worst possible direction for a false statement — so the two cases are pinned to different **return shapes** and a test asserts the two verdicts are unequal. Second, bucketing is a **four-term conjunction**: `orphanState === 'pending'` *and* `resumeSafe === true` *and* both approval-evidence keys (`originalDecision`, `decidedAtMs`) absent. A fifth term rejects any row carrying an **accessor**, and accessors are never invoked at all — reading one means synchronously running someone else's code, and `catch` catches throwing, not *never returning*, so a looping getter would pin the startup thread forever (the row cap does nothing against that shape). The same rule covers the three untrusted reads outside the row as well — the envelope's `crashConverged` key, the carrier's `length`, and every numeric index are read as own property *descriptors* and only data descriptors are used, so accessors and prototype entries read as absent and are never invoked. Such a key is treated as absent: if it was a required field the row is counted as dropped, if it was optional or additive the row survives without it. That also closes the ordering attack, since spreading runs getters in property order and an earlier one could `delete` the approval evidence before it is ever copied (measured before the fix: such a row reached the resume-safe bucket), and the check therefore moves ahead of the read, onto the property descriptors — from which the snapshot is then built directly, because checking descriptors and *then* spreading is two independent observations of the same row, and a non-throwing proxy can make the two `ownKeys` calls disagree (first showing `originalDecision: 'approve'` so the row reads as plain data, then omitting that configurable key so the snapshot loses the evidence; measured before the fix: the dangerous row reached the resume-safe bucket after exactly two enumerations, and after it, one). Keys are written with `Object.defineProperty` rather than plain assignment, because `'__proto__'` is a legal own enumerable key and `o['__proto__'] = x` does not store a value — it calls the prototype setter, letting a row whose own properties are all plain data (so the accessor gate never fires) inject a prototype whose `sessionId` getter deletes the approval evidence from the snapshot during validation; `defineProperty` fires no setter, so the key survives as ordinary additive data and the snapshot keeps `Object.prototype`. A row that simply arrives with a custom prototype is treated the same way, since the snapshot only enumerates own properties: approval evidence sitting on the prototype would never reach it, and a perfectly ordinary object with no proxy and no accessors could otherwise be called safe to re-run — real bodies come from `JSON.parse` and always carry `Object.prototype`, so nothing genuine trips it). Validation itself runs on a **null-prototype** dictionary and the bucketing verdict is carried out of that same pass rather than re-read from the delivered row, because every property lookup on an ordinary `{}` reaches `Object.prototype`: a polluted `sessionId` getter there would delete the approval evidence from the snapshot mid-validation and send the row to the safe bucket (measured before the fix). The row handed to the client is still an ordinary object — the null prototype is an implementation detail of the check, not of the value) — real JSON bodies are all data properties, so only a middle-layer-synthesised payload ever trips it, and it too lands in the human bucket rather than being dropped. The `decided` arm means the human had already approved and side effects may be half-landed, so it always goes to the human bucket, as does `resumeSafe === false` and — the last two terms — any row whose own fields contradict each other, since `pending` claims nothing ran while that evidence says somebody pressed approve. Deciding "not safe" costs one extra question (recoverable); deciding "safe" wrongly has somebody re-run work that already partly happened (not). A 2x2 truth table pins that exactly one cell is resume-safe, so reading either key alone turns red, and the contradictory rows are routed to the human bucket rather than dropped — they are real orphans, and the ones most worth showing. Third, unreadable rows are **dropped and counted**, never thrown and never passed through: the product is declared as `CrashConvergedRow`, so letting a row missing a required field — or carrying one of the wrong type — past would be a lie at the type level, and the closed literal discriminators (`decision` / `cause` / `orphanState`) decide family membership rather than being an open vocabulary. The measuring stick stops at the **type** floor, though: degenerate-but-well-typed values (`ts: NaN`, an empty `toolName`) are kept, because swallowing a real orphan over a decorative field is the worse direction, and the one deliberate exception is `approvalId`, which must be non-empty to be a row identity at all. `dropped` is kept separate from `total` so unreadable rows never inflate "N approvals were affected"; each row is a **one-shot snapshot** — every own enumerable key is read exactly once, and validation, bucketing and the handed-back value all read that same snapshot, so additive upstream keys survive while a **non-idempotent** getter (one that never throws, just answers differently on a second read) can no longer erase the approval evidence between the check and the bucketing (measured before the fix: such a row landed in the resume-safe bucket while its checked value was `"approve"`). Hostile carriers are counted rather than allowed to reject: **every** touch of the carrier is guarded — envelope property reads, `Array.isArray` itself (it throws on a revoked proxy), the `length` read, each indexed read and each row's property reads — and a traversal that dies halfway returns absence rather than a half-counted total. A row that cannot be read never takes the batch with it: its own shape check is inside its own guard, so one revoked-proxy row costs a `dropped` tick rather than collapsing the whole projection to absence — which a client would have read as "this deployment does not offer the surface". Traversal goes by **numeric index, never the carrier's own iterator protocol**, because `for...of` hands the carrier the question of which rows exist: an array carrying an overridden `Symbol.iterator` can yield nothing (measured before the fix: a real orphan became `{total:0}`, which a client reads as "the server said there are none") or swap a dangerous `decided` row for a safe-looking one (measured: `fake-safe` was returned in place of `real-danger`). Row count is capped at 100000 and the cap is checked **before** the walk: requiring only a non-negative integer `length` does not stop a proxy trap reporting a billion, and this surface runs on the startup / `--resume` path, where a synchronous spin freezes the thread (measured before the cap: twenty million rows took 18.3 seconds and twenty million index reads; a billion does not come back). The honest boundary is stated rather than overclaimed — a proxy can still lie in its `length` or index traps, which is the same thing as a host injecting a lying transport — and the widening of `ApprovalsResourceLike.list()` is proven **additive** by really running tsc over a legacy `{ pending }` mock *and* over the real `AgentClient` path — the projector takes `unknown` precisely because a parameter shaped as "an object with an optional `crashConverged`" is a TypeScript weak type that the installed SDK's own `list()` return shape shares no property with, which only a real-client compile would have caught — with a known-red control so a clean run means the checker spoke |
|
|
367
|
+
| `scripts/run-crash-converged-projection-test.mjs` | The `crashConverged` read face on `GET /v1/approvals` (L-38): what the *previous life* of a crashed local engine left behind, projected for every client. Three judgements are pinned. First, **absence is not an empty list** — a missing key (an older server, deps not present, or a carrier that is not an array at all) returns `undefined`, and the client renders nothing; an empty array returns a present zero-count object, which is the server actually saying "none". Folding the first into `{total:0}` would have the client assert "nothing was left behind" on a surface a person uses to decide whether it is safe to re-run something — the worst possible direction for a false statement — so the two cases are pinned to different **return shapes** and a test asserts the two verdicts are unequal. Second, bucketing is a **four-term conjunction**: `orphanState === 'pending'` *and* `resumeSafe === true` *and* both approval-evidence keys (`originalDecision`, `decidedAtMs`) absent. A fifth term rejects any row carrying an **accessor**, and accessors are never invoked at all — reading one means synchronously running someone else's code, and `catch` catches throwing, not *never returning*, so a looping getter would pin the startup thread forever (the row cap does nothing against that shape). The same rule covers the three untrusted reads outside the row as well — the envelope's `crashConverged` key, the carrier's `length`, and every numeric index are read as own property *descriptors* and only data descriptors are used, so accessors and prototype entries read as absent and are never invoked. Such a key is treated as absent: if it was a required field the row is counted as dropped, if it was optional or additive the row survives without it. That also closes the ordering attack, since spreading runs getters in property order and an earlier one could `delete` the approval evidence before it is ever copied (measured before the fix: such a row reached the resume-safe bucket), and the check therefore moves ahead of the read, onto the property descriptors — from which the snapshot is then built directly, because checking descriptors and *then* spreading is two independent observations of the same row, and a non-throwing proxy can make the two `ownKeys` calls disagree (first showing `originalDecision: 'approve'` so the row reads as plain data, then omitting that configurable key so the snapshot loses the evidence; measured before the fix: the dangerous row reached the resume-safe bucket after exactly two enumerations, and after it, one). Keys are written with `Object.defineProperty` rather than plain assignment, because `'__proto__'` is a legal own enumerable key and `o['__proto__'] = x` does not store a value — it calls the prototype setter, letting a row whose own properties are all plain data (so the accessor gate never fires) inject a prototype whose `sessionId` getter deletes the approval evidence from the snapshot during validation; `defineProperty` fires no setter, so the key survives as ordinary additive data and the snapshot keeps `Object.prototype`. A row that simply arrives with a custom prototype is treated the same way, since the snapshot only enumerates own properties: approval evidence sitting on the prototype would never reach it, and a perfectly ordinary object with no proxy and no accessors could otherwise be called safe to re-run — real bodies come from `JSON.parse` and always carry `Object.prototype`, so nothing genuine trips it). Validation itself runs on a **null-prototype** dictionary and the bucketing verdict is carried out of that same pass rather than re-read from the delivered row, because every property lookup on an ordinary `{}` reaches `Object.prototype`: a polluted `sessionId` getter there would delete the approval evidence from the snapshot mid-validation and send the row to the safe bucket (measured before the fix). The row handed to the client is still an ordinary object — the null prototype is an implementation detail of the check, not of the value) — real JSON bodies are all data properties, so only a middle-layer-synthesised payload ever trips it, and it too lands in the human bucket rather than being dropped. The `decided` arm means the human had already approved and side effects may be half-landed, so it always goes to the human bucket, as does `resumeSafe === false` and — the last two terms — any row whose own fields contradict each other, since `pending` claims nothing ran while that evidence says somebody pressed approve. Deciding "not safe" costs one extra question (recoverable); deciding "safe" wrongly has somebody re-run work that already partly happened (not). A 2x2 truth table pins that exactly one cell is resume-safe, so reading either key alone turns red, and the contradictory rows are routed to the human bucket rather than dropped — they are real orphans, and the ones most worth showing. Third, unreadable rows are **dropped and counted**, never thrown and never passed through: the product is declared as `CrashConvergedRow`, so letting a row missing a required field — or carrying one of the wrong type — past would be a lie at the type level, and the closed literal discriminators (`decision` / `cause` / `orphanState`) decide family membership rather than being an open vocabulary. The measuring stick stops at the **type** floor, though: degenerate-but-well-typed values (`ts: NaN`, an empty `toolName`) are kept, because swallowing a real orphan over a decorative field is the worse direction, and the one deliberate exception is `approvalId`, which must be non-empty to be a row identity at all. `dropped` is kept separate from `total` so unreadable rows never inflate "N approvals were affected"; each row is a **one-shot snapshot** — every own enumerable key is read exactly once, and validation, bucketing and the handed-back value all read that same snapshot, so additive upstream keys survive while a **non-idempotent** getter (one that never throws, just answers differently on a second read) can no longer erase the approval evidence between the check and the bucketing (measured before the fix: such a row landed in the resume-safe bucket while its checked value was `"approve"`). Hostile carriers are counted rather than allowed to reject: **every** touch of the carrier is guarded — envelope property reads, `Array.isArray` itself (it throws on a revoked proxy), the `length` read, each indexed read and each row's property reads — and a traversal that dies halfway returns absence rather than a half-counted total. A row that cannot be read never takes the batch with it: its own shape check is inside its own guard, so one revoked-proxy row costs a `dropped` tick rather than collapsing the whole projection to absence — which a client would have read as "this deployment does not offer the surface". Traversal goes by **numeric index, never the carrier's own iterator protocol**, because `for...of` hands the carrier the question of which rows exist: an array carrying an overridden `Symbol.iterator` can yield nothing (measured before the fix: a real orphan became `{total:0}`, which a client reads as "the server said there are none") or swap a dangerous `decided` row for a safe-looking one (measured: `fake-safe` was returned in place of `real-danger`). Row count is capped at 100000 and the cap is checked **before** the walk: requiring only a non-negative integer `length` does not stop a proxy trap reporting a billion, and this surface runs on the startup / `--resume` path, where a synchronous spin freezes the thread (measured before the cap: twenty million rows took 18.3 seconds and twenty million index reads; a billion does not come back). The honest boundary is stated rather than overclaimed — a proxy can still lie in its `length` or index traps, which is the same thing as a host injecting a lying transport — and the widening of `ApprovalsResourceLike.list()` is proven **additive** by really running tsc over a legacy `{ pending }` mock *and* over the real `AgentClient` path — the projector takes `unknown` precisely because a parameter shaped as "an object with an optional `crashConverged`" is a TypeScript weak type that the installed SDK's own `list()` return shape shares no property with, which only a real-client compile would have caught — with a known-red control so a clean run means the checker spoke Every row in the `needsHuman` bucket also carries a closed-set `needsHumanReason` (`approved_then_interrupted` / `pending_unsafe` / `unstable_row` / `contradictory`) derived from the same single read that bucketed it; `resumeSafe` rows never carry it, a same-named key on the supplied row is overwritten or removed, and the bucketing itself is unchanged. |
|
|
368
368
|
| `scripts/run-self-orchestration-denial-test.mjs` | The three judgements behind a **denied self-orchestration request** (server 7.57.0), each of which all three clients would otherwise get wrong on their own. First, whether to retry at all is a **conjunction that may not be loosened**: HTTP 501 *and* an `errorCode` that is **exactly** `capability.self_orchestration_required`. That code shares its shape with every other `capability.*` 501, so dispatching on the prefix would drag "some other capability is not wired up" into the retry arm — those requests do not become acceptable once the two keys are gone, so the client would spend a request and then tell the user the wrong reason. Negative controls cover all four directions: a sibling `capability.*` code, a truncated or suffixed variant of the right one, a codeless 501 (it decides nothing, so it decides nothing — no guessing), and the right code under 500 / 400 / 503 or a string `"501"`. The classifier reads structurally rather than by `instanceof` (a host may inject its own transport; across realms or duplicate SDK instances an understandable error would read as unreadable), so a class instance, a bare `{status, errorCode}` literal and an error carrying those fields on its **prototype** all reach the same verdict — and a hostile proxy or a throwing getter yields `null` instead of throwing, because this classifier runs inside a `catch` block where anything it throws escapes the caller's own guard. Second, removing the intent is a **structural** operation, not wording: `selfOrchestration` sits at the top level while `ultracode` sits under `settings` — two different stamping legs — and a client hand-writing `delete` will miss the second one, which costs the user the same failure twice. The single stripper is pinned to touch exactly those two: other `settings` sub-keys and their values survive byte for byte, `deferTools` is left alone (pulling `Workflow` out would be a behaviour change, not a removal of intent), additive unknown keys survive at both levels, the input object is never mutated, `settings` is only dropped entirely when `ultracode` was really there and nothing else remains (an already-empty one is left as is), a non-object `settings` is not touched at all, an `ultracode` that only exists on the prototype does not count, and the whole thing is idempotent. The end-to-end leg runs a real `buildTaskRequest` product through it and asserts the stripped body still passes the registration gate key by key. Third, on the capabilities body, **absence is not "switched off"**: a pre-7.57 server has no `workflowsGate` key at all, so reading absence as "the engine says no" asserts something the server never said, and the mirror-image disease is folding an **unrecognised** `denial` into `null`, which would have the client render "nothing was denied" when the truth is "denied, for a reason I do not recognise". Five shapes are pinned — caps unreadable, gate absent, closed-set member, unknown value, accessor — with the unknown arm carrying the raw token (or an empty one when the value is not even a string) and never collapsing to `null`. All four untrusted reads go through own **data descriptors** only, and the guard pins the getter invocation count at zero, since `catch` catches throwing but not *never returning*; a descriptor trap that throws and a revoked proxy both yield honest absence rather than an exception — though *what* absence means differs by field, and the guard pins that split rather than a blanket rule: an accessor on `workflows`, `workflowsGate` or `engineCan` reads as absent, while an accessor on `denial` reads as `{unknown:''}`, because a key that is **not there** is the gate saying "nothing was denied" whereas a key that is there but cannot be read is "denied, and I could not read why" — folding the second into the first is exactly the false statement this face exists to prevent. Two further pins came out of an adversarial review. The exported retry list is **frozen at runtime**, not merely `as const`: the verdict hands out that same reference, so any consumer splicing it once would poison every later verdict in the process — the guard asserts `Object.isFrozen`, that four different mutation attempts leave it byte-identical, and that a verdict issued *after* those attempts still carries the original two entries. And the classifier reads `denial` only **after** both criteria have passed, since it is not a criterion but an extra field on the verdict: the guard pins the getter invocation count at zero for any error that does not match and at most one for an error that does. The scope line is drawn explicitly rather than overclaimed — "no getter ever runs" holds for `projectWorkflowsGate`, which reads **wire JSON** where every field is an own data property by definition, but not for the classifier, which reads a **thrown value** that may well be an SDK `APIError` class instance carrying `status` and `errorCode` on its prototype; insisting on own data descriptors there would report a perfectly readable error as unreadable, so that side promises only that it never throws. A final pin covers the **integration document's own worked example** rather than the library: the shipped SDK's `tasks.stream()` is an `async` generator, so calling it issues no request at all — the POST happens inside `streamRaw` on the first iteration, and a `try` wrapped around the `stream(...)` call itself can never catch the 501. A client following a submit-shaped recipe on the streaming leg would never run the classifier, and the whole strip-and-retry path would silently do nothing. The guard drives the **real** `TasksResource` against a fake transport, offline, and pins both halves: the synchronous leg is in flight the moment it is called, the streaming leg has issued zero requests after the call and raises on the first `next()` — and it does so through the **real** error path, with `openStream` returning an actual 501 `Response` that the SDK's own `errorFromResponse` turns into the typed error, pinning the `openStream`→`errorFrom` call order so a transport that stops minting `errorCode` cannot pass. The documented recipe is then **executed** rather than keyword-counted: exactly one retry, a second body that really lost both keys while every other setting survives byte for byte, the caller's own request object left untouched, one disclosure and only one, a second 501 propagating with the request count still at two, and — after the first 501 — an abort leaving the count at one with nothing disclosed. A last leg is type-level: `stripSelfOrchestrationIntent` carries an SDK `TaskRequest` overload, because the wide `Record<string, unknown>` form erases the caller's type and the document's "strip and resubmit" line would not compile without an unsafe cast; a real tsc run over a virtual file proves both the narrow and the wide path, with a known-red control — and it compiles the document's two recipes **verbatim**, extracted from the section itself, because a recipe that does not compile is a recipe that was never given: `{ transientOk: true, signal }` is a TS2379 under `exactOptionalPropertyTypes`, which no amount of prose review had caught. The last thing pinned is the one that would have been quietest of all: the SDK's `stream()` returns only on a `done` or `failed` frame, so a stream truncated mid-run — or yielding nothing at all — ends the `for await` just as normally as a completed one. The documented `runOnce` therefore tracks whether it ever saw a terminal frame and raises when it did not, the guard's success fixture emits a real terminal and asserts the handler received it, and a truncated-stream control asserts that shape is reported as a failure with no retry and nothing disclosed. That terminal-frame rule then needed one more turn of its own: the underlying reader returns *normally* when the signal is aborted, so the check as first written rewrote a user's cancellation into a generic stream fault — a client keying off `AbortError` to suppress the error would instead have shown a failure, or resubmitted. Cancellation is therefore checked first, a real-SDK case aborts from inside the handler and asserts the original `AbortError` survives with no retry and nothing disclosed, and the document is checked for that ordering. The harness runs the documented `handle` and `transcript.note` as real spies rather than pushing frames itself, the drive loop rethrows exactly as the document does, and the disclosure ledger is proven to be the caller's own array by a positive identity assertion — without which the cancellation leg's "nothing disclosed" would have been vacuously true. Each recipe is compiled **on its own**, with a preamble that declares only what a host supplies and injects no library symbol, since compiling them together let the second one borrow the first one's imports, and the preamble's own types are decoupled from what the recipes import so the "remove the imports and it must fail" control fails for the right reason — which is checked by attribution, not merely by redness. Ordering is the last thing to get right: the cancellation check must come before the truncation error but **both** must sit behind the terminal-frame test, because a cancellation that lands after the run already reported `done` would otherwise overwrite a real outcome — one that may have already had effects — with "cancelled", and a person reading that will run it again. Aborting from inside `handle(done)` and `handle(failed)` are both pinned to still report success, and the ordering assertion is anchored inside the streaming `runOnce` body rather than the section, since the section's first `throwIfAborted` belongs to the synchronous recipe and would have made a reversed streaming recipe pass — and that ordering check is now anchored on the TypeScript AST rather than on text, since a comment reproducing the two statements in the right order let a genuinely reversed body pass. One more timing fact had to be written into the recipe: a single SSE read buffers several frames and the SDK yields them back to back, so checking the signal only after the loop lets a cancelled run keep consuming the rest of the chunk — measured, an abort inside `handle(turn_start)` still swallowed the `done` that followed and reported success. The recipe therefore re-checks after every non-terminal frame. Finally, the behavioural matrix is no longer run against a copy of the recipe: both recipes are extracted from the document, transpiled, and **executed** with injected host objects, so the disclosure assertion really exercises the document's own `transcript.note(disclose(...))` line, and the synchronous leg gets the same full matrix the streaming one does |
|
|
369
369
|
| `scripts/run-package-hygiene-test.mjs` | Everything `package.json` `files` ships — dist JS/typings and the Markdown docs — is screened line-by-line against a deny-list of strings that must never reach a public tarball (internal hostnames, codenames, person names, collaboration-process words, other repos' ledger ids and repo names; opaque ticket ids `CC-nnn` and post numbers `[nnnn]` are allowed as traceability references). Since 0.77.2 the build strips comments (`removeComments`; enforced by `run-dist-comments-test.mjs`), so what this gate screens in dist is code, string literals and type-level text. Markdown docs are enforced forward-only (CHANGELOG from 0.77.2, the integration doc from §81) because published sections are frozen.
|
|
370
370
|
| `scripts/run-integration-doc-freshness-test.mjs` | The **integration contract** (`docs/INTEGRATION-CLIENTS.md`) and the **changelog** (`CHANGELOG.md`) checked against the code, because a document with no guard rots — this one had a whole nest of drift found on it within a day of being written. Five directions, each a claim a machine can actually evaluate. (1) *Counting discipline*: the version-anchor row for the guard count may no longer carry a hand-copied number at all — it changes every time a guard is added, and writing it down is planting a timer; the export counts that are still hand-copied (the surface total, the test-hook count, the sentence describing the surface's internal composition, the sum of the sixteen domain rows, and the three sub-counts) are each compared against a value **derived** from `public-export-baseline.json`, which is the drift a human reviewer caught last time. (2) *Coordinates alive*: every `src/` `scripts/` `docs/` path the doc quotes must be on disk **and tracked by git** — on disk is not in the repo, and a doc that points readers at a file living only in its author's working tree sends every clone to nothing. A file landing in the same commit takes a named carve-out that **stops applying** the moment the file is really tracked (it can no longer let anything through, and the guard prints a line asking for it to be deleted) — deliberately not a red, since turning red on the very commit that lands the file would just manufacture a break that only a follow-up commit could clear. (3) *Arm tables*: the `hitl_out_of_slice` row and the `not_in_slice` fenced list must equal, name for name and in **both** directions, the case labels that really fall into those two buckets — read through the **TypeScript AST**, since which bucket an arm lands in is decided by the argument to `nothing(...)` and by nothing a comment says. The extractor is anchored to the one production projector: exactly one function named `eventToSdkMessage`, exactly one `switch (ev.type)` inside it, and no repeated case label — anything else is a broken anchor rather than a verdict, because a second same-shaped switch elsewhere in the file would otherwise overwrite the real one's conclusions and leave the doc agreeing with a switch nobody runs. The list is delimited by a machine-readable fence rather than by section headings, because the same section also names the terminal arms as a counter-example and prose boundaries cannot tell a member from a foil. (4) *Released sections are frozen*: an **append-only ledger** carries every version ever published — its number, the commit it was published from, and the sha256 of its section — and each one is checked, not just the current release, since pinning only the latest would set every earlier version free the moment the next one ships. The ledger cannot vouch for itself either: each recorded hash is **re-derived from that release commit** through git, so editing an old section and its constant together no longer passes — and the commit the row names is in turn checked against the `gitHead` npm recorded at publish time, which is the one value this repository cannot rewrite, so pointing an old version at a freshly written commit does not pass either. The *set* of versions that must be frozen comes from the registry too, so deleting an old row together with its section — which would otherwise remove that version from every set the guard looks at — is red rather than invisible. A failed registry call is classified rather than swallowed, and the classification consults the registry's own status code *before* it considers connection-level symptoms, so an auth refusal whose body happens to mention the network is still red rather than a skip. The version set is compared as full SemVer including prereleases — matching only `x.y.z` would silently drop a published `0.30.0-beta.1` and reopen the very hole this direction closes — and section headings are matched on a whole-version boundary so a stable release cannot bind itself to the release-candidate section sitting above it. Publishing itself is a two-phase protocol rather than a paradox: before a release, exactly one row may be marked pending and must name the current `package.json` version, exempt from the checks whose inputs do not exist yet; once the registry has that version the row must be promoted, so the temporary state cannot survive its own release. And because the pending exemption rests entirely on "this version is not out yet," it is refused outright when the registry cannot be reached to confirm that — an unverifiable premise is not a licence. Three reverse directions close the rest: a section claiming to be released but absent from the ledger, a ledger entry whose section has vanished, and a `package.json` version that was never frozen. Publishing appends a row; it never rewrites one. (6) *Sentinels*: the readers §5a hands hosts for "is this port installed" are checked against what the source actually declares it returns — `hasXxx()` is a `boolean`, the card port / HITL surface / wire target return `T \| null`, the `installHost` family returns `T \| undefined`. Testing a `null`-returning reader for `!== undefined` is *always true*, and a self-check that passes whether or not the port is installed is worse than none, because hosts retire their own fallback on the strength of it. Both directions are red: an implementation that changes its sentinel without the doc following, and a doc that names the wrong one. The roster covers the zero-argument readers and their `*For` variants alike — a multi-session host reads the variants, so leaving them off would let exactly the surface desktop depends on drift unwatched — and the §5a table and the §8-B checklist line are each checked against the source, because hosts tick the checklist, and a guard that only watches the prose table misses the line people actually follow. (5) *Packaging*: the README ships with the package and opens by pointing hosts at the integration doc, and the checklist names two more files as required reading before an upgrade — all three must really appear in the `npm pack` manifest, or an npm consumer follows a relative link that npmjs rewrites onto a private repository. Missing tooling never takes the whole verdict down with it: when git, npm or the registry is unreachable those legs print the `SKIPPED-SECTION` marker and the rest still judges, while a release commit the ledger names but git cannot resolve is red rather than skipped. The guard says in its own header what it does **not** do: it judges counts, coordinates, arm sets, released bytes and the packing list — whether a sentence is *right* is still for review and for the hosts to report (7) *Retired names*: every name in the per-version `removed` ledger of `scripts/export-liveness.json` may appear in the live sections of the integration doc only where a retirement note follows the name inside the same clause (or the table row's label cell is itself a retirement label); the scan is by identifier boundary after invisible text (HTML comments, link targets, reference-link labels, tag attributes) has been stripped, so a signature line in a code block, an inline `NAME = 4096`, a hidden note, or a note that belongs to a neighbouring name all count as a bare recommendation and go red. Frozen sections (a numbered section whose heading carries a version already recorded as released) are historical and never rewritten, so a retired name there is allowed only if the live retirement catalog has a row for it: a retirement label, the name, the version it left in (matching the ledger) and what to use instead, next to the sentence stating that frozen sections are historical records and the catalog is authoritative. A section whose version cannot be read, or is not yet released, is judged as live. |
|
|
@@ -454,6 +454,8 @@ guard still cross-checks the table by name).
|
|
|
454
454
|
| `scripts/run-inherit-env-wire-test.mjs` | The request field that lets the model-driven shell inherit this machine's whole environment (`"all"`) or keep the default scrubbed environment (`"scrub"`). Both words go out exactly as given on the two user lanes, an absent value is never filled in, and any other value (including a list of variable names) is refused before anything is sent. The value reaches a request only through a helper that checks the engine's version: an engine too old to know the field, or one whose version cannot be read, gets nothing, and the caller receives the reason plus one sentence saying what the run's shell gets instead, including that leaving the field out also clears an earlier whole-environment choice on engines that know it. An invalid-field refusal of a request that asked for `"all"` is read by code with one conditionally worded sentence, because the same code has other causes. |
|
|
455
455
|
| `scripts/run-submit-refusal-projection-test.mjs` | Four kinds of refusal that mean the request itself needs fixing, read by code with one sentence each. Permission rule lists that are too long or hold a bad entry are named precisely (which list, how many entries against the limit, or which entry by position) for a new request and, with different wording, when parked work cannot continue with the request stored for it; the same sentence is appended to a decision that fails for that reason. A tool declaration the engine refuses at the start of a run is read by its code rather than its HTTP status, so a status that looks like a temporary outage is not presented as one: the sentence says the same request will be refused the same way. A deployment refusing a single-user-only setting is attributed from the request the caller sent (the whole-environment shell setting, bypassPermissions, both, or unknown) and always says nothing ran and the local settings are not at fault. A body-shape refusal exposes its two lists of unrecognized and unsupported keys exactly as sent (only well-formed lists are read; a missing list is not treated as empty), with a three-way check of whether any listed key is a settings path. |
|
|
456
456
|
| `scripts/run-mandated-card-values-test.mjs` | Approval cards that newer servers mark as mandatory for four kinds of question (a deployment command policy, an operator's never-auto list, an exhausted durable budget, and supervisor routing) are built with the server's own code and fed through both card paths: the card carries the mandatory mark and the matching reason word, the shared check reports it as mandatory, no mandate word is invented, and a session-wide allow that the server declines to remember produces the single not-remembered notice. An ordinary approval-list card stays non-mandatory and remembered, and the older card shapes are shown to read the other way. This check needs an installed server package to run and reports itself as skipped otherwise. |
|
|
457
|
+
| `scripts/run-park-failed-reason-projection-test.mjs` | The fail-soft true cause when a parked approval cannot be decided and the engine's own `done{suspended}` terminal is the one handed back: that terminal is still returned as-is (no second terminal), but it carries `_sema_park_failed_reason` — the same sentence, byte for byte, that the durable leg puts on its synthesized terminal — and the result frame carries it through with credentials washed the same way `errors[]` is (value only; same sentence as the durable leg after washing). The key is left off when this host has no decision surface at all (no approval card port, a port installed under another session key, or no question overlay): that sentence's way out is to decide on the card, which such a host never shows. Pinned with a policy-fold refusal on decide, both stall exits, a host-retracted card, credential samples through the full bridge on both legs, the no-decision-surface forms, and the absent forms (user interrupt, malformed values, ordinary terminals). |
|
|
458
|
+
| `scripts/run-tool-history-mismatch-projection-test.mjs` | When a provider rejects every request of a session because a tool call in the conversation history no longer pairs up with its result, the terminal error row carries a machine-readable key naming the kind of mismatch and the first tool call id the provider named; a host that offers the rewind command gets the same recovery sentence CC shows (unless that host names the engine and the engine reports it cannot rewind the conversation), plus a count from the second time in a row, while the print and utility lanes, hosts without that command, and the result frame's error text stay unchanged. Only the run's own failure is read, never assistant text, and only when the provider sentence opens the provider's message: other statuses, other 400s and quoted copies do not match. The count is per session and per run, ignores replays, resets on any other outcome and never overstates when it cannot tell; the row-resolving helper that finds the prompt to go back before is checked for the target, the first-turn case and every undecidable case. The check also feeds outcomes built and published by an installed server package when one is available and reports that part as skipped otherwise. |
|
|
457
459
|
|
|
458
460
|
Each suite carries a floor that only moves up — a refactor that stops executing a group of
|
|
459
461
|
assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
|
package/dist/adapt/arms.js
CHANGED
|
@@ -10,11 +10,19 @@ import { registerSubagentAlias, registerSubagentContentAlias } from '../subagent
|
|
|
10
10
|
import { readAsyncLaunchedAgentReceipt } from '../toolResult.js';
|
|
11
11
|
import { isWorkflowAgentTaskId, recordWorkflowAgentTaskId } from '../workflow.js';
|
|
12
12
|
import { isTerminalStatus } from '../runTerminal.js';
|
|
13
|
+
import { readToolHistoryMismatchKey } from '../toolHistoryMismatch.js';
|
|
13
14
|
import { resolveEnginePanelTaskId } from '../engineAgentPanelStore.js';
|
|
14
15
|
import { fleetRowAgentType } from '../fleet/fleetRowAgentType.js';
|
|
15
16
|
import { chrome, mainLane, transcript, messageIdentityOf } from './ids.js';
|
|
16
17
|
import { wiringManifestViewOf } from '../adapter/downstream/wiringManifestView.js';
|
|
17
18
|
import { CANCEL_MESSAGE, decisionOf, flattenWireOutput, REJECT_MESSAGE, sanitizeToolUseBlock, shortTaskLabel, TASK_TOOL_NAMES, WORKFLOW_TOOL_NAMES, } from './wireShapes.js';
|
|
19
|
+
function toolHistoryKeyCarry(m) {
|
|
20
|
+
if (m._sema_api_error_message !== true)
|
|
21
|
+
return {};
|
|
22
|
+
const raw = m._sema_tool_history_mismatch;
|
|
23
|
+
const key = readToolHistoryMismatchKey(typeof raw === 'object' && raw !== null ? raw : undefined);
|
|
24
|
+
return key === undefined ? {} : { _sema_tool_history_mismatch: key };
|
|
25
|
+
}
|
|
18
26
|
const assistantArm = function* (m, { ctx, idOf, text, cards, inst }) {
|
|
19
27
|
const message = m.message;
|
|
20
28
|
const blocks = Array.isArray(message?.content)
|
|
@@ -120,6 +128,7 @@ const assistantArm = function* (m, { ctx, idOf, text, cards, inst }) {
|
|
|
120
128
|
session_id: ctx.sessionId,
|
|
121
129
|
parent_tool_use_id: null,
|
|
122
130
|
...(m._sema_api_error_message === true ? { _sema_api_error_message: true } : {}),
|
|
131
|
+
...toolHistoryKeyCarry(m),
|
|
123
132
|
}, ctx.now());
|
|
124
133
|
text.markEmittedText(textBlock.text);
|
|
125
134
|
}
|
|
@@ -499,6 +499,7 @@ function errorResult(ctx, parts) {
|
|
|
499
499
|
...selectedModelParts(parts.model),
|
|
500
500
|
...(parts.salvagedResult !== undefined ? { _sema_salvaged_result: parts.salvagedResult } : {}),
|
|
501
501
|
...(parts.outcomeUnknown === true ? { _sema_outcome: 'unknown' } : {}),
|
|
502
|
+
...(parts.parkFailedReason !== undefined ? { _sema_park_failed_reason: washCredentials(parts.parkFailedReason) } : {}),
|
|
502
503
|
});
|
|
503
504
|
}
|
|
504
505
|
export function doneToSdkResult(ev, ctx, observed) {
|
|
@@ -513,7 +514,9 @@ export function doneToSdkResult(ev, ctx, observed) {
|
|
|
513
514
|
const durationMs = elapsedMs(ctx);
|
|
514
515
|
const errorCode = runTerminalCode(terminal);
|
|
515
516
|
const degraded = degradedOf(r);
|
|
516
|
-
const
|
|
517
|
+
const parkFailedRaw = ev._sema_park_failed_reason;
|
|
518
|
+
const parkFailedReason = typeof parkFailedRaw === 'string' && parkFailedRaw.length > 0 ? parkFailedRaw : undefined;
|
|
519
|
+
const errorBase = { durationMs, stats, model: r.model, errorCode, degraded, observed, facts: effectiveFactParts(rec), parkFailedReason };
|
|
517
520
|
if (terminal?.kind === 'failed') {
|
|
518
521
|
const failedCode = terminal.code;
|
|
519
522
|
const failedMessage = terminal.message;
|
|
@@ -5,9 +5,11 @@ import { readRunCostFacts, terminalToSdkResult, TOOL_INPUT_JOIN_MAX_CALLS } from
|
|
|
5
5
|
import { coerceOutput, publishSubagentContentEvent } from '../subagentContentStore.js';
|
|
6
6
|
import { ACTIVE_RUN_BUSY_ERROR_CODE, OUTPUT_INVALID, isCorruptSessionCode, isLimitsExceededCode } from '../engineErrorCodes.js';
|
|
7
7
|
import { corruptSessionContent } from '../corruptSessionCopy.js';
|
|
8
|
+
import { classifyToolHistoryMismatch, noteToolHistoryTerminal, openToolHistoryGeneration, TOOL_HISTORY_REWIND_SENTENCE, toolHistoryMismatchRowKey, toolHistoryRepeatSentence, toolHistoryRewindOffered, } from '../toolHistoryMismatch.js';
|
|
8
9
|
import { isReviewPark, readRunTerminal, runTerminalCode } from '../runTerminal.js';
|
|
9
10
|
import { washCredentials } from '../displayUntrusted.js';
|
|
10
11
|
import { ccToolDenialKindForToolEnd, gateDeniedBy, gateOutcomeOf } from '../gateOutcome.js';
|
|
12
|
+
import { resumeReopenContent, resumeReopenFromError, suspendedReopenOf } from '../hitl/suspendedReopen.js';
|
|
11
13
|
import { isCcToolDenialKind, isCcToolDenialKindADenial, isGateDeniedByWord } from '../gateVocabulary.js';
|
|
12
14
|
const mainLane = () => ({ lane: 'main' });
|
|
13
15
|
function emitChromeFireAndForget(ctx, event) {
|
|
@@ -19,6 +21,27 @@ function emitChromeFireAndForget(ctx, event) {
|
|
|
19
21
|
}
|
|
20
22
|
catch { }
|
|
21
23
|
}
|
|
24
|
+
function suspendedReopenedNotice(ev) {
|
|
25
|
+
try {
|
|
26
|
+
const reopen = suspendedReopenOf(ev);
|
|
27
|
+
if (reopen.kind !== 'reopened')
|
|
28
|
+
return undefined;
|
|
29
|
+
const detail = resumeReopenFromError({ errorCode: reopen.code, retriable: true });
|
|
30
|
+
const rawTaskId = ev.taskId;
|
|
31
|
+
const eventId = eventSeq(ev);
|
|
32
|
+
return {
|
|
33
|
+
kind: 'suspended_reopened',
|
|
34
|
+
laneProof: mainLane(),
|
|
35
|
+
reopen,
|
|
36
|
+
...(detail !== null ? { content: resumeReopenContent(detail) } : {}),
|
|
37
|
+
...(typeof eventId === 'string' && eventId.length > 0 ? { eventId } : {}),
|
|
38
|
+
...(typeof rawTaskId === 'string' && rawTaskId.length > 0 ? { taskId: rawTaskId } : {}),
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
catch {
|
|
42
|
+
return undefined;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
22
45
|
export function pendingGateIsProvablyDifferent(prev, next) {
|
|
23
46
|
const a = prev?.checkpointId;
|
|
24
47
|
const b = next?.checkpointId;
|
|
@@ -98,7 +121,7 @@ export function isModelOutputErrorRowText(text) {
|
|
|
98
121
|
return text.trimStart().startsWith(`${MODEL_OUTPUT_ERROR_PREFIX}:`);
|
|
99
122
|
}
|
|
100
123
|
export const OUTCOME_UNKNOWN_ROW_PREFIX = 'Outcome unknown';
|
|
101
|
-
function syntheticTerminalRow(ctx, text) {
|
|
124
|
+
function syntheticTerminalRow(ctx, text, historyKey) {
|
|
102
125
|
return stamp(ctx, {
|
|
103
126
|
uuid: undefined,
|
|
104
127
|
session_id: undefined,
|
|
@@ -106,8 +129,22 @@ function syntheticTerminalRow(ctx, text) {
|
|
|
106
129
|
message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text: washCredentials(text) }] },
|
|
107
130
|
parent_tool_use_id: null,
|
|
108
131
|
_sema_api_error_message: true,
|
|
132
|
+
...(historyKey !== undefined ? { _sema_tool_history_mismatch: historyKey } : {}),
|
|
109
133
|
});
|
|
110
134
|
}
|
|
135
|
+
function engineRunIdOf(result) {
|
|
136
|
+
const v = result?.runId;
|
|
137
|
+
return typeof v === 'string' && v.length > 0 ? v : undefined;
|
|
138
|
+
}
|
|
139
|
+
function engineSessionIdOf(result) {
|
|
140
|
+
const v = result?.sessionId;
|
|
141
|
+
return typeof v === 'string' && v.length > 0 ? v : undefined;
|
|
142
|
+
}
|
|
143
|
+
function providerFailureStatusOf(result) {
|
|
144
|
+
const af = result?.apiFailure;
|
|
145
|
+
const status = typeof af === 'object' && af !== null ? af.status : undefined;
|
|
146
|
+
return typeof status === 'number' && Number.isFinite(status) ? status : undefined;
|
|
147
|
+
}
|
|
111
148
|
export function isOutcomeUnknownRowText(text) {
|
|
112
149
|
return text.trimStart().startsWith(`${OUTCOME_UNKNOWN_ROW_PREFIX}:`);
|
|
113
150
|
}
|
|
@@ -192,6 +229,7 @@ async function* runStreamInner(events, ctx, handle = {}) {
|
|
|
192
229
|
const seen = new Set();
|
|
193
230
|
if (ctx.startedAtMs === undefined)
|
|
194
231
|
ctx.startedAtMs = Date.now();
|
|
232
|
+
const historyGeneration = openToolHistoryGeneration();
|
|
195
233
|
const nestedUsageByTask = new Map();
|
|
196
234
|
let usageMissingObserved = false;
|
|
197
235
|
const toolInputByCallId = new Map();
|
|
@@ -478,7 +516,37 @@ async function* runStreamInner(events, ctx, handle = {}) {
|
|
|
478
516
|
const sessionUnreadable = ev.type === 'failed'
|
|
479
517
|
? isCorruptSessionCode(ev.errorCode)
|
|
480
518
|
: doneTerminal?.kind === 'failed' && isCorruptSessionCode(doneTerminal.code);
|
|
481
|
-
|
|
519
|
+
const historyMismatch = !busy.busy && !governance && !isModelOutputErrorText(errText)
|
|
520
|
+
? classifyToolHistoryMismatch({
|
|
521
|
+
code: ev.type === 'failed' ? ev.errorCode : runTerminalCode(doneTerminal),
|
|
522
|
+
status: ev.type === 'done' ? providerFailureStatusOf(ev.result) : undefined,
|
|
523
|
+
message: errText,
|
|
524
|
+
})
|
|
525
|
+
: null;
|
|
526
|
+
const historyEchoedSid = ev.type === 'done' ? engineSessionIdOf(ev.result) : undefined;
|
|
527
|
+
const historyRepeat = noteToolHistoryTerminal({
|
|
528
|
+
sessionId: historyEchoedSid ?? ctx.sessionId,
|
|
529
|
+
sessionEchoed: historyEchoedSid !== undefined,
|
|
530
|
+
generation: historyGeneration,
|
|
531
|
+
runId: ev.type === 'done' ? engineRunIdOf(ev.result) : undefined,
|
|
532
|
+
replay: ev.type === 'done' && ev.replay === true,
|
|
533
|
+
mismatch: historyMismatch,
|
|
534
|
+
});
|
|
535
|
+
const rewindLine = historyMismatch !== null && ctx.lane !== 'print' && ctx.lane !== 'utility' && toolHistoryRewindOffered(ctx.offersRewind)
|
|
536
|
+
? `\n${TOOL_HISTORY_REWIND_SENTENCE}${historyRepeat !== undefined && historyRepeat >= 2 ? ` ${toolHistoryRepeatSentence(historyRepeat)}` : ''}`
|
|
537
|
+
: '';
|
|
538
|
+
yield syntheticTerminalRow(ctx, `${sessionUnreadable ? `${rowText}\n${corruptSessionContent()}` : rowText}${rewindLine}`, historyMismatch === null ? undefined : toolHistoryMismatchRowKey(historyMismatch, historyRepeat));
|
|
539
|
+
}
|
|
540
|
+
else {
|
|
541
|
+
const echoedSid = ev.type === 'done' ? engineSessionIdOf(ev.result) : undefined;
|
|
542
|
+
noteToolHistoryTerminal({
|
|
543
|
+
sessionId: echoedSid ?? ctx.sessionId,
|
|
544
|
+
sessionEchoed: echoedSid !== undefined,
|
|
545
|
+
generation: historyGeneration,
|
|
546
|
+
runId: ev.type === 'done' ? engineRunIdOf(ev.result) : undefined,
|
|
547
|
+
replay: ev.type === 'done' && ev.replay === true,
|
|
548
|
+
mismatch: null,
|
|
549
|
+
});
|
|
482
550
|
}
|
|
483
551
|
if (ev.type === 'done' && isReviewPark(doneTerminal) && ctx.emitChrome) {
|
|
484
552
|
try {
|
|
@@ -514,6 +582,11 @@ async function* runStreamInner(events, ctx, handle = {}) {
|
|
|
514
582
|
yield resultFrame;
|
|
515
583
|
return;
|
|
516
584
|
}
|
|
585
|
+
if (ev.type === 'suspended') {
|
|
586
|
+
const notice = suspendedReopenedNotice(ev);
|
|
587
|
+
if (notice !== undefined)
|
|
588
|
+
emitChromeFireAndForget(ctx, notice);
|
|
589
|
+
}
|
|
517
590
|
const projection = eventToSdkMessage(ev, ctx);
|
|
518
591
|
if (projection.kind === 'message') {
|
|
519
592
|
yield projection.message;
|
package/dist/adapter/types.d.ts
CHANGED
|
@@ -18,6 +18,9 @@ export interface EmitContext {
|
|
|
18
18
|
startedAtMs?: number;
|
|
19
19
|
model?: string;
|
|
20
20
|
lane?: RequestLane;
|
|
21
|
+
offersRewind?: boolean | {
|
|
22
|
+
readonly engineBaseUrl: string;
|
|
23
|
+
};
|
|
21
24
|
emitChrome?(event: ChromeEvent): void | Promise<void>;
|
|
22
25
|
onDroppedFrame?(info: DroppedFrameInfo): void;
|
|
23
26
|
}
|
package/dist/decideReceipt.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { type GateOutcomeView } from './gateOutcome.js';
|
|
2
|
-
import { type ResumeReopenDetail } from './
|
|
2
|
+
import { type ResumeReopenDetail } from './hitl/suspendedReopen.js';
|
|
3
3
|
export interface DecideReceiptView {
|
|
4
4
|
status?: string;
|
|
5
5
|
taskId?: string;
|
package/dist/decideReceipt.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { gateOutcomeOf } from './gateOutcome.js';
|
|
2
|
-
import { resumeReopenFromError } from './
|
|
2
|
+
import { resumeReopenFromError } from './hitl/suspendedReopen.js';
|
|
3
3
|
import { DECIDE_CLAIM_LOST, DECIDE_GATE_MOVED, DECIDE_NOT_PARKED, DECIDE_WORKFLOW_HOST_UNKNOWN, DECIDE_WORKFLOW_LANE_CODES, DECIDE_WORKFLOW_REMEMBER_UNSUPPORTED, isDecideParkMovedCode, } from './engineErrorCodes.js';
|
|
4
4
|
import { effectiveWireErrorCode } from './wireErrorTriage.js';
|
|
5
5
|
import { readErrorString } from './wireFailureShape.js';
|
package/dist/detachWire.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type EnvLike } from './hostEnv.js';
|
|
2
|
+
import { type WireAuthTokenSource } from './engineWireSdk.js';
|
|
2
3
|
export declare const DETACH_HEADER = "x-detach-on-disconnect";
|
|
3
4
|
export declare const HEADLESS_DETACH_ENV = "SEMA_HEADLESS_DETACH";
|
|
4
5
|
export declare const DETACH_WIRE_MIN_ENGINE: readonly [number, number, number];
|
|
@@ -36,7 +37,11 @@ export interface DetachCancelArm {
|
|
|
36
37
|
principal: string | undefined;
|
|
37
38
|
getTaskId: () => string | null;
|
|
38
39
|
}
|
|
39
|
-
export declare function armDetachCancel(arm: DetachCancelArm
|
|
40
|
+
export declare function armDetachCancel(arm: Omit<DetachCancelArm, 'authToken'> & {
|
|
41
|
+
authToken: WireAuthTokenSource | {
|
|
42
|
+
mode: 'loopback-unauthed';
|
|
43
|
+
};
|
|
44
|
+
}): void;
|
|
40
45
|
export declare function isDetachArmed(): boolean;
|
|
41
46
|
export declare function detachedTaskId(): string | null;
|
|
42
47
|
export declare function detachCancelArm(): DetachCancelArm | null;
|