@sema-agent/client-core 0.82.7 → 0.83.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +97 -0
- package/README.md +15 -7
- package/dist/adapt/arms.d.ts +3 -2
- package/dist/adapt/arms.js +17 -11
- package/dist/adapt/ids.js +12 -8
- package/dist/adapt/textStream.d.ts +3 -2
- package/dist/adapt/toolCards.d.ts +2 -3
- package/dist/adapt/toolCards.js +1 -1
- package/dist/adapt.d.ts +1 -1
- package/dist/adapt.js +13 -6
- package/dist/adapter/activeRunSelfHeal.js +34 -17
- package/dist/adapter/downstream/eventToSdkMessage.js +1 -1
- package/dist/adapter/downstream/terminalToSdkResult.d.ts +2 -4
- package/dist/adapter/downstream/terminalToSdkResult.js +46 -8
- package/dist/adapter/runStream.js +25 -8
- package/dist/clientSlice.d.ts +0 -5
- package/dist/displayUntrusted.d.ts +24 -0
- package/dist/displayUntrusted.js +1437 -0
- package/dist/engineNoticeCodes.js +4 -0
- package/dist/fleet/fleetRowAgentType.d.ts +0 -1
- package/dist/fleet/fleetRowAgentType.js +0 -3
- package/dist/fleetTaskDesc.js +3 -3
- package/dist/gateVocabulary.js +1 -1
- package/dist/hitl/approvalDecisionNoteAudit.js +4 -2
- package/dist/hitl/approvalResolution.js +2 -0
- package/dist/hitl/frameRouter.d.ts +3 -2
- package/dist/hitl/sessionPolicyDeliverable.d.ts +14 -0
- package/dist/hitl/sessionPolicyDeliverable.js +80 -0
- package/dist/hitl/toolApprovalWire.d.ts +1 -0
- package/dist/hitl/toolApprovalWire.js +56 -10
- package/dist/hooksWireCaps.d.ts +6 -1
- package/dist/hooksWireCaps.js +220 -13
- package/dist/host.d.ts +25 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +3 -0
- package/dist/liveInitToolFace.d.ts +4 -2
- package/dist/liveInitToolFace.js +2 -2
- package/dist/model/catalogLoader.d.ts +1 -6
- package/dist/panelRunningHistory.d.ts +3 -2
- package/dist/peerFrames.d.ts +0 -1
- package/dist/peerFrames.js +2 -1
- package/dist/pluginHooksWire.d.ts +92 -0
- package/dist/pluginHooksWire.js +429 -0
- package/dist/printInitToolFace.js +1 -1
- package/dist/request/printNotification.js +26 -26
- package/dist/seam.d.ts +8 -2
- package/dist/seam.js +18 -7
- package/dist/subagent/engineSubagentResume.js +2 -1
- package/dist/systemReminderTag.d.ts +0 -1
- package/dist/systemReminderTag.js +0 -3
- package/dist/toolResult.js +1 -0
- package/dist/uuidV5.d.ts +3 -0
- package/dist/uuidV5.js +178 -0
- package/dist/webSearchWireCaps.d.ts +0 -3
- package/dist/webSearchWireCaps.js +21 -51
- package/dist/wireFailureShape.d.ts +2 -1
- package/docs/INTEGRATION-CLIENTS.md +794 -13
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -49,6 +49,103 @@
|
|
|
49
49
|
> 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
|
|
50
50
|
> commit 漏转在发布前就红,不再靠人记。
|
|
51
51
|
|
|
52
|
+
## 0.83.1(2026-09-26)
|
|
53
|
+
|
|
54
|
+
> 主题:patch —— 三件公共逻辑归包,只增公面、另有几处行为订正。① 展示层安全出口 `displayUntrusted` 三端单源,本包合成的终局行与结果帧 `errors[]` 在铸点洗掉凭据(CC-112 / CC-187);② 会话规则记录的无损判定 `sessionPolicyDeliverable`(CC-117);③ 插件 hook 逐条判「投给引擎 / 本客户端执行 / 如实不跑」,开机「不会执行」清单与 `/hooks` 标注措辞单源,同批不再把带参数的设置来源 command hook 条目送上请求、传了计划时也不再送 `mcp_tool` 条目(CC-174)。根公面运行期导出 1238 → **1254**(+16),公面类型 +26,`SettingsPort` +1 可选成员,`hooksForWire` +1 可选参;peer sdk 地板 `>=11.3.0` 不动。
|
|
55
|
+
|
|
56
|
+
### Added
|
|
57
|
+
|
|
58
|
+
- **`displayUntrusted(text, opts?)`**(CC-112):wire 派生文本上屏前的合成出口,三端一只。五张网默认全开 —— 凭据 URL 结构面(userinfo 整段、query 的每个值、fragment、路径段参数值、以已知凭据前缀打头的路径段)、凭据词级面(`Authorization: Bearer <值>`、`Authorization: token <值>` 一类非 bearer / basic 方案(方案词留、凭据段换记号;参数表形方案 —— Digest、签名算法形、OAuth 1.0 —— 参数名一律保留,`response` / `signature` / `oauth_signature` 的值换记号)、`api_key=<值>` / JSON 引号形 / `OPENAI_API_KEY=<值>` 这类键值对的值,分隔符两侧的空白不设上限;散文里不带标签的已知凭据字面形 —— `sk-…` / `ghp_…` / `AKIA…` / JWT 三段形 —— 整词换记号(随机段不足 16 位的 `sk-video` 这类名字不算);PEM 私钥块(含 PGP 私钥块)的主体换成一枚记号,头尾两行与它们旁边的换行保留,没有 END 行时遮到块尾,凭据标签后面紧跟私钥块时同样整块处理;诊断词与 `max_tokens` 一族计量单位不洗,方案词后跟的是一张窄词表里的常见英文词(`Invalid bearer token`;表外的词照遮)、反引号里点名的凭据变量名、裸复数 `tokens:` 后的纯整数计数、裸复数 `keys:` 后的键名清单也不洗)、控制符(C0 / DEL / C1 / 孤代理项)、双向与格式字符(格式类整类,ZWNJ / ZWJ 除外;含行 / 段分隔符)、行折平;顺序为凭据 → 字符 →(字符面改动过字节时)凭据 → 字符 … 到不动点。凭据两网在一份「判别视图」上匹配:ANSI 转义序列(按 ECMA-48 整族认:CSI、`ESC ( B` 一类字符集指定、`ESC 7` / `ESC M` 一类单字功能、已结尾的 OSC / DCS 控制串)、格式字符、控制符(以及本出口自己铸的 `\uXXXX` 转义)夹在标签、分隔符与值之间照样认得出,遮盖映回原文 —— 序列留在记号两侧,不会留下半截转义加记号的误导形;已结尾控制串的内容(窗口标题、超链接地址)另作一段文本洗;序列的最后一个字符恰是凭据词的首字母时(`<ESC>Bearer <值>`),按「终端吞掉这个字符」与「这个字符是词首」两种读法各判一遍,任一种认出就遮。整段文本线性处理(十万字符量级的标签密排 / 控制串引导符密排文本在百毫秒量级完成)。RGI 地区旗(黑旗加标签字符的整串)原样保留。参数表形方案的参数名与等号之间、等号与值之间带空白(`username = "bob"`)照认,头值折行续写照认(Negotiate / NTLM / Basic / Bearer 一类的首段照旧整段换记号);无 scheme 的 `user:pass@host` 与凭据标签后的转义反斜杠都不设长度上限(此前口令超过 256 字符 —— 例如把 JWT 当口令 —— 或 JSON 转义套三层以上时整段认不出、原样上屏)。选项:五张网各自可关(`credentialUrls` / `credentialWords` / `controls` / `bidi` / `foldLines`)、`keepLayout`(多行正文保 `\t` `\n`)、`mark`(危险字符呈现为可见转义 `\uXXXX`〔astral 为 `\u{…}`〕/ 点 / 空格,默认可见转义)、`max`(按输出封长,截点不劈转义记号、不劈代理对;不是有限数时不封)。幂等(带 `max` 时同样);干净文本原样返回(默认形开着行折平:换行、连续空白与首尾空白仍会折平);只管呈现、不参与任何判定。新增公面类型 `DisplayUntrustedOptions` / `DisplayMark`。接入文档 **§101a D-1 / 101a-1**。
|
|
59
|
+
- **会话规则记录的无损判定 `sessionPolicyDeliverable(behavior, rules)`**(CC-117):给一批同一 behavior 的规则串,答「能不能原样写进这条会话的规则记录」。只有一类能无损对上 —— 不带限定、恰好是一个工具名的 deny,写进 `toolDeny`(逐字、序不变、重复保留);其余每一条带一个成因词,闭集 `SESSION_POLICY_WITHHELD_WHY` 五词:`ask_no_bucket` / `qualified_deny` / `peer_wide` / `wildcard` / `not_a_tool_name`。🔴 **整批可送才送**:批里只要有一条送不了,`deliverable` 就是 `{}`,绝不给子集。坏形入参(表外 behavior、不是数组、读的时候抛错)不抛、一条都不送。每个成因词一句用户面话,单源 `sessionPolicyWithheldNotice(why)`,只收成因词、结构上回显不了规则串。判据与整批语义与此前宿主侧那一份相同,只有三处更严(都是此前判能写、而会话规则记录按原字节比一条都拦不住的形,本版扣下):① 首尾带空白的名字(如 ` Read`,`not_a_tool_name`);② 串里任何位置带 `*` 的规则(如 `*` / `Bash*` / `Web*` / `mcp__*__get_user` / `mcp__srv__get_*`,此前只扣下 MCP 工具段恰为 `*` 那一形;`wildcard`)—— 带 `*` 的规则在 CC 的规则语义里是通配(终端现行的匹配器只认 MCP 工具段恰为 `*` 那一形),而会话规则记录按精确名比、一条都拦不住;没有哪只工具的名字带 `*`,扣下的代价只是晚一拍生效。带括号限定的规则(如 `Bash(git push:*)`)仍报 `qualified_deny`:括号里的 `*` 是参数样式,不是工具名通配。③ 覆盖一个服务器或对端全部工具的规则不只认 MCP:引擎认得的每个协议命名空间(今天是 `mcp__` 与 `a2a__`)的 `<命名空间>__<服务器或对端>` 一律扣下(`peer_wide`)—— 此前 `a2a__payroll` 这类被判能写,而引擎把它当作覆盖该对端全部工具的规则,会话规则记录按精确名比一条都拦不住。成因词的用户面话 `sessionPolicyWithheldNotice(why)` 对任何入参都不抛(包括转成字符串时会抛错的值),表外的值一律回通用句。整批读不懂的入参(表外 behavior、不是数组、读的时候抛错)结果带 `unreadable: true`,与空批可分。新增公面类型 `SessionPolicyDeliverability` / `SessionPolicyWithheldRule` / `SessionPolicyWithheldWhy`。接入文档 **§101a P-1 / P-2 / 101a-3**。
|
|
60
|
+
- **插件 hook 每轮计划 `hooksWirePlan(opts?)`**(CC-174):设置来源投影 + 每一只插件 hook 的判定(`engine` / `shell` / `not_run` 加原因)+ 被治理筛掉的插件 hook + 被拿掉的设置来源 hook + 这一轮请求体 `settings.hooks` 的确切内容(`plan.wire`)。入参是宿主报的三条读数:`baseUrl`(读引擎的 `taskSettings.pluginHooks` 能力位)、`engineOwnedByThisShell`(引擎是不是本机由本客户端自起的)、`shellHookEvents`(本客户端自己的本地执行器为插件 hook 触发哪些事件)。宿主没报 `engineOwnedByThisShell`、或引擎能力还没探到时,判定原因与计划上的引擎事实都如实写「不知道」(原因 `engine_locality_unknown` / `engine_capability_unknown`,`plan.engine` 上对应位为 `null`),不说成「不是本机起的」「不支持」。传 `null` 与不传同处置。宿主交来的插件数据在读的时候抛错(会抛的 getter、已撤销的代理)与读口抛错同处置:这一轮 `pluginReader: 'failed'`、零插件判定与插件条目,设置来源照发,不外抛。🔴 每轮算一次、同一份用到底:先按它决定本地跳过哪些插件 hook,再把同一份交给 `hooksForWire({ plan })` 组请求体。接入文档 **§101a H-2 / 101a-4 / 101a-5**。
|
|
61
|
+
- **`SettingsPort.enabledPluginHooks?(): PluginHooksReading`**(可选,CC-174):已启用插件的 hooks(`hooks/hooks.json` 与 manifest `hooks` 按宿主加载后的形)、插件 id / 名 / 根目录 / 数据目录 / 是否由 managed 设置启用 / 声明的选项(敏感选项**只报声明,型面上没有值位**),加安全·bare 模式位。不实现 ⇒ 插件条目照旧不投,每个已装的端口告警一次。接入文档 **§101a H-1 / 101b**。
|
|
62
|
+
- **`hooksForWire(opts?)` 新增可选 `{ plan }`**:给了就原样返回 `plan.wire`(同一个对象);不给时不读插件口、不投插件条目。传 `null`(或 `{ plan: null }`)与不传同处置,不抛;`plan` 位上不是计划的值(空对象、数、串等)按不传处置 —— 设置来源照发,不会整份变空;把计划直接当第一个参数传入(`hooksForWire(plan)`)认得出,按 `{ plan }` 处置。接入文档 **§101a H-2**。
|
|
63
|
+
- **插件 hook 的查询口、措辞单源与闭集**(CC-174):查询口 `pluginHookVerdictOf(plan, pluginId, event, groupIndex, hookIndex)`(本地跳过同一只用);开机一次性「不会执行」清单 `hooksNotRunNotice(plan)`(按插件 × 原因一行,只含插件名、事件名与固定短句;名字过 `displayUntrusted` 的字符面,零宽 / 标签字符 / 软连字符渲成可见转义,引号转义成 `\"`,超过 120 个字符截断加省略号,转义与封长单遍完成)、`/hooks` 执行方标注 `pluginHookExecutorLabel(verdict)`、原因短句 `pluginHookReasonText(reason)` / `settingsHookDropText(reason)`;事件判定口 `isEngineFiredHookEvent(event)`;闭集 `ENGINE_FIRED_HOOK_EVENTS` / `PLUGIN_HOOK_DISPOSITIONS` / `PLUGIN_HOOK_REASONS` / `PLUGIN_HOOK_EXCLUSION_REASONS` / `SETTINGS_HOOK_DROP_REASONS` 与对应类型。判定原因闭集 `PLUGIN_HOOK_REASONS` 共 13 词,其中「引擎不是本机由本客户端起的」(`engine_not_local`)与「引擎不支持插件 hook」(`engine_capability_absent`)只在读数确实这么说时用;宿主没报 / 能力没探到走「不知道」两词(`engine_locality_unknown` / `engine_capability_unknown`)。接入文档 **§101a H-3 / H-4 / H-8 / 101a-7**。
|
|
64
|
+
- **插件 hook 投给引擎的投影形**(CC-174):本机自起的引擎报 `taskSettings.pluginHooks === true`、条目是 command、插件没声明敏感选项、并进去不超引擎每事件上限时,条目带 `plugin: { name, root, dataDir, options? }` 投出(`options` 只带已存的非敏感值)。🔴 截至本版,引擎尚未报出这一能力位,`plugin` / `args` 两键在引擎契约上也还没有定形 ⇒ 实际零投;判定、告知与本地跳过今天就生效。能力位与契约形在引擎侧同版出现时,本包在那一版核对键名与值形之后再放开;形与这里不同则按引擎的改。接入文档 **§101a-6**。
|
|
65
|
+
|
|
66
|
+
### Changed
|
|
67
|
+
|
|
68
|
+
- 🔴 **结果帧错误信封 `errors[]` 洗掉凭据 —— wire 可见的行为变化**(CC-187):`errors[]` 逐条过凭据两网,凭据位换成闭形记号(`«redacted:userinfo»` / `«redacted:query»` / `«redacted:fragment»` / `«redacted:secret»`),条数、顺序与其余字节不变。方向只会更安全(原来原样带出的凭据值不再带出);按 `errors[0]` 渲失败原因的端零改动即得净文本。射程按载体划:被挡原因不论是引擎、模型(经报告被挡的工具)还是 hook 反馈写的,进了合成终局行与 `errors[]` 就照洗;assistant 正文行、成功臂的 `result` 与错误信封的 `_sema_salvaged_result` 一个字节不碰。`errors[]` 里的非字符串元素原样放回。要在错误文本里抠 URL / 键值的消费方请按记号处理。接入文档 **§101a D-3 / 101a-2**。
|
|
69
|
+
- **本包合成的终局行正文在铸点洗掉凭据**(CC-187):终态错误行(`API Error:` / `Run stopped:` / `Run blocked…` / `Model output error:` 各形与「会话被占」那一句)与 `Outcome unknown:` 行,正文里的凭据位换成同一组记号;主机、端口、路径、query 键名、行首身份与其余文字逐字节不动。print 车道与交互车道是同一份产物;身份判定仍按引擎原话判完再洗,洗消只改呈现字节,不做字符面(那归呈现边界)。此前本包不洗,洗消只在终端自己的出口做,且只认 0.83.0 已改名的旧行类旗。`Outcome unknown:` 行与 `errors[0]` 洗后仍是同一句。接入文档 **§101a D-2 / 101a-2**。
|
|
70
|
+
- **设置来源 hook 条目上手写的 `plugin` 键在发出前剥掉**(CC-174):值为 `undefined` 也剥,条目其余部分照发;设置里写的 hook 不是插件,不许借这一键冒充插件上下文,请求体上的 `plugin` 位只由本包按插件读口交来的身份铸。宿主交来的设置文档本身不改,剥了会留一行 debug。接入文档 **§101a H-7 / 101a-6**。
|
|
71
|
+
- **字符面三只旧口改由同一只引擎实现,输出逐字节不变**(CC-112):`escapeDisplayControlChars` / `collapseLabel` / `capForDisplay`、同伴消息署名规范化与 hook 故障横幅共用 `displayUntrusted` 的字符面引擎,字符集保持各自旧集(双向族按枚举,不含零宽 / 软连字符 / 标签字符;署名规范化只折 C0 / DEL / 行段分隔符),全部 BMP 码元与代理组合对拍逐字节相同。它们窄于 `displayUntrusted` 的默认,新代码请用新口。接入文档 **§101a D-5**。
|
|
72
|
+
|
|
73
|
+
- **强制审批卡的那一句说明改口**(`mandatedApprovalDetail()`):此前那一句说「…so it is asked every time」,与引擎契约不符 —— 强制卡的约束是「存下的 allow 规则与记住的回答都清不掉,每一次调用要各自被回答」,而这一次的回答者可以是人,也可以是 hook 或部署运行的自动裁决(全域放行席、自动模式分类器、沙箱准入),所以这张卡未必每次都摆到人面前。新句:`no saved allow rule and no remembered answer can retire this question; each call is settled on its own — by you here, by a hook, or by an automatic check the deployment runs — and an answer covers only that call`。仍是零参数,仍不指人去写规则。按旧句逐字断言的测试改锚。接入文档 **§101a G-1**。
|
|
74
|
+
|
|
75
|
+
### Fixed
|
|
76
|
+
|
|
77
|
+
- **`failed` 帧上不是字符串的 `errorMessage` / `errorCode` 让流在合成终局行这一步抛错**:现在按缺席处理(回落到下一位,两位都缺席时正文为 `run failed`),流照常收尾。接入文档 **§101a D-2**。
|
|
78
|
+
- **交互车道重建的合成终局行丢了行类旗**(CC-187 同批):适配器把整条 assistant 消息重建成转录行时,上游帧带 `_sema_api_error_message: true` 的,重建行此前把它丢掉(合成行在交互车道上只剩 `model: '<synthetic>'` 可认),现在同样带上;模型行不新增任何键。接入文档 **§101a D-4**。
|
|
79
|
+
- **审批回决备注与子代续跑收据放过了双向 / 格式字符**(CC-112 残留收编):`readDecisionNoteAudit(ack).note` 与 `decisionNoteAuditLine(...)` 引用的备注,此前清掉控制字符却放过双向重排 / 格式字符(U+202E、U+2066、零宽、标签字符)、行 / 段分隔符与孤代理项,能把一条拒绝理由在屏上重排成另一句,现在一并折成空格(清洗后为空的备注与纯空白同处置:不算回显正文,不再渲一对空引号);`resumeSettledSubagent(...)` 成功时的 `receipt` 与失败时的调试日志行,此前只清 C0 / DEL,现在 C1(如 U+0085)、双向 / 格式字符与孤代理项一并换成 `.`,收据封长时不再截出半个代理对。除这几类字符外输出逐字节不变。接入文档 **§101a D-6 / D-7**。
|
|
80
|
+
- **设置来源的 `mcp_tool` hook 条目让整个请求被拒**(CC-174):引擎的 hooks 契约不认这一型,原样发出时整份 hooks 解析失败、整个请求被拒。现在:① 按计划组请求体(`hooksForWire({ plan })`)时不发,逐条记进 `plan.settingsDropped`,因此哪都不跑的进「不会执行」清单;② 不传计划的 `hooksForWire()` 在引擎点亮的事件上**照旧发出** —— 这类宿主看不到清单,整个请求被拒是它们唯一看得见的信号(被静默拿掉的守卫 hook = 用户以为在生效、其实没跑),有意保留;③ 引擎不点亮的事件上一律不发(引擎本来不跑这些事件,发出去只换来一次整份被拒)。接入文档 **§101a H-5**。
|
|
81
|
+
- **设置来源带参数的 command hook 条目被引擎只拿可执行名去跑**(CC-174):`type` 为 `command` 且带参数的条目 —— 非空数组(exec 形),或串 / 对象 / 数这些非数组值(上游本身不认,但照发同样会被剥)—— 引擎报出 `taskSettings.pluginHooks` 之前会静默丢掉 `args`、把 `command` 交给 shell 去跑(参数丢了,`command: "bash"` 这类会把 hook 的输入当脚本执行);现在能力位到货之前不发这类条目(不传计划时按未报判),到货之后原样发(非数组形由引擎响亮拒)。prompt / http 等条目上的 `args`、值为 `null` 的 `args` 不算,照旧原样发出;空数组 `args: []` 也不算带参数、照旧原样发出 —— 没有参数可丢,引擎按 shell 形跑的就是同一个可执行文件(如托管设置里的 `{ command: "/opt/guard/deny-dangerous", args: [] }`);只有命令串含空白、引号或 shell 特殊字符(shell 会拆词、展开或串接,与直接执行不等价)时才按带参数拿掉,判定只认由字母、数字与 `_ . / : + -` 组成的命令串为等价;其余条目(含形状不对的)照旧原样发出,由引擎响亮拒。拿掉原因与用户面话同一句(`exec_form_unsupported`)。不传计划的 `hooksForWire()` 拿掉这类条目、且那个事件由引擎点亮时,经日志口告警 —— 每个 settings 端口 × 事件 × 原因恰一次。接入文档 **§101a H-6**。
|
|
82
|
+
|
|
83
|
+
### Gates
|
|
84
|
+
|
|
85
|
+
- 新增常驻门三道:`run-display-untrusted-projection-test.mjs`(凭据两网的判据与反例样本、合成出口、旧口收编逐字节回归、残留收编、两车道合成行洗消、单源普查)、`run-session-policy-deliverable-test.mjs`(向量表、成因闭集、整批语义、坏形不抛、措辞纪律;在场时与宿主侧那一份逐例差分,分歧只许是首尾空白、带 `*` 的通配与非 MCP 命名空间整对端三类(逐类计数);另对实装引擎包的协议命名空间表双向对账)、`run-plugin-hooks-projection-test.mjs`(真端口夹具 + 真能力缓存进真 dist 的计划、请求体与措辞口)。`gates-manifest.json` 137 → 140,README「Guards」表同批 +3 行。
|
|
86
|
+
|
|
87
|
+
### Known limits(本版新增)
|
|
88
|
+
|
|
89
|
+
- 展示层:fragment 被整段换成记号之后,值里未编码的停字符(`|` `"` `<` 等)右边那一截落在洗法外(`#api_key=AB|<尾>` ⇒ `#«redacted:fragment»|<尾>`;真实 URL 进文本前已百分号编码);旧三口不跟随 `displayUntrusted` 的宽字符集(放宽是另一次行为变更);服务端逐串脱敏后投出的透传文本(引擎通告、审批卡正文与入参、回决备注、续跑收据)本包不再洗第二遍凭据;凭据面认不出的形 —— 驼峰名标签(`secretAccessKey`)、无前缀的 40 位 AWS 秘钥串、字面反斜杠写法的转义(`\x1B[33m`)夹在标签与值之间、括号或 YAML 块标记包住的值(块标记那一形里真值在下一行原样可见)、口令里含未编码 `/` `#` 的 userinfo、`Authorization` 方案词后换行再跟的普通值、标签与分隔符之间隔着空白又紧贴在上一只值后面的内层标签;仍会多遮的形 —— 标签后的普通词(`key: model`、`Missing key: ANTHROPIC_API_KEY`、`x-ratelimit-reset-tokens: 6m0s`)、方案词后的表外英文词(`basic subscription`)、引号里的键名、文档地址的 query / fragment 值、标签在行尾隔空行后的下一段首词(功能词 / 诊断词开头的除外)、`Bearer realm="…"` 挑战形里的 `realm=`;机读面(`errors[]` / 合成终局行)在记号两侧保留原有的控制符 / 格式字符(字符面归呈现边界)。与终端此前自带的那一只相比,凭据面上本包几乎只在更严一侧有差(控制符 / 格式字符 / 着色序列夹带、分隔符后长空白、`Authorization` 的非 bearer / basic 方案、散文里的已知凭据字面形),随机语料差分里更松的个例均为对方靠记号里的冒号误吃,或剥掉不可见字符后两边同样不遮。
|
|
90
|
+
- 会话规则:CC 旧工具名(如 `Task`)与带转义括号 / 反斜杠的名字会被判「能写」而系统里没有一层拦得住(与此前宿主侧同答);`mcp__<server>`、以及带 `*` 的规则(`mcp__<server>__*` / `mcp__<server>__get_*` / `Bash*` 等)在较新的引擎上有的其实能按集合命中,但 wire 上没有位说出对面是哪一代引擎,照旧不写(代价是晚一拍生效)。
|
|
91
|
+
- 插件 hook:引擎侧能力位与执行器尚未到货,插件 hook 在引擎腿上仍然不跑,本版做到的是如实说与判定就绪;守卫类插件 hook 投不出去时只如实告知,不改成逐次询问;插件 hook 模块(`register(on)` 形)与 skill / frontmatter hook 不在判定范围;exec 形里 `${user_config.KEY}` 由本包先替换、路径占位由执行方后替换,与上游次序相反;由引擎执行的插件 hook 没有进度与成功输出的展示面;设置来源的 hooks 在安全模式下的处置本包仍没有读口;不传计划的宿主若在引擎点亮的事件上配了 `mcp_tool` hook,整个请求照旧被拒(有意保留的响亮失败,改按计划组请求体即可);设置来源 command 条目上值为 `null` 的 `args` 照旧原样发出(与不写 `args` 同跑,上游本身不认这一形);不传计划的宿主上,被拿掉的 exec 形设置来源 hook(包括守卫类)只经日志口告警一次 —— 宿主若把日志只写进调试记录,用户看不到这一句,换钉前请改按计划组请求体并渲「不会执行」清单(见接入文档 §101y);空参数数组的 command 条目只在命令串由字母、数字与 `_ . / : + -` 组成时照发,引擎在 Windows 上改用 PowerShell 执行时这一等价判据未实测。
|
|
92
|
+
- 完整台账见接入文档 §101 末行「包侧缺口」。
|
|
93
|
+
|
|
94
|
+
## 0.83.0(2026-09-26)
|
|
95
|
+
|
|
96
|
+
> 主题:🔴 **minor**(行为面与型面都有 BREAKING)—— CC 形消息上 24 项非 CC 键按裁定 C-R103([8200] / [8232])改 `_sema_` 名、删除或改走 chrome 臂,转录 id 改成 UUID 形,0.82.7 标过渡的三只联网搜索旧读口到期删除,公面类型 −4;同版另有结果帧 CC 键 `terminal_reason`、用户层 `disableAllHooks` 在引擎腿上生效、退化审批卡三形拒收改写、注入件自愈句整族重写与引擎通告码册 +2;根公面运行期导出 1240 → **1238**,peer sdk 地板 `>=11.3.0` 不动。
|
|
97
|
+
|
|
98
|
+
### BREAKING
|
|
99
|
+
|
|
100
|
+
- **结果帧(`type:'result'`)上四个非 CC 键改名**([8200] 第 2–5 项):降级链 `degraded` → `_sema_model_degraded`(成功臂与错误信封都改;不叫 `_sema_degraded`,那是工具结果记录上另一个意思的键)、错误码 `errorCode` → `_sema_error_code`、成功臂上引擎选定的模型 `model` → `_sema_selected_model`、错误信封上从写出窗抢救回来的正文 `result` → `_sema_salvaged_result`(`result` 只留给成功臂)。值与缺席语义逐位不变(唯一例外:成功臂上的空串 `model` 此前原样铸出,`_sema_selected_model` 对空串缺席,见 Added),旧名自本版起零铸,旧转录不迁移、不改写。接入文档 **§100a K-2–K-5 / 100b**。
|
|
101
|
+
- **合成终局行的行类旗 `isApiErrorMessage` → `_sema_api_error_message`**([8200] 第 10 项):终态错误行与「结局不知道」行两处合成点同改,只在合成行上在场、恒 `true`;包内 print 车道 init 判定闸同批改按新名判「run 已结束」。🔴 端上凭 `isApiErrorMessage === true` 认出合成行、再对正文做凭据洗消的,必须同批改读新名 —— 不改读则洗消整段跳过,报错正文里的凭据(URL 里的用户名口令、`api_key=…` 这类值)原样上屏与落盘。接入文档 **§100a K-10 / 100d**。
|
|
102
|
+
- **工具结果记录删掉过渡名 `toolUseResult`**([8200] 第 1 项,0.81.0 起预告):只留 SDK 面的 `tool_use_result`,值与缺席语义不变,交互记录与 `-p` 帧在结构化结果这一位上从此同形。端内部渲染链沿用驼峰名的,请在自己的入口从 `tool_use_result` 同值起别名。接入文档 **§100a K-1**。
|
|
103
|
+
- **system 行三处退役**([8200] 第 6 / 8 / 9 项):压缩分割线 `compact_boundary` 的附带文件键 `attachedFiles` → `_sema_attached_files`(值原样);本包不再给 system 行补到达时戳 `timestamp`(`user` / `assistant` 行照旧恒带,入参自带的原样过境),也不再在任何 CC 形消息上铸 `isMeta`。包内把附带文件投成「Referenced file」附件行的读点本版两名并读(新名优先),0.84.0 起只认新名。接入文档 **§100a K-6 / K-8 / K-9**。
|
|
104
|
+
- **print 车道完成通知帧的十三个归因位平铺成 `_sema_` 蛇形名**([8200] 第 11–23 项):`taskNotificationToPrintFrame` 出口帧上 `stoppedBy` / `resumable` / `partial` / `exitCode` / `diagnostics` / `result` / `lines` / `recentSteps` / `editedFiles` / `task_type` / `source` / `seq` / `injected` 依次改为 `_sema_stopped_by` / `_sema_resumable` / `_sema_partial` / `_sema_exit_code` / `_sema_diagnostics` / `_sema_result` / `_sema_lines` / `_sema_recent_steps` / `_sema_edited_files` / `_sema_task_type` / `_sema_source` / `_sema_seq` / `_sema_injected`,取值与缺席纪律逐位不变。CC 已命名的六位(`task_id` / `status` / `summary` / `tool_use_id` / `output_file` / `usage`)、交互车道的完成通知与 wire 读法都不动,`PRINT_NOTIFICATION_FIELD_MATRIX` 的 `field` 列同批换名(`from` 列仍是 wire 原名)。接入文档 **§100a K-11–K-23**。
|
|
105
|
+
- **记忆写入从转录面撤下,改发 chrome 臂 `memory_saved`**([8200] 第 7 / 24 项):一次成功的 `Remember` 写入此前在转录面铸一条 `system{subtype:'memory_saved', writtenPaths:[…]}`,本版起转录面零行,改发 `{kind:'memory_saved', laneProof, id, notes}`(新型 `MemorySavedChromeEvent`;`notes` 是笔记原文、不是路径;`id` 由这次写入的 wire 稳定键派生,同一组 wire 事件重投同值、不与同一张卡的工具结果行撞 id)。`CHROME_ARMS` 把这一臂登记为 `required: true`:要保留「Saved N memories」那一行的宿主必须接它,否则那一行换钉当天从转录里消失 —— 按 kind 分派 chrome 事件、带兜底臂的端(兜底臂丢弃未知 kind)与只放行几只渲染位臂的端,不接这一臂就静默丢。接入文档 **§100a K-7 / K-24 / 100a-2 / 100d**。
|
|
106
|
+
- **转录 id 改成 UUID 形**(CC-147):`deriveTranscriptId` 与适配器铸出的转录行 `uuid` 此前是 `wid_<帧 id>` / `wseq_<seq>` / `wtc_<toolCallId>` 前缀串,现在是由同一个稳定键确定性派生的 UUID 形(小写 8-4-4-4-12);同一条派生路径上的 tool_use 卡 `message.id` 后半段(仍以 `msg_sema_` 开头)、段身份 `_sema_segment_id` / `text_segment_end.segmentId`、段末事件的 `committedUuid` / `committedUuids` 与 `attachment` 事件的 `id` 同批换形。只承诺 UUID 形、同一份适配器输入重放得到同一串 id、同一稳定键得到同一 id(不读时钟、不读随机源);同一组 wire 事件经投影口重投时,只有按 wire `eventId` 取键的行(带 `eventId` 的工具结果行、`memory_saved`)跨重投稳定(0.82.x 同样如此),派生参数不在承诺内,旧转录不迁移、新旧两形共存。按前缀形认「是不是本包铸的」的判据作废,改为比相等。接入文档 **§100a U-1 / 100a-6**。
|
|
107
|
+
|
|
108
|
+
### Added
|
|
109
|
+
|
|
110
|
+
- **`deriveTranscriptId(frame, ctx, suffix?)` 多一个可选第三参**:同一个稳定键派生多条消息时的去撞后缀,三个键空间都生效(此前只有帧 id 空间生效,seq / toolCallId 空间把它丢掉 ⇒ 同一帧派生的两条消息同 id);不传时与此前逐字同。接入文档 **§100a U-1**。
|
|
111
|
+
- **耐久 park 卡透传待决行的 `mandated`**(CC-183):待决列表行顶层带 `mandated: true` 时(引擎 7.31.0 在 park 行铸这一位,服务端 7.101.0 起投到行顶层),卡请求严格透传 `mandated: true`,判据口 `approvalIsMandated` 在 park 卡上从此能答 `true`;其它值与嵌套同名键不认,永不铸 `false`;对更早的服务端行为不变。接入文档 **§100a A-2**。
|
|
112
|
+
- **结果帧铸 CC 键 `terminal_reason`**(CC-175):只铸能从引擎终局推出的四个词 —— 成功 ⇒ `completed`,`limits.max_turns_exceeded` ⇒ `max_turns`,`limits.max_cost_exceeded` ⇒ `budget_exhausted`,`output.invalid` ⇒ `structured_output_retry_exhausted`;其余情形(墙钟 / token 预算到限、分类器拒到限额、取消、未知码、被挡、停泊、结局不知道、409 拒绝信封)键缺席,绝不铸 CC 闭集外的词。判据从已选定的 CC `subtype` 单源派生(success ⇒ completed、三个到限 subtype 各对一词),扁平形回放行上不会与 subtype 自相矛盾。新增公开读口 `terminalReasonForResult(msg)`,与铸点同一只判据(按 `subtype` 读、不读错误码),给端自拼的结果帧与旧转录用。接入文档 **§100a R-1 / 100a-1**。
|
|
113
|
+
- **错误信封也带 `_sema_selected_model`**(C-R103 Q5):done 帧的结果记录带非空串 `model` 时,错误信封同样铸出引擎选定的模型;`failed` 事件帧没有结果记录,这一位如实缺席。它与供应商自报名 `_sema_response_model` 是两位,不合并、不互相回落;空串 / 非串的 `model` 两臂都不铸(此前成功臂会把空串原样铸出)。接入文档 **§100a K-4′**。
|
|
114
|
+
- **`SettingsPort` 新增可选方法 `mergedDisableAllHooks()`,用户层 `disableAllHooks` 在引擎腿上生效**(CC-162):宿主返回设置合并之后的 `disableAllHooks`(与本机 hooks 执行器读的同一个值),为 `true` 时 `hooksForWire()` 只发 managed 设置里的 hooks —— 非 managed 设置关得掉自己的 hooks,关不掉组织下发的;会话目标的 Stop 钩子随之不发,显式开启的终局校验不再因为被挡掉的用户 Stop hook 让位。宿主没实现这个方法 ⇒ 行为同 0.82.x,每个已装的 settings 口经宿主日志口告警一次(`warn`);方法抛错 ⇒ 按合并值为真处置(只发 managed 的 hooks,不是一条都不发)。接入文档 **§100a H-1 / 100a-5**。
|
|
115
|
+
- **`resolveLiveInitToolFace` 新增可选 `opts.webSearchStamped`**:宿主明说这次请求盖没盖 `settings.webSearch`,明说的一律优先(从 settings 段采纳、env 缺席的那一形只有宿主知道);缺省按新判决推(`webSearchVerdictFromEnv(env).kind === 'honored'`),print 车道 init 帧回落面不再跟随旧读口。选项形命名导出为 `LiveInitToolFaceOptions`。接入文档 **§100a W-2**。
|
|
116
|
+
- **引擎通告码册 +2**(CC-186):`ENGINE_NOTICE_CODES` 加 `config.context_settings_swapped`(引擎 7.30.0:上下文覆写表随模型目录换档)与 `config.secret_env_scrubbed`(引擎 7.31.0:子进程环境里按名剥掉了凭据形变量,只列名、不带值),audience 都是 `operator`,码册 69 → 71、顺序与上游同源。开发依赖钉引擎 `~7.31.0`;引擎 7.31.0 的 BREAKING 项 `deprecatedLayers` 本包零读点,peer 地板不动。结构化卡白名单 `STRUCTURED_DETAIL_TYPES` 同批 +`read-inbox`(引擎 7.30.0 的 `ReadInbox` 工具卡,只在挂了跨会话收件车道的部署上出现;漏了这一词那一类卡会退回按模型面文本解析)。接入文档 **§100a C-1**。
|
|
117
|
+
- **`ADAPTER_DIVERGENCES` 多两条并改写两条**:新增 DIVERGENCE-11(system 行到达时戳只由宿主接收口补)与 DIVERGENCE-12(记忆写入改走 chrome 臂);DIVERGENCE-9(工具结果行 uuid 改为 UUID 形)与 DIVERGENCE-10(只铸 `tool_use_result`、驼峰名删除)同批改写文案,常量由 10 条变 12 条。接入文档 **§100a X-2**。
|
|
118
|
+
|
|
119
|
+
### Changed
|
|
120
|
+
|
|
121
|
+
- **忙碌会话自愈里「注入件」一族上屏句整族重写**:`activeRunSelfHealRow(…, 'injected')` 的十二句不再用「A system notification … the run …」这类内部词,改说「A follow-up message sema sent on its own (not one you typed)」与「the reply already in progress / an earlier reply」,每句只讲这是什么、当时什么挡住了它、要不要你动手;句柄从 `(run <id>)` 改为 `(id <id>)`。「不用你动手」只在两档说(宿主声明已跟上那一轮回复;会话已放开、正在重发),「会发」只在候决断那一档说,其余一律如实说没送达 / 模型没看到、不承诺重投。接入文档 **§100a E-1 / 100a-3**。
|
|
122
|
+
- 交接按 `delivery` 分句,修正两处不实:`queued` 时那一轮其实停在一道决断上(按重开判决再分卡已呈 / 答案在路上 / 卡呈不出三句),`parked_for_wake` 时那一轮早已结束、消息没到模型 —— 此前这两形都说「交给了正在工作的那一轮,请看着它」。
|
|
123
|
+
- `resending` 此前说「前一轮被取消了」,对「权限规则自己决了、那一轮自己跑完」那一形不成立,改说「会话已放开」;`not-delivered` 此前说「没能把决断卡摆上屏」,对仍在跑 / 引擎不认得 / 已结束 / 读不出状态几形不成立,改说「屏上没有一张答了就能放开会话的卡」。
|
|
124
|
+
- 处置分类(`selfHealSubmissionDisposition`)、续跟意图(`steerFollowIntent`)与用户形文案逐字节不变,`copy.rowFor` 整行覆盖照旧优先;按旧句做文本匹配的端需改锚。
|
|
125
|
+
- **联网搜索判决跟随服务端 7.101.0**(CC-184):`endpoint` 带 userinfo(用户名或口令任一非空,含只带用户名、含省了 `//` 的 `https:u:p@host`)⇒ 判形错(`malformed`,出错字段 `endpoint`),env 车道同判;空 userinfo(`https://@host`)照收、值原样。原因句逐字镜像服务端的 wire 句:provider 三形「(it is missing | it is not a string | it is not one of them)」、searxngParams「— entry #N …」(重名点出先出现那一项的序号)与 endpoint 的 userinfo 句,都不回显任何用户配的值。对 7.100.x 服务端:带 userinfo 的 endpoint 本包先拒、服务端仍采纳(本包更严);原因句已与服务端 7.101.0 的判官逐字对拍一致(对拍用 7.101.0 发布标签源码构建的判官)。接入文档 **§100a W-3**。
|
|
126
|
+
|
|
127
|
+
### Fixed
|
|
128
|
+
|
|
129
|
+
- **合成终局行的 `uuid` 改成 UUID 形、不再读墙钟**(转录 id 换形的同形存量):终态错误行与「结局不知道」行此前以 `err-<毫秒时间戳>` 当 `uuid`,同一毫秒里合成两行就撞 id(按 `uuid` 去重的宿主会吞掉第二行),也不是 UUID 形;现在与结果帧同一铸法。接入文档 **§100a U-2**。
|
|
130
|
+
- **入参不是工具真实入参的审批卡:三形一律拒收改写、如实说明**(CC-135):流内审批帧的入参超上限被省略或根本没带、且事件流上也没有这只调用的入参时,以及耐久待决行上的入参缺席或被换成超限标记 `{ truncated, bytes }` 时,卡请求带 `argsUnavailable: true`,卡回「编辑后批准」一次都不发(respond / decide / 会话级放行都不发),上屏 `EDIT_REFUSED_ON_BLIND_ASK_WARN_TEXT` 一次,这只审批仍挂着、可照原样批准或拒绝。修前这道闸只覆盖悬挂审批一形,另两形上的改写会原样上 wire,把工具的真实入参换成卡上那份残缺入参;纯批准与拒绝不受影响,有真实入参时改写照常转发。接入文档 **§100a A-1 / 100a-4**。
|
|
131
|
+
- 结局:流内帧腿 `{ decision: 'unresolved', editRefused: true }`(既有形);耐久腿新结局 `{ kind: 'failed', stage: 'card', editRefused: true }`(`FsApprovalOutcome` 的 `failed` 臂 +1 可选位),`approvalResolutionOf` 把它读成 `not_sent` / `edit_refused`(此前读成 `card_unavailable`)。
|
|
132
|
+
- 卡上说明句如实:流内帧入参超上限时按「path 解出来没有」分两句,都不再说「下面的 diff 是从问句重建的」;耐久待决行入参缺席 / 超限各一句,超限标记不再当入参渲上卡。只认与上游标记完全同形的对象(自有键恰好 `truncated: true` 与非负数 `bytes` 两个),近似形照旧当入参渲。
|
|
133
|
+
|
|
134
|
+
### Removed
|
|
135
|
+
|
|
136
|
+
- 🔴 **联网搜索旧三读口 `webSearchFromEnv` / `webSearchFromSettings` / `resolveWebSearch` 退出公面**(0.82.7 标过渡、本版到期):对 7.100.0 及以后的服务端,它们把拼错或缺席的 `provider` 整段丢掉,服务端看到「没带」、改用部署后端。换读 `resolveWebSearchVerdict(env, 原始段, apiKeyFor?)` / `webSearchVerdictFromEnv(env)` / `judgeWebSearchSettings(raw, apiKeyFor?)`(§99),还在调旧读口的端编译期即红。接入文档 **§100a W-1**。
|
|
137
|
+
- 🔴 **公面类型 −4**(CC-171):孤儿类型 `SemaNestedUsageByTask` 删除(从没有产出点真正返回过这个具名形;读终帧上的 `_sema_nested_usage_by_task` / `_sema_nested_usage_by_task_partial` 两个 wire 键即可,键本身不变);`ClientVerbSpec` / `CatalogCacheEnvelope` / `PeerFrameLane` 三个纯类型不再从包根导出(形状不变)。四名在已知各端源码里零取用,运行期导出零变化;同批删掉两个从未导出、零引用的内部函数,并给 11 个本就不在公面上的类型去掉多余的 `export`。接入文档 **§100a X-1**。
|
|
138
|
+
|
|
139
|
+
### Gates
|
|
140
|
+
|
|
141
|
+
- **包侧归层对账门**(CC-176,不改出包面):新门 `run-layering-shadow-export-test.mjs` 从本包一侧同时看四个消费本包的端:终端、桌面端、网页端、管理台。它用 TypeScript 语法树取出每个端**自己声明**的顶层运行时导出,与本包公面运行时导出名求交 —— 同名就是「同一件公共逻辑住了两处」,端上那份不在豁免表(`scripts/layering-shadow-exemptions.json`)⇒ 红。对本包的转口(`export { X } from '@sema-agent/client-core'`)是正形,不算。
|
|
142
|
+
- 读源:本机该端克隆的 `origin/main`,没有这个 ref 时退回已提交的 HEAD。只读,不 fetch,不看工作区。每端打印一行所读 ref、提交号、提交日期;所读提交早于 7 天时多打一行陈旧告警(不判红):本机克隆陈旧,这一端的结果只代表那一刻。`origin/main` 在却读不成提交(ref 悬空)按环境故障处理,不会静默改读 HEAD。
|
|
143
|
+
- 豁免行不常驻:每行必带到期版本(本包版本到了即红;最远只许写到当前 minor 之后第三条 minor 线)与本包票号;端已删掉那份而行还在 ⇒ 红。某个端的源码树不在场时该端打 `SKIPPED-SECTION`(部分跑,不算通过)。
|
|
144
|
+
- 尺子自证:内存里的假端植入四种形状的影子必须恰被抓到、五种合法形一条不抓;读源选择两形、陈旧告警(用构造的旧日期)各有一格;在一个临时仓上验证真实的 ref 解析(不在 ⇒ 退回 HEAD,悬空 ⇒ 故障);某端扫描面为零 ⇒ 报工具故障,不给「零命中」。
|
|
145
|
+
- 基线(09-25,各端本机 `origin/main`):终端 16 条、桌面端 0 条(本机克隆陈旧,只代表 08-13 那一刻)、网页端 6 条、管理台 3 条,全部登记,到期 0.85.0(逐条清单与迁移建议见接入文档 **§100e**)。0.85.0 之前端上既不删、也不改名的,本包门在 0.85.0 当天红。
|
|
146
|
+
- **联网搜索判决门的服务端对拍段**:对拍用的服务端判官可由环境变量 `SEMA_WEBSEARCH_ORACLE_ROOT` 指向一份发布包目录;默认读的服务端构建若早于判官源码最后一次提交,这一段打 `SKIPPED-SECTION` 并点名「构建陈旧」(此前按版本号去比陈旧构建,会误报不一致);读不了该树的提交史按环境故障退出。末行总结如实写这一段跑了没有、对的是哪一版。
|
|
147
|
+
- 登记物:`gates-manifest.json` 136 → 137、README「Guards」表 137 行、负控文档「自动化」表 23 行 = 负控套 `CASES` 23(新增一枚:注入一行已到期的豁免 ⇒ 门必须红且点名);公面基线不动。
|
|
148
|
+
|
|
52
149
|
## 0.82.7(2026-09-24)
|
|
53
150
|
|
|
54
151
|
### Added
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.
|
|
38
|
+
**Version:** 0.83.1
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -81,7 +81,8 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
81
81
|
pure-transient goes to chrome."*
|
|
82
82
|
- **Deterministic transcript ids** — derived from wire stable keys (frame id > seq >
|
|
83
83
|
toolCallId); `ctx.uuid()` only for keyless synthetic frames. Invariant (guarded):
|
|
84
|
-
same stream replayed ⇒ same id sequence.
|
|
84
|
+
same stream replayed ⇒ same id sequence. Since 0.83.0 the ids are UUID-shaped; only the
|
|
85
|
+
shape, same-stream determinism and same-key-same-id are promised, not the derivation.
|
|
85
86
|
- **Lane discipline as a type** — chrome events require a `LaneProof`; a historical ghost-row bug
|
|
86
87
|
family is structurally impossible to reintroduce.
|
|
87
88
|
- **Field-level round-trip guard** — every semantic wire field either maps into
|
|
@@ -305,7 +306,7 @@ guard still cross-checks the table by name).
|
|
|
305
306
|
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails |
|
|
306
307
|
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source |
|
|
307
308
|
| `scripts/run-permission-rule-issue-codes-test.mjs` | The rule-lint refusal codes an engine reports when it will not compile a permission rule. The SDK publishes neither a schema nor a type for them, so the package mints the table from the engine's own bytes and the guard pays the cost of that copy instead of leaving it to somebody remembering: it parses the codes the engine actually mints and reconciles them against the table in both directions, so a code added upstream (the user would see a bare code) and a code only the package believes in (a branch that can never fire) both fail. It also reconciles the table plus a small retired ledger against the engine's declared union, which is deliberately not the same set — one member was renamed and its old name is still declared — so reviving a code the engine will never mint again is impossible and a future stale member shows up immediately. Sentences are pinned one per code, mutually distinct, and split by family: a rule that is wrong and a rule that is legal but unsupported on this lane are different next steps and may not share a sentence. The engine's own message rides along as prose — sanitized and capped after escaping, never matched on |
|
|
308
|
-
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, nine words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because the sets differ in kind: one is genuinely closed on the wire (an out-of-set record is withheld by the engine, so reading one means the record is damaged) while the other is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine) Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. |
|
|
309
|
+
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, nine words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because the sets differ in kind: one is genuinely closed on the wire (an out-of-set record is withheld by the engine, so reading one means the record is damaged) while the other is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine) Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer; it must not say the question is asked every time — an answer for this one call may come from the person, a hook or an automatic check the deployment runs — and its wording is checked against the engine package's own description of the mandate A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. |
|
|
309
310
|
| `scripts/run-engine-identity-test.mjs` | The engine generation anchors on `/health` (`pid`, `instanceId`, `startedAt`; engine >=7.67.0). `/health` is the one unauthenticated door and its heartbeat is always green, so "another host restarted the shared engine" used to be discoverable only by having some authenticated request hit a 401 first — a path that misreads a restart as a network fault. The reader narrows each anchor independently (one malformed field never hides the other two) and always hands back a reading object rather than an absence, because the caller is asking which anchors answered, not whether there was a response. The comparison is a three-word verdict, not a boolean: `unknown` when the two readings share no comparable anchor at all — an empty intersection means nothing could be compared, never that nothing changed — and the boolean convenience is pinned so that only `true` is an assertion. Any comparable anchor differing decides `changed`, so a reading whose `startedAt` matches while its `instanceId` does not cannot be waved through as the same life; precedence only decides which anchor gets named in the diagnosis |
|
|
310
311
|
| `scripts/run-posture-knob-projection-test.mjs` | The three deployment knobs on the operator face (`serverGates.durableApproval` / `streamAskWindowMs` / `sessionAutoTitle`, engine >=7.67.0), each read as a value **plus who set it plus one operator-facing pointer** rather than a bare value — a bare boolean cannot answer why this particular machine is on this setting or how to pin it back, and a default that flips with the deployment shape is invisible without that. A worker too old to report readings still sends a bare boolean; the reader folds it into the same shell so consumers keep one branch, but raises a `legacy` bit, answers `undefined` from the machine-readable source accessor, and mints a sentence that contains no source word at all — claiming a source nobody reported is worse than admitting the worker cannot say. The other two knobs are honestly absent on such a worker rather than defaulted, a malformed side knob drops only itself while the anchor knob drops the whole reading, and the four sentences are pinned literally distinct so an operator can tell "not observed" from "not reported" from a real value. The last leg reads the installed SDK's `openapi.yaml` and `types.d.ts` directly, including a pin that exactly one knob on this face is numeric — the premise the millisecond-to-prose rendering rests on |
|
|
311
312
|
| `scripts/run-terminal-facts-projection-test.mjs` | The four unconsumed terminal-receipt facts: `TaskResult.effectiveReasoning` / `effectiveMemoryScopes` are narrowed into `_sema_effective_reasoning` / `_sema_effective_memory_scopes` on the CC-shaped `result` (success and error envelopes alike; a malformed value mints nothing, never a default tier), the resume **reopen** family (`resume.env_failed` / `tool_unavailable` / `tool_contract_mismatch`) is a frozen closed set with a reader and three-sentence copy that is disjoint from the refusal and retry-later sets, and `routePairingVerdict` reads `ModelInfo.routePairing` as ok / broken / unknown without policing the open set. A fifth section pins the structured-output key on the success result: the CC-spelled `structured_output` is the only home for the value the wire calls `structuredOutput`. The camelCase spelling this package used to mint on its own — a misspelling of the CC field, not an additive field of our own — rode alongside it for exactly one release (0.79.1) and is **absent from 0.80.0 on**, pinned both by own-key and by `in`, so a consumer still reading the old name sees `undefined` rather than a stale copy. The wire position is read exactly once, so a value-changing accessor is only ever asked for its first answer; absence stays absence; a wire key that is present but `undefined` mints nothing, since a key whose value is `undefined` makes a consumer that tests presence read "the engine produced nothing" as "the engine produced an empty result"; falsy-but-present values such as `null`, `0`, `""` and `false` are still minted, and so are shapes that are not records at all — an empty array, a populated array, a string, a number, a boolean — each carried through by the same reference, because the shape of that value is decided by the caller's own schema and the package does not get to filter it; and the error envelope carries no such key, because the CC error arm has no such field. Which spelling CC itself declares is witnessed from the mirror's own syntax tree rather than a constant copied into the guard, so the day that field is renamed upstream the guard says so. |
|
|
@@ -323,7 +324,7 @@ guard still cross-checks the table by name).
|
|
|
323
324
|
| `scripts/run-approval-card-retract-test.mjs` | The approval card's **decision-free retraction** and the in-stream frame leg's **outcome hand-back**: a host that must withdraw a card that no longer has a decision channel (session switch, engine switch, a tracker reporting the ask gone) answers `{ kind: 'retracted' }` and the package sends nothing on any of the three legs (in-stream frame, suspended ask, durable park), reporting `decision: 'unresolved'` with a `retracted` flag; `aborted` / `failed` / `deny` keep their meaning (a real deny is still posted), and `onToolApprovalOutcome` hands every in-stream outcome back to the host exactly once, tolerating a throwing or rejecting callback Also the single source for the host-side approval-outcome note (`approvalOutcomeNoteOf`): `settled` is whether the decision was delivered, `retracted` is an independent key present only when the card was retracted, and `detail` is the retraction / edit-refused sentence or the refusal code and message — never a fabricated sentence. |
|
|
324
325
|
| `scripts/run-memory-spec-wire-test.mjs` | The per-agent **memory spec** (`agents[].memory`) read once for every client, plus the judge for the engine's **closed** key list. Two states are kept apart that clients habitually collapse: an absent `scopes` means *no layers were specified*, never "zero layers", and an explicit `writeScope: null` is a positive fact — this run has memory **read-only** (no remember tool, no consolidation write; recall still works) — which is neither "unspecified" nor "memory off". Each of the four keys is read once, on own properties only (an inherited key never reaches the wire, so reading one would report a value the engine cannot see), and a key that is present but unreadable stays in its own slot instead of collapsing into "unspecified"; `enabled` must be a strict boolean and `scopeContract` is an open-set verbatim word. A spec that cannot be read at all answers *undefined*, kept distinct from an agent that simply has no spec. The judge earns its keep on the consequence: the engine checks this spec against a closed list, so one unlisted key — most often the retired singular `scope` — is refused together with the **whole agent definition**, not just that key, and the single sentence minted here says so. What counts as "on the wire" is decided by the bytes, not by the shape of the in-process object: both the reader and the judge work off a `JSON` snapshot of the spec taken **in its property position** (wrapped under the same key, never serialized as a root value — otherwise a `toJSON(key)` that branches on the key hands us one shape and the engine another: one such input made the snapshot say *read-only, no violations* while the real bytes carried the retired key and a writable scope), because `Object.keys` and `hasOwn` disagree with the serializer in ways that change the answer — a non-enumerable `writeScope: null` would otherwise be reported as "memory is read-only for this run" while the engine receives *unspecified* and may still write; a key whose value is `undefined` would be reported as a violation that never leaves the process; a `toJSON` (even inherited) adds keys that `Object.keys` cannot see, including the retired singular one; and a throwing getter would let the judge claim it had looked when the spec cannot be serialized at all. A spec that fails to serialize is reported as unreadable by both ports, and the snapshot is taken once, so every getter runs exactly once. The judge answers in three states, never two: `[]` is an assertion (*looked, nothing unlisted* — including an agent that carries no spec at all), a non-empty list is what it saw, and *undefined* means it could not read the spec (a non-object item, an array, an unreadable `memory`, a throwing getter) — an unreadable spec never poses as a clean one, and an array is not a spec so its index keys are noise rather than findings. It reports only the snapshot's string keys, sorted and bounded, so a prototype, symbol, non-enumerable or `undefined`-valued key is never blamed while a `__proto__` that really does serialize is; the empty-string key is kept rather than dismissed as noise, because it does serialize and dropping it left a non-empty violation list with nothing said about the consequence; every key name in the sentence is quoted and escaped one code point at a time, so no escape is ever cut in half (a half-cut escape used to make the closing quote itself look escaped) and an empty name, a key literally named `""`, a key containing a backslash and a real control character versus a literal `\uXXXX` all read as different violations; a name too long to show is marked `(truncated)` outside the quotes with a pointer to the judge's verbatim list, so a prefix is never presented as the whole key — two long names sharing a prefix do show the same, which is why the mark and the pointer are there; key names are sanitized and bounded on the way into the sentence while the judge itself hands back the verbatim key, because sanitizing belongs in prose and never in a verdict. The announced future key `projectKey` is still unlisted today and is reported as such, with a sentence saying it is not a typo. The two construction-time refusals (`config.memory_project_key_spelling` — a spelling, 400; `config.memory_write_scope_mismatch` — a conflict with the scope already in force, 409) join the existing `config.` recognition table rather than a second word list, and each gets one sentence stating that the refusal landed **before the run started**, so nothing ran; the engine owns the triage and an unrecognised code gets no sentence at all. The write face is widened in the same batch so the package can actually mint what the reader can read: `TaskAgentWireMemory.writeScope` is now an optional `string \| null`, since a reader that understands "memory is read-only for this run" while the writer cannot express it is worse than no reader at all — it makes the support look real. Minting `null` survives serialization and reads back as read-only, minting `undefined` drops the key and reads back as unspecified, and the projector still pins `writeScope` explicitly every time. The accepted key list is reconciled against an upstream witness rather than a second local copy: the guard reads the SDK's own declaration comment for this key, requires the two sets to match in both directions, requires that comment to still name the singular `scope` as retired, and requires it to still not mention `projectKey` — so the day upstream admits that key, the guard goes red instead of the package quietly continuing to promise a 400. Each port takes its own snapshot, so a consumer that wants one self-consistent answer about a spec that can still change under it should read `spec.unknownKeys` off the reader — which comes from the same snapshot as the four slots — and send that materialized data rather than the live object. |
|
|
325
326
|
| `scripts/run-approvals-feed-unknown-test.mjs` | The approvals feed tells three states apart: **N items waiting**, **nothing waiting**, and **this fetch did not come back, so we do not know**. Every way a fetch can fail (the call throwing or rejecting, a body that is not an object, a `livePending` section that is not an array — including the `null` seen in the field, a `pending` that is not an array or holds a malformed row) publishes `{kind:'unknown', why, at, mode}` on the subscription — never an empty snapshot and never silence. Real snapshots carry `kind:'snapshot'`; `snapshot()` still answers only with the last real one (a fact about the past) while `reading()` answers whether it is current (`unobserved` / `present` / `unknown`). Recovery always publishes a real snapshot again, even when the contents are byte-identical to before the failure. An unknown reading is never counted as zero: the awaiting-decision counts read `null`, the view is empty, and the tracker reports no removals, so cards on screen are not retracted for a failed fetch. Retry, backoff and circuit-breaking are unchanged. |
|
|
326
|
-
| `scripts/run-approval-resolution-test.mjs` | The single discriminated union for **how an approval decision ended** (`ApprovalResolution`: `decided` / `not_sent` / `unsettled`) and its one mapping entry `approvalResolutionOf`: every outcome of the durable-park leg (12 shapes) and of the in-stream frame / suspended-ask leg (4 shapes) lands on exactly one arm and cause; the three meanings of `decision: 'unresolved'` (retracted card, refused edit, respond that never settled) land on three different arms, with `retracted` winning when both flags are set; an interrupted durable card really posts a deny, so it is `unsettled` (`interrupted`), never `not_sent`; a safety stop never claims the decision left the package, and a refusal is only attributed to the engine when the outcome carries positive evidence (a wire error code, or the pointer key the engine mints on a rejection body) — an aborted or code-less decide failure is reported as a plain decide failure; the decision word is passed through without re-validating the closed set; an unreadable outcome is `unsettled` (`unreadable`), never guessed as `decided`; both cause vocabularies are frozen tuples with every word covered by a case, plus the three predicates; the approval-outcome note (`approvalOutcomeNoteOf`) is now derived from the union and compared key-by-key against a reference copy of its previous logic over the released inputs, with a self-check that the comparison can fail; a source-text pin asserts every `return` carrying `respondRefusal` also carries `'unresolved'`. No behaviour change: the existing outcome types and keys are untouched. From 0.80.0 that last pin reads the syntax tree instead of scanning lines: the same return had been rewritten across several lines with conditional spreads, a shape a line-wise search misses entirely, which would have quietly turned the pin into a check of nothing. |
|
|
327
|
+
| `scripts/run-approval-resolution-test.mjs` | The single discriminated union for **how an approval decision ended** (`ApprovalResolution`: `decided` / `not_sent` / `unsettled`) and its one mapping entry `approvalResolutionOf`: every outcome of the durable-park leg (12 shapes) and of the in-stream frame / suspended-ask leg (4 shapes) lands on exactly one arm and cause; the three meanings of `decision: 'unresolved'` (retracted card, refused edit, respond that never settled) land on three different arms, with `retracted` winning when both flags are set; an interrupted durable card really posts a deny, so it is `unsettled` (`interrupted`), never `not_sent`; a safety stop never claims the decision left the package, and a refusal is only attributed to the engine when the outcome carries positive evidence (a wire error code, or the pointer key the engine mints on a rejection body) — an aborted or code-less decide failure is reported as a plain decide failure; the decision word is passed through without re-validating the closed set; an unreadable outcome is `unsettled` (`unreadable`), never guessed as `decided`; both cause vocabularies are frozen tuples with every word covered by a case, plus the three predicates; the approval-outcome note (`approvalOutcomeNoteOf`) is now derived from the union and compared key-by-key against a reference copy of its previous logic over the released inputs, with a self-check that the comparison can fail; a source-text pin asserts every `return` carrying `respondRefusal` also carries `'unresolved'`. No behaviour change: the existing outcome types and keys are untouched. From 0.80.0 that last pin reads the syntax tree instead of scanning lines: the same return had been rewritten across several lines with conditional spreads, a shape a line-wise search misses entirely, which would have quietly turned the pin into a check of nothing. From 0.83.0 the parked leg's card-stage failure carrying `editRefused` (an edited approval on a card whose tool arguments were not available) reads as not sent with the `edit_refused` cause, the same cause the live leg already had; the flag counts only as a strict `true` on the card stage. |
|
|
327
328
|
| `scripts/run-panel-cycle-identity-test.mjs` | The **cycle identity** on agent-panel events and the fleet ledger's **departure read-out**: a background agent may be revived under the same id, so `fleet-row` and `end` events now carry the wire's own `cycleSeq` / `startedAt` when present (absent means the row has no notion of generations, never "generation one"), `isStaleEngineAgentPanelEnd` is the single rule for ignoring a late `end` from a previous cycle (only when both sides carry a comparable identity; absence never drops a real terminal), a changed `cycleSeq` is a new cycle for usage stickiness and buffer coalescing, the notification lane carries `cycleSeq` only when the wire really sent `seq`, and `task_remove` frames reach the host through `onTaskRemoved` with `removeReason` / `cycleSeq` verbatim, a stale previous-generation removal leaving the newer row in place. A terminal row held back because the consumer has no such row yet is also released by the keys it carries itself (its transcript id, or the delegating call id of a subagent already on screen under its wire id), since the key tables are only written once a row has actually been published — the release still goes through the one funnel, so the subagent stays one row; a running frame arriving after such a held terminal row is a stale snapshot when both sides carry a comparable generation and it matches (no event, the held row keeps its final usage), a revival when the frame is provably newer (the held row is dropped), and is treated as a revival when neither side can be compared. A subagent lifecycle event carries `agentType` only from an honest source — the fleet row's own agent type, recorded before it is folded into the row label — and omits the key when there is none, never substituting the display name |
|
|
328
329
|
| `scripts/run-workflow-size-warning-test.mjs` | The **workflow size warning** verdict shared by every host footer / panel: a three-state result (`warn` / `ok` / `unknown`) read off the optional fleet view keys, where an unknown size is never reported as a normal one (absent `totalCount` / `tokens` without positive over-cap evidence is `unknown`, naming the missing keys), positive evidence on either axis wins regardless of absent keys, the per-agent denominator uses the engine's started count only when it is not below done+failed (a smaller value is a stale reading), otherwise falls back to the done+failed lower bound only when both keys are present — and a lower-bound denominator only yields an upper bound of the projection, which can prove *within cap* (`ok`, flagged) but never *over cap* (`unknown`, with the upper bound exposed) — and the prior is used only when the engine itself reports zero started agents; cap precedence env > explicit guideline > default, prototype keys never act as a guideline, the env reader is pure and does not fall through to the second name on a bad first value; caps, guideline table, env names and the three copy variants are single-sourced |
|
|
329
330
|
| `scripts/run-approvals-stream-live-capability-test.mjs` | The engine's live-approval-push self-description (`capabilities.approvalsStreamLive`, engine ≥7.87.1), read the same four-state way as its four sibling capability readers: an absent key is reported as not reported (never folded into `false`), the value must be a strict boolean, and the one decision the feed consumer needs — whether it must keep pulling suspended asks itself — is answered by `livePendingNeedsReconcile`, which only says no when the engine explicitly says it pushes. |
|
|
@@ -350,7 +351,7 @@ guard still cross-checks the table by name).
|
|
|
350
351
|
| `scripts/run-usage-verbatim-channel-test.mjs` | The two complementary usage disciplines (core 3.0.0 metering semantics): the CC `ModelUsage` mirror stays pure (five pinned keys, `totalInputTokens` has no seat), while the sema-owned channel forwards the engine `turn_end.usage` object **verbatim** (six keys, incl. `totalInputTokens`) via `last_turn_usage.engineUsage` / `handle.latestEngineUsage` — honest absence on pre-3.0.0 engines, no fabricated zeros |
|
|
351
352
|
| `scripts/run-plan-review-decide-verify-test.mjs` | `decidePlanReview`'s post-decide honesty ([2315]/[2316], engine RB-471 family): a 2xx from the decide endpoint is **not** a terminal — the wire re-pulls the task status and words the outcome by the real shape (still-locked / legal new gate / genuinely left park / unverified), never claiming success it hasn't earned; when the engine answers that the session's stored resume context cannot be read, the outcome names the unreadable row and says the decision was not applied. Driven against a real fake-engine HTTP server through the shipped dist. The outcome queue item also carries a machine-readable `_sema_planReviewOutcome` (task id, a package-minted dispatch number, decision, effect) so a host can tell which in-flight decision an outcome belongs to without searching the prose; the prose is byte-identical, the number is minted only for a decision the in-flight latch admits, and a caller that passes no metadata gets no key. |
|
|
352
353
|
| `scripts/run-shell-gate-durable-allow-test.mjs` | #110: the durable approval leg for **shell** gates. The tool_end HOLD/REJECT predicate must cover Bash the same way park detection already does (otherwise the park poison frame `Operation aborted` hits the transcript, `endedCalls` swallows the real replayed result, and the user who pressed Yes watches a command that really ran be reported as aborted); a replayed, already-decided park must resume reading the stream instead of being reported as a failed turn; `lastEventId` must track numeric `seq` too. Mutation-proven: each of the three fixes reverted turns the gate red |
|
|
353
|
-
| `scripts/run-hitl-gate-honesty-test.mjs` | [2393] the four HITL disciplines that a passing type-check cannot see. (1) The park predicate and the `tool_end` predicate must cover the **same** set — the park side admits a first-class `kind:'tool_approval'` gate for *any* tool name, and a `tool_end` frame carries no `kind`, so the frame-level judge falls back to the engine's exact abort marker; otherwise the poison frame hits the transcript and `markEnded` swallows the real replayed result (the #110 disease, reopened on kind-only gates). (2) The already-decided identity criterion is **one-shot**: its two inputs are monotonic, so without consumption one successful decide makes every later park failure — including a real `approvals.list` outage — read as "already resolved" until the 24-hop budget runs out and reports a cause that has nothing to do with what happened. (3) A `plan_review` card dismissed without an answer must be re-presentable: the idempotent re-arm short-circuit re-publishes the still-armed card, and a stale armed id (responder gone) re-arms from scratch rather than presenting a card nobody can answer. (4) `HitlSafetyError` is a safety signal — the `remember` fallback arm must re-raise it instead of auto-retrying the decide, while a plain unknown-key 400 still falls back. (5) The polling leg reschedules after an escaping throw and flips `mode()` to `idle` once it consistently fails, so the honesty surface stops reporting a dead feed as live. (6) The live-frame leg carries the fact behind "you are being asked because the auto-mode classifier could not run" all the way to the card port. Transit narrows on SHAPE only — a non-empty cause string is taken verbatim, an open set, because the word table's owner is the engine and re-checking a closed table at the package boundary would drop a legal value the day a new cause word appears, which is exactly the information worth keeping. A malformed carrier degrades to absence rather than half-minting, and absence stays absence: it covers "the classifier answered", "this ask never qualified" and "this deployment has no classifier" at once, so nothing may render it as reassurance. The guard also pins the division of labour that makes the open set safe — the same word that transits is judged again by the public display reader, which narrows to the availability axis, so a word the engine says it never stamps on this fact renders no sentence while still being visible on the card for triage A later section pins the split this release introduced on the deny close-out frame. Until now every denied tool call was stamped with the same sentence — the one that says *the user* does not want to proceed — including the calls denied automatically on a lane that has no approval surface at all, where nobody was ever asked. The guard drives all three shapes (a person pressed No, a rule settled it, nobody said which) through both close-out arms and the durable park leg, and pins that the third shape is byte-identical to the previous release: an attribution nobody supplied is not evidence for either answer. The rule-settled shape carries the shell's own reason on a second line when there is one and stands alone when there is not, because a blank line where a reason should be reads worse than no line at all. The attribution is read from own data properties only, so neither a polluted prototype nor a getter can make an automatic denial claim a person made it — and the getter case is pinned to never run at all. The transcript classification word is minted only on the two paths where the upstream transcript format really carries one; the three classifier words and the two abort words are left absent, with the abort words pinned against the strings this package actually normalises interruptions to, which are different strings. A final section pins the decide-operation observer: one `start` in the same tick as the first request and exactly one `end` after the last attempt has settled, across success, retried timeouts, exhausted transient failures, semantic refusal, binding mismatch and both kinds of caller abort, with nothing between retries; an observer that throws or rejects — even when logging that fault fails — never changes what is sent or returned, and the stream-level dependency reaches both durable park legs. A package-internal re-delivery of the same decision (plain approve after an older server rejects the session-scope flag, or a re-send without the attribution key) is reported as one operation with a single start and end. An operation handle only groups sends for the same session and bound call — a send for another gate through the same handle is its own operation — and closing a handle while a send is still in flight defers the end until that send settles. When a person picks "allow for this session" on a parked card and the grant is known not to have been stored — the server answers so, which newer servers do for the gates they can recognise from the parked row as needing a person each time, or the server refuses the session-wide grant with one of the refusal codes that are fixed by the row or the deployment and the package falls back to a plain approval — the parked path now says so with the same line the live path uses, and the receipt carries the server's bit for hosts that call that path directly; an answer without the bit, or any other failure — including a conflict that an internal retry can hit after an earlier send already stored the grant — is treated as unknown and says nothing, a plain approval that never asked for the grant says nothing, and a host logger that throws after a successful decision, on either card path or in the bridge's retry step, can no longer turn it into a second send or a failure. |
|
|
354
|
+
| `scripts/run-hitl-gate-honesty-test.mjs` | [2393] the four HITL disciplines that a passing type-check cannot see. (1) The park predicate and the `tool_end` predicate must cover the **same** set — the park side admits a first-class `kind:'tool_approval'` gate for *any* tool name, and a `tool_end` frame carries no `kind`, so the frame-level judge falls back to the engine's exact abort marker; otherwise the poison frame hits the transcript and `markEnded` swallows the real replayed result (the #110 disease, reopened on kind-only gates). (2) The already-decided identity criterion is **one-shot**: its two inputs are monotonic, so without consumption one successful decide makes every later park failure — including a real `approvals.list` outage — read as "already resolved" until the 24-hop budget runs out and reports a cause that has nothing to do with what happened. (3) A `plan_review` card dismissed without an answer must be re-presentable: the idempotent re-arm short-circuit re-publishes the still-armed card, and a stale armed id (responder gone) re-arms from scratch rather than presenting a card nobody can answer. (4) `HitlSafetyError` is a safety signal — the `remember` fallback arm must re-raise it instead of auto-retrying the decide, while a plain unknown-key 400 still falls back. (5) The polling leg reschedules after an escaping throw and flips `mode()` to `idle` once it consistently fails, so the honesty surface stops reporting a dead feed as live. (6) The live-frame leg carries the fact behind "you are being asked because the auto-mode classifier could not run" all the way to the card port. Transit narrows on SHAPE only — a non-empty cause string is taken verbatim, an open set, because the word table's owner is the engine and re-checking a closed table at the package boundary would drop a legal value the day a new cause word appears, which is exactly the information worth keeping. A malformed carrier degrades to absence rather than half-minting, and absence stays absence: it covers "the classifier answered", "this ask never qualified" and "this deployment has no classifier" at once, so nothing may render it as reassurance. The guard also pins the division of labour that makes the open set safe — the same word that transits is judged again by the public display reader, which narrows to the availability axis, so a word the engine says it never stamps on this fact renders no sentence while still being visible on the card for triage A later section pins the split this release introduced on the deny close-out frame. Until now every denied tool call was stamped with the same sentence — the one that says *the user* does not want to proceed — including the calls denied automatically on a lane that has no approval surface at all, where nobody was ever asked. The guard drives all three shapes (a person pressed No, a rule settled it, nobody said which) through both close-out arms and the durable park leg, and pins that the third shape is byte-identical to the previous release: an attribution nobody supplied is not evidence for either answer. The rule-settled shape carries the shell's own reason on a second line when there is one and stands alone when there is not, because a blank line where a reason should be reads worse than no line at all. The attribution is read from own data properties only, so neither a polluted prototype nor a getter can make an automatic denial claim a person made it — and the getter case is pinned to never run at all. The transcript classification word is minted only on the two paths where the upstream transcript format really carries one; the three classifier words and the two abort words are left absent, with the abort words pinned against the strings this package actually normalises interruptions to, which are different strings. A final section pins the decide-operation observer: one `start` in the same tick as the first request and exactly one `end` after the last attempt has settled, across success, retried timeouts, exhausted transient failures, semantic refusal, binding mismatch and both kinds of caller abort, with nothing between retries; an observer that throws or rejects — even when logging that fault fails — never changes what is sent or returned, and the stream-level dependency reaches both durable park legs. A package-internal re-delivery of the same decision (plain approve after an older server rejects the session-scope flag, or a re-send without the attribution key) is reported as one operation with a single start and end. An operation handle only groups sends for the same session and bound call — a send for another gate through the same handle is its own operation — and closing a handle while a send is still in flight defers the end until that send settles. When a person picks "allow for this session" on a parked card and the grant is known not to have been stored — the server answers so, which newer servers do for the gates they can recognise from the parked row as needing a person each time, or the server refuses the session-wide grant with one of the refusal codes that are fixed by the row or the deployment and the package falls back to a plain approval — the parked path now says so with the same line the live path uses, and the receipt carries the server's bit for hosts that call that path directly; an answer without the bit, or any other failure — including a conflict that an internal retry can hit after an earlier send already stored the grant — is treated as unknown and says nothing, a plain approval that never asked for the grant says nothing, and a host logger that throws after a successful decision, on either card path or in the bridge's retry step, can no longer turn it into a second send or a failure. From 0.83.0 it also pins approval cards whose arguments are not the tool's real input: a live frame whose arguments were omitted over the size cap or never sent (the card holds at most a path recovered from the question), and a parked approval whose stored input is missing or replaced by the upstream size marker. On every such card an edited approval sends nothing at all — no respond, no decide, no session-grant attempt — ends as edit-refused, raises the same notice once and flags the card so hosts do not offer editing; both production entry points are driven end to end, and the parked path neither mistakes it for an already-decided gate nor reconnects. The explanatory line is pinned word for word in its four forms: a recovered path is the only thing it claims to have recovered, an unrecoverable one says nothing was recovered, neither claims a reconstructed diff, and the parked forms each say which of the two gaps it is, with near-miss shapes of the size marker still rendered as ordinary input. Real input on the stream, the frame or the record keeps edits flowing, and plain approvals and denials are untouched. |
|
|
354
355
|
| `scripts/run-park-hop-progress-test.mjs` | L-80: the park re-attach loop budgets **stalled** rounds, not parks. A turn where the model keeps hitting gates and every one of them is really decided (a card was answered, the engine really moved on) must never be cut off by the hop budget — the budget counts consecutive rounds that produced no progress, and "the engine revived and immediately parked again on the same coordinates" is not progress. The three non-progress arms (already-resolved, decide-transport-exhausted, and a re-scan that was adopted but led nowhere) share one same-cause limit instead of one arm having a limit and the others having none, and every non-progress re-attach is announced once through the host callback rather than only to the debug log. When the limit is spent the resolver reads the approval queue once more and puts whatever is decidable in front of the user before it gives up; only when there is genuinely nothing to show does it fail soft, and the terminal message then carries the real cause and a real way out instead of a sentence about a budget. On the self-heal side, a reopen verdict that reports `decidedWithoutCard` — the chain settled the gate by rule, so there was no card to present — is progress, not a reopen failure, and the user is not told their message was NOT sent. Negative control: a genuinely empty queue with a run that never moves still fails soft |
|
|
355
356
|
| `scripts/run-notif-fleet-honesty-test.mjs` | [2393] the five notification/fleet disciplines a green type-check cannot see, each proven by reverting the fix. (1) The workflow-side dedup `return` keeps a count and a trace — without it "suppressed by design" and "a real completion swallowed because the runId minting changed" are the same observation. (2) `seq` normalisation has exactly one mint point, so a 0-based or fractional wire `seq` cannot make the watcher lane and the frame lane key the same completion differently (which would feed the model twice). (3) The TTL sweep defers to a probe arm that is still inside its own deadline — an entry recorded as "abandoned" must not be delivered a moment later — while an arm that has outlived its deadline never blocks the sweep, so the headless exit gate keeps its liveness. (4) The reset hook really clears every ledger it claims to (the sticky `prompt` ledger leaked across cases). (5) The fleet ledger counts all three drop paths (malformed / unknown frame type / isolation drop), and the panel projection's settled recycling is anchored on the settle instant and skips still-present rows, so the dedup token is never carried off with the entry (which would re-emit `end`) |
|
|
356
357
|
| `scripts/run-public-surface-test.mjs` | The outward promises: the npm export surface baseline (an **exact set**, both directions — a new export that never entered the baseline is one nobody watched leave, and deleting it later would not be red), the peer floor witness, and this README's claims |
|
|
@@ -371,7 +372,7 @@ guard still cross-checks the table by name).
|
|
|
371
372
|
| `scripts/run-rules-side-test.mjs` | The persisted-permission-rules lane's shared decision half. The two capability bits are checked as **two independent gates** — a worker can honestly advertise the rules lane while predating the revoke routes, and that shape must *hide* the governance surface rather than render a dead entry. Failure classification is by **disposition, not cause**: the two 404s (route missing vs. dead ticket) never share a bucket, a 503 `rule_import_retry` means *the ticket is still alive* (the opposite handling of a dead one), and a stale-cursor 400 drops the cursor and re-lists from the top exactly once — never resuming a stale keyset, never surfacing a partial governance list, and never paging past the hard cap. The persist-ack reader is **merged into** `readToolApprovalRespondAck`: the three-state verdict (`persisted` / `refused` / `unknown`) is derived only from an ack that passed the package's structural narrowing, and a half-shaped object such as `{rulePersisted: true}` with no `delivery` reads as `unknown` — the pre-merge shell read would have said `persisted`, which is precisely the double-ledger drift this file closes, so that case is pinned in reverse. The local-allow-rule skeleton pins all five narrowings (whole-tool, tool-name match, literal anchor with the escaped-star counter-example, bare interpreter prefix consulted only for Bash, and the canonical dangerous-pattern overlay) **with their refusal strings byte-for-byte** — the cli's 128-assertion suite anchors the same strings, so a one-character edit here changes observable behaviour on three clients — and asserts the parse is a pure function of its input, because the same call backs both "render the option" and "resolve the selected value" |
|
|
372
373
|
| `scripts/run-park-decision-layer-test.mjs` | The decision layer behind the "stuck behind a card" family, shared by every client. A pending row that is **not in the queue** is three states, not one: a bounded, interruptible re-probe loop distinguishes *a decidable row*, *not born yet* (no positive evidence that anything settled — an empty queue proves nothing) and *settled elsewhere*, always probes at least once so a zero budget keeps the pre-fix semantics verbatim, cuts a hung read face off at the window rather than only noticing afterwards, and reports the honest failure when the window is spent instead of inventing a decision. The decision-note reader is likewise three-state: an explicit `noteRecorded: false` outranks an echoed note body, absence renders **no line at all**, and untrusted note text is flattened and bounded before it ever reaches a renderer. Row routing anchors on the deciding quantity — a row carrying `gateKind: "human"` with `toolName: "Write"` is a tool gate, because `human` is the engine's *generic* "someone must decide", not a synonym for a question — and the queue scan refuses to surface a row it cannot positively prove belongs to this session. A chain that fails after the row vanished is split by whether a card was ever presented: decided-elsewhere, or not-its-turn-yet. A row-level single-flight makes "at most one card per pending item" structural rather than incidental. The resume three-way card pins the option **order** (the zero-effect choice sits at index 0, because the frame carries no default-focus field and a stray Enter must not attach or cancel), renders only options the wired verbs can honour, collapses every ambiguous answer to zero action, omits the liveness line entirely when the engine gave no evidence, and — when there is no card lane at all — prints three real routes and exits on a dedicated code rather than reporting success |
|
|
373
374
|
| `scripts/run-selfheal-reopen-test.mjs` | The 409 active-run self-heal decision chain: `governanceForced` narrows on strict `true` only; triage prefers the wire's `pendingGate.kind` and falls back to the status table (an off-table kind is never guessed into a card arm — hands-off plus the honest wording); a first-sight card makes zero closed/reopened claims and a host presentation receipt of `presented: false` demotes the outcome to reopen-failed; park-row ownership is a fail-closed positive proof (own-run ledger or session id — unprovable is not owned); the three gate-identity key literals live in exactly one mint (`hitl/gateIdentity.ts`, AST string-token scan); the armed-gate presentation ledger is per-session; and the `plan_review` reopen arm shares the arm arm's card body, three-state verdict and delivery pipe, consuming the presentation history once a decision is delivered. The same chain also carries the `running` three-way card: both plan-family gate kinds route to the plan arm and all four ask-family kinds to the ask arm (an off-table kind still never gets guessed into either); the card is offered only for verbs that can actually be honoured and a missing presenter means zero action rather than a silent cancel; a steer is sent **exactly once** with its three delivery outcomes worded apart (a `queued` receipt is the wire correcting the triage input, so the named park word decides which card gets reopened, and an unrecognised park word drives neither arm), and a steer failure is split into *provably not delivered* (4xx) and *delivery unknown*, because telling a user to resend a non-idempotent instruction that may already have landed is how duplicates get made. After a user-chosen cancel, "the session is free" is asserted only from a whitelist of terminal states — park states hold the claim, an unrecognised state word is not a release, a failed read is *unknown* rather than a release, and only a 404 counts as one — and the honest timeout line quotes how long it really waited. The two "card could not be reopened" rows can carry a host-declared way to keep the conversation, which says the card comes back on resume only if it is still waiting: it is placed before the route that abandons it, never offered for an injected submission, while a decision is still on its way, or once the pending approval has been proven gone (the outcome then carries a flag saying so; the proof only counts before the cleanup card is shown, so a fallback after the card carries no flag unless a fresh read finds the run finished, and a recheck that finds the approval back clears it), the host function is not even called in those cases, and it is treated as unavailable when it throws or returns an empty value; a host can also switch off the engine decide route on the interactive rows while the cancel route stays, and with neither given all four rows are pinned byte-for-byte to the text the previous release produced. |
|
|
374
|
-
| `scripts/run-terminal-identity-copy-test.mjs` | Terminal-state **identity**, in both lanes where a stop gets a name. A run stopped by this deployment's own governance knobs — the open-set `limits.*` family, `output.invalid`, and the `blocked` contract terminal a ReportBlocked agent produces — is not a provider failure, and labelling it `API Error:` sends the reader to check the network, the key and the quota when the handle is the `--max-turns` they passed themselves. Those terminals now render a neutral row; the reverse direction is guarded just as hard, because asserting "this is *not* an API error" on a code the package does not recognise is the same misfiling pointed the other way — a real `gateway HTTP 502`, a `conflict.session_active_run` and any unknown code all keep the `API Error:` prefix, and the row keeps its `isApiErrorMessage` class flag so brief-mode visibility filtering does not silently drop it. The second half is who the rejected submission belonged to: the self-heal copy told every caller "Your message was NOT sent … send it again", which is three separate untruths for a system injection (a plan-review outcome, a cron wake-up, a task notification) — not the user's message, and not re-sendable, since a host queue marks those non-editable and non-recallable. The injected form says so instead, and the one sentence that promises re-delivery is pinned to the single disposition that earns it: `selfHealSubmissionDisposition` is the same function the host consults before putting the item back on its queue, so the promise and the behaviour cannot drift apart, and the arms where no card could be surfaced state plainly that nothing was delivered and nothing will retry. Since 0.72.6 the same gate pins the **follow intent** after a steer (): a message handed to a live run only pays off if someone tails that run's own event stream, so `steerFollowIntent` decides from the delivery word whether to tail now, after the pending decision, or only after a wake — and the "watch that run" sentence turns into a factual "sema is following that run" **only** when the host declares it attached that tail, so a shell that did not wire it can never claim it did |
|
|
375
|
+
| `scripts/run-terminal-identity-copy-test.mjs` | Terminal-state **identity**, in both lanes where a stop gets a name. A run stopped by this deployment's own governance knobs — the open-set `limits.*` family, `output.invalid`, and the `blocked` contract terminal a ReportBlocked agent produces — is not a provider failure, and labelling it `API Error:` sends the reader to check the network, the key and the quota when the handle is the `--max-turns` they passed themselves. Those terminals now render a neutral row; the reverse direction is guarded just as hard, because asserting "this is *not* an API error" on a code the package does not recognise is the same misfiling pointed the other way — a real `gateway HTTP 502`, a `conflict.session_active_run` and any unknown code all keep the `API Error:` prefix, and the row keeps its `isApiErrorMessage` class flag so brief-mode visibility filtering does not silently drop it. The second half is who the rejected submission belonged to: the self-heal copy told every caller "Your message was NOT sent … send it again", which is three separate untruths for a system injection (a plan-review outcome, a cron wake-up, a task notification) — not the user's message, and not re-sendable, since a host queue marks those non-editable and non-recallable. The injected form says so instead, and the one sentence that promises re-delivery is pinned to the single disposition that earns it: `selfHealSubmissionDisposition` is the same function the host consults before putting the item back on its queue, so the promise and the behaviour cannot drift apart, and the arms where no card could be surfaced state plainly that nothing was delivered and nothing will retry. Since 0.72.6 the same gate pins the **follow intent** after a steer (): a message handed to a live run only pays off if someone tails that run's own event stream, so `steerFollowIntent` decides from the delivery word whether to tail now, after the pending decision, or only after a wake — and the "watch that run" sentence ("watch that reply" on the rows about follow-up messages sema sent on its own) turns into a factual "sema is following that run" ("… that reply") **only** when the host declares it attached that tail, so a shell that did not wire it can never claim it did. Since 0.83.0 the rows about follow-up messages sema sent on its own carry no engine-internal words and say what happened per delivery shape, promising a resend or "nothing for you to do" only where the code guarantees it |
|
|
375
376
|
| `scripts/run-additive-key-passthrough-test.mjs` | The one disease shape behind two legs: a **closed whitelist / flattening arm** dropping a fact that is already on the wire, while both sides of the seam look correct. (1) The `task_progress` projection carries a registered **key ledger** — a frame populated with every key the service really projects is pushed through the shipped `eventToSdkMessage`, and the set of wire keys that survive must equal the registered pass-through list **name for name in both directions**, so quietly forwarding one more key is as red as quietly dropping one. `model` (the child run's model id, minted by core as `prepared.model.id` and projected by the server since 7.52.1) is the key this batch adds, with the same conditional the server itself applies: a non-empty string or no key at all — an empty string is neither a model id nor "unknown". The ledger is also checked against the fenced list in `docs/INTEGRATION-CLIENTS.md` §3d, so a doc that still says seven keys while the code forwards eight is red rather than merely stale. (2) The decide-failure arms carry the server's S-02 `currentPending` pointer key from a 409 `approval_stale` refusal onto the outcome the host reads. The reader is structural rather than `instanceof`, because the client is host-injected and the class identity is not this package's to assume; a half triple never mints (half a pointer cannot relocate anything), an empty string is not presence, and `checkpointToken` never transits. Both the allow and the deny leg are driven end to end through the real durable approval path — as is the accept-session leg, where a refusal carrying the pointer key must now re-raise instead of silently re-sending the human's answer for the **old** card as a plain approve (one decide call, pointer preserved), while a legacy 400 still falls back exactly as before — and all three flattening points must call the one shared reader — the same-shape residue check that makes "fixed one arm and left the twin" red instead of invisible. (3) The same disease growing on the REQUEST side: the `.mcp.json` → server-spec projection rebuilds each server key by key, and the settings schema deliberately leaves some keys parse-transparent — whatever JSON the file carries reaches the engine untouched, because validating them where the whole domain parses all-or-nothing would let one bad declaration take every server down silently. The whitelist had no row for the newest of them, so an operator's per-tool declarations — the ones the write fence reads — were stripped at the package boundary while both sides looked correct. The criterion is not "is that key handled" but the transparent-key table read out of the INSTALLED schema at runtime, reconciled name-for-name against this leg's ledger, so the day upstream adds a third one this turns red and forces an explicit decision. Behaviour is pinned on both transports, by object identity rather than deep equality (a rebuild would be a second judge), and malformed values must transit UNCHANGED rather than be refused here — the engine refuses them loudly and names the server, whereas a package-side judge can only swallow a declared protection quietly. Absence still mints no key, unknown keys still never reach the wire (the fix is the dropped key, not the gate), and the one transparent key this leg deliberately does not forward is a ledger entry with its own exit condition: it belongs to the deployment plane, and the day the request-plane type declares it the entry's premise is gone and the gate says so |
|
|
376
377
|
| `scripts/run-esc-halt-plan-test.mjs` | The Esc stop decision every client shares: fire the **turn-level** halt first, and escalate to a **run-level** cancel in exactly two cases — the engine itself answered with a 409 from the closed code set (it is saying "there is no in-flight turn here; use cancel for a run-level stop"), or that shot came back with no verdict at all *and* the shell can independently prove a permission card was on screen. Everything else does not escalate. The asymmetry is the whole point and every negative control guards the same direction — deciding *not* to escalate costs the user one more choice on a busy-session card (recoverable), deciding to escalate wrongly tears down a run that was alive and takes every in-flight tool with it (not). So: the closed code set is a **frozen** value, not a `ReadonlySet` — type-level immutability does not stop a consumer's `.add()`, and the guard proves it by really trying to mutate the exported value and then checking the verdict did not drift; the escalation gate is the **conjunction** of that closed set and the 409 status, since honouring the code alone lets a 500 that merely quotes it drive a destructive call; `interrupt.not_held` and `steering.not_running` are deliberately outside the set (the first means *this replica* has no live face — the run may be perfectly alive on another); an unreadable code falls to the no-escalation side; a `parked` flag never overrides a verdict the engine did give, and only strict `true` counts when it did not. The first shot is unconditional by construction — it does not consult `parked`, because the 409 it earns is exactly the verdict the gate wants — and the verdict itself is a closed machine-readable reason word, not display copy. A third escalating case was added once tearing the stream stopped reaping the run: with detach armed, a shot that never lands leaves the run going all the way to the end of the turn, so the Esc the user pressed has no effect at all and nothing on screen says so — the old behaviour had a silent backstop (tearing the stream ended the run) and that backstop is gone. The new fact is held to the same three disciplines as `parked`: it is read only where the engine gave no verdict, it is judged **after** `parked` so an existing host's reason word does not change under it, and only strict `true` counts. Absence is proven to be a no-op rather than asserted — the guard carries its own reference implementation of the previous version's table, runs the full grid through both, requires zero divergence when the new field is omitted, and first shows the comparison really does report a difference on the one cell where the two versions are meant to differ |
|
|
377
378
|
| `scripts/run-peer-frame-projection-test.mjs` | The three engine-injected lanes design/385 puts on the **one** `task_notification` carrier, which are not the same kind of thing at all: a delegated child's uplink (`agentMessage`), another session's message drained from this session's own box (`crossSessionMessage`), and a receipt about one of *this* session's own outbound messages (`crossSessionNotice`). The engine renders none of them inside a `<task-notification>` shell, so a client that projects them as the generic completion card shows "background task finished" while the model read a colleague's sentence — two faces describing different events. The discriminator is pinned to the **typed carrier being present**, never to the `summary` text: those carriers can only be minted by the engine's injection legs (the external `notify()` input is a strict subset of the payload and can wear none of them), while `summary` is filled by every notification there is — so anchoring on text would let any background task impersonate a colleague's message by writing `<agent-message from="…">` into its own summary, and a positive control asserts exactly that payload still projects as the generic card. Fail-closed has two tiers rather than one: a broken **required** field (empty `from`, a non-string `body`, a notice `kind` outside the closed set) returns absence so the caller falls back to the generic card — an honest downgrade where the user still sees the notification — while a broken **optional** field drops only itself, because losing an attribution note and losing a colleague's whole message are not the same magnitude. The provenance side record is **required and must agree on four points** (`kind` matches the lane; `from`/`taskId`/`seq` are present and equal the carrier/payload — each equality is anchored on a core mint site and pinned by the cli wire-anchor A-K24), so a carrier signed with a trusted name but a disagreeing provenance falls back to the generic card; peer bodies pass the same authority-envelope neutralization core applies (`<task-notification>` etc. are defused) so a colleague's text can never seed the resume dedup ledger. Lane precedence copies the engine renderer's own order, because the model already read the frame in that order and a client ordering of its own would put a card on screen that disagrees with the frame the model saw. Rendering and parsing of the transcript line live in the same module and are round-tripped in both directions, including a body carrying a forged closing tag (a parser fooled there hands half a message to the next row) and a quote inside the sender label (which must not forge a second attribute); the notice lane is deliberately kept **out** of the parser, since recognising it would mean anchoring the `[Cross-session …]` prefix and a user typing that same line would be rendered as engine speech. Hostile carriers are read as own **data** descriptors only and accessors are never invoked at all — `catch` catches throwing, not never returning — proven by a counting getter that must stay at zero calls, alongside a revoked proxy and a prototype-only carrier; and four legacy payload shapes assert the no-carrier path is byte-identical to before, which is the executable form of "zero difference for an older host". A re-supplied cross-session message — same task id, status and sequence as the first delivery, handed to the model again after compaction — is not rendered a second time, because the first delivery is still on the user's screen; a record with the next sequence number still renders |
|
|
@@ -420,7 +421,14 @@ guard still cross-checks the table by name).
|
|
|
420
421
|
| `scripts/run-file-history-capture-capability-test.mjs` | The engine's file-history-capture self-description (`capabilities.fileHistoryCapture`), read the same four-state way as its sibling capability readers: an absent key is reported as not reported (never folded into `off`), words are taken as an open set so a newer mode is not mistaken for a malformed answer, `fileHistoryCaptureMode` recognises only `off` and `on-always`, and the wording for `off` speaks about capture only — whether code can be rewound is left to the rewind readings. |
|
|
421
422
|
| `scripts/run-model-identity-resolvability-test.mjs` | The model-identity judgement a client makes before letting anyone in: can the engine it is about to use start with a model name? Each end reports what it read from each place that can feed a model name to a local engine (complete, partial — a gateway address or a credential but no model name —, absent, or unreadable), or, for an engine that runs elsewhere, whether that engine has been seen answering; `modelIdentityResolvability` answers resolvable, unresolvable or unknown. Having part of an upstream configuration is not having enough of one, so partial lanes never add up to resolvable; a lane that was not reported or could not be read makes the answer unknown rather than unresolvable; an engine that runs elsewhere is never judged unresolvable and local lanes are never consulted for it (it does not start without a model name, so seeing it answer is enough to call it resolvable). `modelSetupDecision` combines that answer with whether this end can configure a model at all: setup is offered only for unresolvable on an end that can configure one, an end that cannot says so and points at whoever runs the engine, and unknown never opens setup. The detail and notice sentences are checked to be pairwise distinct, unknown sentences neither claim a model is configured nor that it is not, and the module is checked to import no platform I/O |
|
|
422
423
|
| `scripts/run-cloud-effective-projection-test.mjs` | The cloud control plane's effective-configuration response beyond its four configuration domains, and what a locally started engine does with the models document derived from it. Three top-level keys are read with the meaning their producer gives them: `warnings` (degradation warnings from the build that produced the served view — an empty list is a clean build, an absent key is an older server that cannot tell), and `budget` / `runtimeCaps`, where `null` means two different things: nothing resolves for this principal when you view yourself, and values withheld when the response previews another principal. A missing key or a wrong type is a third state, unknown, and none of the three is ever folded into a zero, a `false` or "no budget". A malformed warning row, budget field or cap costs only itself, and a known budget field of the wrong type is named as not shown rather than silently read as "no limit on this axis"; warning kinds are an open set, so a kind this client does not recognise still produces a warning line. The budget and cap readers are reconciled against the installed settings schema. The single wording source puts degradation warnings first and keeps every "not set / withheld / unknown" sentence free of digits, while a zero the server really sent is shown as a zero. On the models side, when the host injects an entry check the models document carries the default model, the tier groups and the active tier group (without the check none of the three is written), and every catalog reference the local engine's schema would reject — a default, role, @-mention entry, tier binding or active group that names something outside the catalog served to this principal — is dropped and recorded, because one dangling reference makes the local engine discard the whole models domain and fall back to its environment catalog; this is proven by reading the produced document with the installed file store. An @-mention allowlist that would be pruned to empty is kept as sent, since an empty list means "everything may be mentioned"; that case is recorded, produces its own warning that a locally started engine will reject the cloud model settings and use its environment catalog instead, and the gate reads the document with the installed file store to confirm exactly that outcome, so the sentence turns red the day the local reader becomes lenient. A per-model budget in which no field could be read is never described as having no limits. Registry annotation keys on model entries (`origin`, `overridesTeam`) are removed before the models document is written: the local engine's schema does not accept them, and a configuration refresh would otherwise be rejected as a whole. A model entry the host-injected entry check rejects is left out of the document and references to it are dropped: the local engine drops such an entry at startup, but a refresh rejects the whole configuration over it, so the gate requires a clean read of the produced document; when no entry passes the check, the catalog is kept as sent and gets its own warning, which the gate proves by reading the document back. The check receives a copy, so it cannot alter what is written. Tier words outside the local schema's closed set are dropped and recorded as unsupported, and the package's tier word list is reconciled against the installed schema in both directions; an active tier group is judged against group names, never model names. |
|
|
423
|
-
| `scripts/run-websearch-verdict-test.mjs` | The per-request web-search configuration a host puts on the wire. Newer servers refuse the whole request when that section is malformed — and a missing or misspelled search provider now counts as malformed, because the section names where the searches go and an unknown destination is refused rather than silently swapped for the deployment's own backend. The old readers in this package dropped such a section without a word, which let the server swap destinations after all. The guard pins the new three-way verdict (absent, honoured, malformed with the field that is wrong) against the server's own judge, vector by vector, whenever that judge is available next to this package; it pins that a half-configured environment is malformed rather than ignored, that a malformed environment never falls back to the settings file (that would change the destination too), and that neither the sentence shown to the user nor the recorded reason repeats an endpoint, a key, or a search-parameter name or value — an unrecognised provider is never echoed either (the sentence lists the valid words instead), so a URL or key pasted into the wrong field does not come back out, even when it happens to be all letters. A host's key store is plugged in through a callback the package calls only after the provider has been recognised, so the precedence between environment and settings stays inside the package. |
|
|
424
|
+
| `scripts/run-websearch-verdict-test.mjs` | The per-request web-search configuration a host puts on the wire. Newer servers refuse the whole request when that section is malformed — and a missing or misspelled search provider now counts as malformed, because the section names where the searches go and an unknown destination is refused rather than silently swapped for the deployment's own backend. The old readers in this package dropped such a section without a word, which let the server swap destinations after all. The guard pins the new three-way verdict (absent, honoured, malformed with the field that is wrong) against the server's own judge, vector by vector, whenever that judge is available next to this package; it pins that a half-configured environment is malformed rather than ignored, that a malformed environment never falls back to the settings file (that would change the destination too), and that neither the sentence shown to the user nor the recorded reason repeats an endpoint, a key, or a search-parameter name or value — an unrecognised provider is never echoed either (the sentence lists the valid words instead), so a URL or key pasted into the wrong field does not come back out, even when it happens to be all letters. A host's key store is plugged in through a callback the package calls only after the provider has been recognised, so the precedence between environment and settings stays inside the package. Since 0.83.0 an endpoint that carries a user name or password is malformed as well (the server judges the same way from 7.101.0), the reason sentences match the server's own word for word, and the three older readers that dropped a misspelled provider are gone. |
|
|
425
|
+
| `scripts/run-hooks-merged-disable-projection-test.mjs` | The fourth governance leg of the hooks projection: `disableAllHooks` set in a non-managed settings source. The value that counts is the **merged** one, read from the host through the optional `SettingsPort.mergedDisableAllHooks()`, because a per-source approximation ("any source says true") reads user `true` with local `false` backwards — the merged value there is `false` and every source's hooks ship. When the merged value is `true`, only managed-settings hooks are sent to the engine: non-managed settings can switch off their own hooks, never the managed ones, and a managed `disableAllHooks` is still judged first and sends nothing at all. A full matrix over the four sources, each true, false or absent, is merged with the reference rule (later sources override earlier ones, managed settings last) and every cell's projection is asserted. The reader is the only authority: per-source values never second-guess it, and only a strict `true` counts. A host that does not implement it keeps the previous behaviour and gets exactly one warning per installed settings port, never one per request, and none on paths where the reader would not have been consulted; a reader that throws is treated as `true`, so managed hooks still ship. The session goal's Stop hook and the final-verification yield rule, which reads the projected hooks, follow the same verdict, and the trust gate and the three managed gates are evaluated before the reader is ever called. The last leg pins the member's declared shape in the built declarations: optional, no parameters, returning a boolean or `undefined`. |
|
|
426
|
+
| `scripts/run-memory-saved-projection-test.mjs` | Engine memory writes (a successful `Remember` tool call) moved off the transcript onto the additive `memory_saved` chrome event, driven through the real pipeline: zero transcript rows for the write (the transcript is byte-identical to the same frames with the write reported as not successful), exactly one event whose `notes` carry the note text verbatim (notes, not file paths) and whose key set is exactly kind / laneProof / id / notes; no event for a missing, empty or non-string note, a non-`true` `ok`, a tool error, a missing result or another tool name; two writes give two events in order with distinct ids; a sub-agent write rides the sub-agent lane and an empty parent id emits nothing rather than falling back to the main lane; the `id` is derived from the write's wire key (the tool-end event id, else the tool-start event id, else the call id; empty ids count as absent), so projecting the same wire events twice gives the same id, and it never collides with the tool result row of the same or another call; the event sits right after the tool result row; the arm is registered as required. |
|
|
427
|
+
| `scripts/run-result-frame-projection-test.mjs` | Result frames and the synthesized terminal rows. The CC key `terminal_reason` is minted on result frames only where it follows from what the engine reported: `completed` on success, `max_turns`, `budget_exhausted` and `structured_output_retry_exhausted` for the three matching engine codes, on both the done-frame path and the failed-event path. Every other outcome leaves the key absent as an own property rather than present with an undefined value: wall-clock and token-budget limits, the classifier denial limit, cancellation, unknown codes, blocked, paused, unreadable or missing terminal records, and the busy-session refusal. The public reader `terminalReasonForResult` shares the minting predicate and is checked to agree with the minted key on every frame the gate produces. Both the minted key and the reader derive the word from the frame's CC subtype (success with `is_error` strictly false, and the three limit subtypes), not from the error code, so a replayed row whose status is paused, blocked or unrecognised never carries a word that contradicts its subtype. The four words are checked against the mirrored CC union, and the minting file is checked to hold no hand-copied code literals. The renamed superset keys (`_sema_error_code`, `_sema_salvaged_result`, `_sema_model_degraded`, `_sema_selected_model`, and the row flag `_sema_api_error_message`) are driven through the real stream pipeline. Each must be present under its new name, the old name must be absent, and every frame the gate saw is swept for old names. The selected model appears on error envelopes whenever the terminal record carries it, and never on a failed event, which has no record. It stays separate from the provider-reported model name. The two in-package readers still work: the interactive result arm reads the salvaged text under its new name (and old-shape frames under the old one), and the print init gate treats the renamed flag as the run having ended. |
|
|
428
|
+
| `scripts/run-layering-shadow-export-test.mjs` | Same-name shadows across the first-party clients that consume this package (terminal, desktop, web and the admin console). Each client's product sources are read at the local clone's `origin/main` (or its HEAD when there is no such ref), without fetching, and parsed with the TypeScript parser; every top-level runtime export the client declares itself is compared with this package's public runtime exports. The guard prints which ref, commit and commit date it read for each client, and warns (without failing) when that commit is more than seven days old, because the result then only describes that older snapshot. A client-side declaration carrying the name of a package export means a piece of shared logic now lives in two places and can drift apart. It fails the guard unless it is listed in `scripts/layering-shadow-exemptions.json`, and a listed row must carry a retire-by version no more than three minor lines ahead (it fails once the package reaches it). It also fails once the client has removed the shadow and the row still stands. Re-exports of this package's own exports are the intended form and never count. A client tree that is not present is reported as a skipped section, not as a pass. The ruler proves itself on an in-memory fake client (planted shadows must be caught, legal forms must not), on a throwaway repository (a missing `origin/main` falls back to HEAD, a broken one is a fault rather than a silent fallback), and refuses to report zero on a client whose scan surface is empty. |
|
|
429
|
+
| `scripts/run-session-policy-deliverable-test.mjs` | Which of a batch of user-written permission rules can be written into a session’s own rule record without changing their meaning, and why each of the others cannot. The record holds whole tool names and command names only, so exactly one class maps across losslessly: a deny rule that names one tool with no qualifier. Everything else is withheld with one word from a closed five-word list — an ask rule (the record has no ask tier), a deny rule with a parenthesised qualifier (recording just the name could block more), a rule covering every tool of one server or agent peer (for every protocol namespace the engine knows, checked against the engine package's own table) or containing a wildcard (*) anywhere (an engine that compares exact names would block nothing), and an entry that is not a tool name — and each word has one sentence, which never echoes the rule itself; asking for the sentence never throws, even with a value that throws when turned into a string. A name with leading or trailing whitespace counts as not a tool name: the record compares exact bytes, so it would block nothing. The guard pins the batch semantics: the deliverable part is either the whole batch or empty, never a subset, so a caller cannot send half a change and report it as saved. It also checks that malformed input never throws and never delivers anything (non-arrays, non-string entries, holes, a polluted array prototype, a length or index that throws, a changing index read once), that a batch which cannot be read at all is marked `unreadable: true` while an empty batch is not, so the two stay tellable apart, that each word is produced by some vector and nothing outside the list is produced, and — when a checkout of the previous in-client implementation is present — that this function gives the same answer on every recorded vector and on tens of thousands of generated rules and pairs, except for three deliberately stricter classes (whitespace-padded names; rules with a wildcard anywhere, which the previous implementation sent as exact names unless the wildcard was the whole tool part of a server rule; and peer-wide rules outside the MCP namespace, which it did not recognise), whose disagreements are counted per class and must match an independent count exactly. |
|
|
430
|
+
| `scripts/run-plugin-hooks-projection-test.mjs` | Plugin hooks: each command hook an enabled plugin declares is decided one by one as running in the engine, running in this client, or not running at all, and the page of hooks sent with a request is built from the same per-turn plan the client uses to skip its own copies, so one hook never runs in two places. Governance is judged first and always wins — a managed hooks switch-off, an untrusted workspace, safe or bare mode, or a governance read that fails sends no plugin hook and does not list it as a gap; managed-hooks-only (set directly, or through a merged non-managed hooks switch-off) keeps only managed plugins; the plugin-only customization lock does not touch plugin hooks. A hook reaches the engine only when this client started the engine on this machine, the engine reports plugin-hook support, the entry is a command, the plugin declares no sensitive option, and the event still fits the engine's per-event limits; the gate walks that matrix cell by cell, including the limit boundaries and a session goal hook counting toward them. A fact that was never read is reported as not known rather than as a fact: a host that does not say where the engine runs gets a "not known whether this client started the engine" reason, an engine whose capabilities have not been read yet gets a "not known yet whether it supports plugin hooks" reason, and the plan's two engine facts are null in those cases, not false. Events the engine never fires run only if the client says it fires them itself, and hooks the upstream behaviour itself refuses (option references in a shell-form command, an unset option in exec form, malformed entries) run nowhere. Exec-form arguments are passed element by element with only saved non-sensitive option references filled in; path placeholders are left for the executor. Sensitive option values never reach the request: with a host that wrongly supplies one, every string in the plan, the request body, the notice, the labels and the log is searched for it across eight cells. A host without the plugin reader keeps the previous request body and gets exactly one warning per settings port; plugin data that throws while it is being read (a throwing getter, a revoked proxy) is treated like a failing reader — no plugin hooks this turn, settings hooks still sent, nothing thrown; the not-running notice names the plugin and events, never a command or an option value, and escapes control characters in names. Command hooks from settings that carry arguments (a non-empty `args` array, which is the exec form, or any other non-null value) are removed from the request until the engine reports support for arguments, because the engine would otherwise drop the arguments and run the bare command through a shell; an empty `args` array is not treated as carrying arguments when the command is made only of letters, digits and `_ . / : + -` (the shell runs the same executable), so such a guard still reaches the engine, while an empty array on a command with spaces or shell characters is removed; `args` on a prompt or http entry, a null `args`, or an entry with no type is left alone, and those go out unchanged. MCP tool hooks, which the engine cannot parse, are removed only from a request built from a plan, whose not-running notice the host shows; a request built without a plan still carries them on engine-fired events, so the engine rejects the whole request loudly instead of a guard hook silently not running — the gate checks both request bodies against the engine's own hooks schema. Without a plan, every removed hook of that kind on an engine-fired event produces one warning per settings port, event and reason. Malformed entries still pass through for the engine to reject loudly, and passing null where the options object goes behaves like passing nothing; a `plan` option that is not a plan is ignored rather than turning the whole page into nothing, and a plan passed directly in place of the options object is recognised and used. A `plugin` key written by hand on a settings hook is stripped before sending (even when its value is undefined), because only hooks that come from the plugin reader may carry plugin context; the settings document itself is left untouched and a debug line records the count. The two hand-copied tables, the engine-fired event list and the engine limits, are checked against their owners. |
|
|
431
|
+
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. |
|
|
424
432
|
|
|
425
433
|
Each suite carries a floor that only moves up — a refactor that stops executing a group of
|
|
426
434
|
assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
|
package/dist/adapt/arms.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ import type { PanelTaskLedger } from './panelTasks.js';
|
|
|
5
5
|
import type { TextStream } from './textStream.js';
|
|
6
6
|
import type { ToolCardLedger } from './toolCards.js';
|
|
7
7
|
import type { TurnFlags } from './turnFlags.js';
|
|
8
|
-
|
|
8
|
+
interface ProjectionArmDeps {
|
|
9
9
|
readonly ctx: AdapterContext;
|
|
10
10
|
readonly idOf: IdOf;
|
|
11
11
|
}
|
|
@@ -17,5 +17,6 @@ export interface ArmDeps extends ProjectionArmDeps {
|
|
|
17
17
|
readonly inst: AdapterInstanceLedger;
|
|
18
18
|
}
|
|
19
19
|
export type ProjectionArmFn = (m: Frame, deps: ProjectionArmDeps) => Generator<AdapterOutput>;
|
|
20
|
-
|
|
20
|
+
type ArmFn = (m: Frame, deps: ArmDeps) => Generator<AdapterOutput>;
|
|
21
21
|
export declare const ARMS: ReadonlyMap<string, ArmFn>;
|
|
22
|
+
export {};
|
package/dist/adapt/arms.js
CHANGED
|
@@ -119,6 +119,7 @@ const assistantArm = function* (m, { ctx, idOf, text, cards, inst }) {
|
|
|
119
119
|
uuid: idOf(m),
|
|
120
120
|
session_id: ctx.sessionId,
|
|
121
121
|
parent_tool_use_id: null,
|
|
122
|
+
...(m._sema_api_error_message === true ? { _sema_api_error_message: true } : {}),
|
|
122
123
|
}, ctx.now());
|
|
123
124
|
text.markEmittedText(textBlock.text);
|
|
124
125
|
}
|
|
@@ -140,7 +141,7 @@ const userArm = function* (m, { ctx, idOf }) {
|
|
|
140
141
|
};
|
|
141
142
|
const systemArm = function* (m, { ctx, idOf }) {
|
|
142
143
|
yield transcript(m, ctx.now());
|
|
143
|
-
const attachedFiles = m.attachedFiles;
|
|
144
|
+
const attachedFiles = m._sema_attached_files ?? m.attachedFiles;
|
|
144
145
|
if (Array.isArray(attachedFiles)) {
|
|
145
146
|
let i = 0;
|
|
146
147
|
for (const f of attachedFiles) {
|
|
@@ -491,7 +492,7 @@ const promptSuggestionsArm = function* (m) {
|
|
|
491
492
|
yield chrome({ kind: 'prompt_suggestions', laneProof: mainLane(), suggestions: list });
|
|
492
493
|
}
|
|
493
494
|
};
|
|
494
|
-
const toolEndResultArm = function* (m, {
|
|
495
|
+
const toolEndResultArm = function* (m, { idOf, cards, panel }) {
|
|
495
496
|
const callId = typeof m.toolCallId === 'string' ? m.toolCallId : undefined;
|
|
496
497
|
recordEngineToolLabel(callId, m.label);
|
|
497
498
|
if (callId !== undefined) {
|
|
@@ -526,14 +527,18 @@ const toolEndResultArm = function* (m, { ctx, idOf, cards, panel }) {
|
|
|
526
527
|
: undefined;
|
|
527
528
|
if (p.name === 'Remember' && m.isError !== true) {
|
|
528
529
|
if (s && s.ok === true && typeof s.note === 'string' && s.note.length > 0) {
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
530
|
+
const memParent = closeParent ?? p.parentToolCallId;
|
|
531
|
+
if (memParent !== '') {
|
|
532
|
+
yield chrome({
|
|
533
|
+
kind: 'memory_saved',
|
|
534
|
+
laneProof: memParent === undefined ? mainLane() : { lane: 'subagent', parentToolCallId: memParent },
|
|
535
|
+
id: (() => {
|
|
536
|
+
const key = [closeEventId, p.eventId].find((v) => typeof v === 'string' && v.length > 0);
|
|
537
|
+
return idOf(key !== undefined ? { id: key } : { toolCallId: p.id }, 'memory-saved');
|
|
538
|
+
})(),
|
|
539
|
+
notes: [s.note],
|
|
540
|
+
});
|
|
541
|
+
}
|
|
537
542
|
}
|
|
538
543
|
}
|
|
539
544
|
if (s && (s.type === 'task' || s.type === 'task-list')) {
|
|
@@ -692,7 +697,8 @@ const resultArm = function* (m, { ctx, idOf, text, cards, panel, flags }) {
|
|
|
692
697
|
return lowerBound && ot === 0 ? undefined : ot;
|
|
693
698
|
});
|
|
694
699
|
yield* text.takeAnswerSegment();
|
|
695
|
-
const
|
|
700
|
+
const salvaged = m._sema_salvaged_result;
|
|
701
|
+
const terminalText = typeof salvaged === 'string' ? salvaged : typeof m.result === 'string' ? m.result : '';
|
|
696
702
|
const committed = text.lastCommittedAnswerText;
|
|
697
703
|
const terminalIdentity = messageIdentityOf(m, ctx);
|
|
698
704
|
const emitTerminal = function* (body, tag) {
|
package/dist/adapt/ids.js
CHANGED
|
@@ -14,12 +14,16 @@ export function messageIdentityOf(frame, ctx) {
|
|
|
14
14
|
};
|
|
15
15
|
}
|
|
16
16
|
export const chrome = (event) => ({ plane: 'chrome', event });
|
|
17
|
-
export const transcript = (message, nowMs) =>
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
:
|
|
22
|
-
|
|
17
|
+
export const transcript = (message, nowMs) => {
|
|
18
|
+
const arm = message.type;
|
|
19
|
+
const stampsArrival = arm === 'user' || arm === 'assistant';
|
|
20
|
+
return {
|
|
21
|
+
plane: 'transcript',
|
|
22
|
+
message: (!stampsArrival || typeof message.timestamp === 'string'
|
|
23
|
+
? message
|
|
24
|
+
: { ...message, timestamp: new Date(nowMs).toISOString() }),
|
|
25
|
+
};
|
|
26
|
+
};
|
|
23
27
|
export function makeIdOf(ctx) {
|
|
24
28
|
return (frame, suffix) => {
|
|
25
29
|
const base = typeof frame.id === 'string' && frame.id.length > 0
|
|
@@ -28,9 +32,9 @@ export function makeIdOf(ctx) {
|
|
|
28
32
|
? frame.uuid
|
|
29
33
|
: undefined;
|
|
30
34
|
return deriveTranscriptId({
|
|
31
|
-
...(base !== undefined ? { id:
|
|
35
|
+
...(base !== undefined ? { id: base } : {}),
|
|
32
36
|
...(typeof frame.seq === 'number' ? { seq: frame.seq } : {}),
|
|
33
37
|
...(typeof frame.toolCallId === 'string' ? { toolCallId: frame.toolCallId } : {}),
|
|
34
|
-
}, ctx);
|
|
38
|
+
}, ctx, suffix);
|
|
35
39
|
};
|
|
36
40
|
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import type { AdapterContext, AdapterOutput, TextSegmentCommittedRow } from '../seam.js';
|
|
2
2
|
import { type Frame, type IdOf } from './ids.js';
|
|
3
3
|
export declare const SEMA_SEGMENT_ID_KEY = "_sema_segment_id";
|
|
4
|
-
|
|
4
|
+
interface TextSegmentReplacement {
|
|
5
5
|
diverged: boolean;
|
|
6
6
|
committedUuids?: readonly string[];
|
|
7
7
|
committedUuid?: string;
|
|
@@ -9,7 +9,7 @@ export interface TextSegmentReplacement {
|
|
|
9
9
|
committedPrefixLen: number;
|
|
10
10
|
committedPrefixDiverged: boolean;
|
|
11
11
|
}
|
|
12
|
-
|
|
12
|
+
interface ThinkingSegmentReplacement {
|
|
13
13
|
diverged: boolean;
|
|
14
14
|
committedPrefixLen: number;
|
|
15
15
|
committedPrefixDiverged: boolean;
|
|
@@ -40,3 +40,4 @@ export interface TextStream {
|
|
|
40
40
|
readonly answerLength: number;
|
|
41
41
|
}
|
|
42
42
|
export declare function createTextStream(ctx: AdapterContext, idOf: IdOf): TextStream;
|
|
43
|
+
export {};
|