@sema-agent/client-core 0.83.0 → 0.83.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +83 -0
- package/README.md +10 -5
- package/dist/adapt/arms.js +1 -0
- package/dist/adapter/downstream/terminalToSdkResult.js +3 -3
- package/dist/adapter/runStream.js +19 -2
- package/dist/displayUntrusted.d.ts +24 -0
- package/dist/displayUntrusted.js +1437 -0
- package/dist/engineAgentAbsence.d.ts +52 -0
- package/dist/engineAgentAbsence.js +131 -0
- package/dist/engineNoticeCodes.js +4 -0
- package/dist/fleetTaskDesc.js +3 -3
- package/dist/gateVocabulary.d.ts +4 -0
- package/dist/gateVocabulary.js +29 -1
- package/dist/hitl/approvalDecisionNoteAudit.js +4 -2
- package/dist/hitl/askSurvivesPosture.d.ts +15 -0
- package/dist/hitl/askSurvivesPosture.js +31 -0
- package/dist/hitl/sessionPolicyDeliverable.d.ts +14 -0
- package/dist/hitl/sessionPolicyDeliverable.js +80 -0
- package/dist/hitl/toolApprovalWire.d.ts +13 -5
- package/dist/hitl/toolApprovalWire.js +16 -2
- package/dist/hooksWireCaps.d.ts +6 -1
- package/dist/hooksWireCaps.js +194 -13
- package/dist/host.d.ts +24 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +5 -0
- package/dist/ownKey.d.ts +1 -0
- package/dist/ownKey.js +6 -0
- package/dist/peerFrames.js +2 -1
- package/dist/pluginHooksWire.d.ts +92 -0
- package/dist/pluginHooksWire.js +429 -0
- package/dist/subagent/engineSubagentResume.js +2 -1
- package/docs/INTEGRATION-CLIENTS.md +552 -13
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -49,6 +49,89 @@
|
|
|
49
49
|
> 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
|
|
50
50
|
> commit 漏转在发布前就红,不再靠人记。
|
|
51
51
|
|
|
52
|
+
## 0.83.2(2026-09-26)
|
|
53
|
+
|
|
54
|
+
> 主题:patch —— 只增公面,两组公共判定归包。① 审批卡一族三件:引擎 7.32.0 提货,卡上认「强制位站在哪一个词上」(`mandate`,六词闭集)并给读口与每词一句说明,通告码册 +2(CC-195);停泊审批行重开的卡带上出身词 `origin`(CC-196);常驻姿态(bypass 一类)下「哪一只 ask 不许被姿态替人放行」的三端单一谓词 `askSurvivesPosture`(CC-193)。② 后台子代「缺席行」的判定、计数、sweep 豁免、回收与措辞收成一组纯判定(CC-194)。根公面运行期导出 1254 → **1273**(+19),公面类型 +13,`ApprovalCardRequest` / `ToolApprovalFrame` 各 +1 可选键;开发依赖引擎 `~7.32.0`;peer sdk 地板 `>=11.3.0` 不动;零 wire 投影臂。
|
|
55
|
+
|
|
56
|
+
### Added
|
|
57
|
+
|
|
58
|
+
- **`askSurvivesPosture(card, facts?)`**(CC-193):会话处在会替人答审批卡的常驻模式(bypass 一类)时,这一张卡能不能由姿态答掉。读两路事实,给三态 `{ kind: 'must_ask', by: 'origin' | 'settings_ask_rule' }` / `{ kind: 'posture_may_answer' }` / `{ kind: 'unknown' }`(返回值冻结):① 卡上的出身词 —— `content_question`(工具自己的提问)/ `org_rule`(组织规则)/ `hook` / `ask_rule`(显式 ask 规则)/ `org_unavailable` / `rule_store_unavailable`(组织治理源 / 规则店读不出)/ `denial_limit_fallback`(分类器连拒到上限后交给人)七个词必须问(后三个是引擎要求只能由真人答的出身);② 宿主对自己 settings 里 ask 规则的匹配读数 `facts.settingsAskRuleMatched` —— `true`(正向命中)必须问,`false`(判过、没命中;命令拆不全或判不出也报 `false`),不报 = 没报。只有宿主报了 `false`、且出身词缺席或是本版认得的其余四个词之一时才答 `posture_may_answer`;宿主没报时,除七个必问词外一律答 `unknown`(settings ask 规则那一路命没命中不知道 —— 出身词只说谁提的问),本版不认识的出身词也答 `unknown`。`unknown` **不是放行**:宿主按自己的 settings 规则判。JSON 形入参不抛;带会抛的取值器或 Proxy 陷阱的对象会把异常原样抛出。它**不读** `governanceForced` / `requiresRealApproval` / `mandated` / `mandate` —— 那几位是更强的另一类事实,宿主既有的检查照旧排在它前面。型 `AskPostureVerdict` / `AskPostureFacts` / `AskPostureSurvivalRoad`。接入文档 **§102a S-1**。
|
|
59
|
+
- **强制位站在哪一个词上:`mandate`**(CC-195,引擎 7.32.0):`ApprovalCardRequest` 与 `ToolApprovalFrame` 各 +1 可选键 `mandate`;闭集 `APPROVAL_MANDATE_WORDS` 六词(`operator_always` / `tool_marks` / `probe_mandate` / `probe_unanswered` / `write_protection` / `write_protection_unresolved`,型 `ApprovalMandateWord`),只在那个词就是强制位的全部理由时在场;活卡帧与停泊审批行两条腿经同一把窄读上卡,只认严格在场的六词之一,表外 / 坏形 / 缺席 ⇒ 卡上没有这一位、不编词。读口 `readApprovalMandate(card)`(缺席答缺席)、成员判据 `isApprovalMandateWord(v)`、每词一句「为什么这张卡必须问」`approvalMandateDetail(word)`(六句互异,不指人去写规则,不说每次都问,不说只有真人能答)。词在场 ⇒ `approvalIsMandated` 答 `true`(见 Changed);卡上的 `mandated` 键仍只透传 wire 原位。🔴 服务端把这一位带上卡之前,卡上恒缺席。接入文档 **§102a M-1–M-3**。
|
|
60
|
+
- **通告码册 +2**(CC-195,引擎 7.32.0):`approval.read_root_granted` / `approval.read_root_grant_rejected`(审批人答复里的读根授权生效 / 没生效;audience 均为 `user`),位置紧跟 `task.interrupt_unconsumed`,七十一码 ⇒ 七十三码。按码册分发的端,这两枚从「码册外 ⇒ 只落调试」变成通用通告行。带授权批准的回答口本版不做(候服务端的能力位与决断体形)。接入文档 **§102a N-1**。
|
|
61
|
+
- **后台子代「缺席行」的判定与措辞归包**(CC-194;[8373] / [8376]):引擎不再上报一只后台子代、又没有终态到达时,本包经缺席订阅口发一条缺席事件(0.72.13 起);收到之后怎么判 —— 这一行算不算缺席、算不算在跑、turn 末 sweep 结不结它、什么时候能删、删的时候说什么 —— 此前各端各写一份。本版收成一组纯判定,三端共读:
|
|
62
|
+
- `isEngineAgentAbsentRow(row)`:判缺席标在不在,不看 status(缺席行的 status 恒是最后已知态);
|
|
63
|
+
- `engineAgentAbsenceMarkOf(row, ev)`:只在 `running` 行上、只在首次缺席时落标(第二条事件不许把停表点往后推;真终态先到的行丢弃缺席事件);`engineAgentAbsenceClockStopMs(row)`:计时停在最后一次见到它的时刻;
|
|
64
|
+
- `engineAgentPanelEventClearsAbsence(ev, current?)`:行回来(`tick` / `fleet-row`)或当代真终态(`end`)撤标;上一周期迟到的 `end` 与 `sweep` 不撤;
|
|
65
|
+
- `engineAgentRowCountsAsRunning(row)` / `tallyEngineAgentRows(rows)`:在跑数排除缺席行,缺席行单独一格,不进 completed / failed / stopped 任一格;
|
|
66
|
+
- `engineAgentTurnSweepSettles(taskId, row)`:turn 末 sweep 不结缺席行、常驻行与非 running 行;
|
|
67
|
+
- `reapEngineAgentAbsentRows(rows, { nowMs, ttlMs?, owned? })` / `engineAgentAbsenceTtlDueAtMs(row, ttlMs?)`:grace 窗不回收;自最后一次见到它起满 30 分钟(`ENGINE_AGENT_ABSENCE_TTL_MS`)**只是去看后台代理登记读数的时刻** —— 每行的 `registry` 读数(`ENGINE_AGENT_REGISTRY_READINGS`:`running` / `ended` / `not_listed` / `unknown`)为 `ended` 或 `not_listed` 才回收,以回收事实(`taskId` / `lastSeenAtMs` / `absentForMs` / `dueAtMs` / `registry`;`absentForMs` 是缺席事件检出时的读数,约 30 秒,不是回收时已缺席多久 —— 要后者用 `nowMs − lastSeenAtMs`)交给宿主删行并上一行说明;`running`、`unknown`、没给读数或给了认不出的词一律不删,也不排复查(下一拍同判,不起常驻轮询);有人正看着(`viewing`)且登记确认离场的行单列在 `heldByViewer`。按会话读后台代理登记的读口尚未提供,在它到来之前宿主只能给 `unknown`,因此**缺席行一行都不删**;
|
|
68
|
+
- 措辞单源:`ENGINE_AGENT_ABSENT_ROW_TEXT`(行上那一句,`engine no longer reports this agent · it may still be running`)、`engineAgentAbsenceDroppedLine(label, ttlMs?, removeReason?)`(回收那一句;引擎给过离场原词就原样说出并声明它不是结局)、`engineAgentTerminalAfterDropLine(label, status, hasReport)`(回收之后真终态才到的那一句);三句都不替缺席行断言结局或「已结束」;展示名、离场原词与回收后终态句的状态词呈前消毒。
|
|
69
|
+
- 型:`EngineAgentAbsenceMark` / `EngineAgentAbsenceRowFacts`(三位:`status` / `absence?` / `viewing?` —— 「有人正看着」,不是服务端后台登记行上的 `retained`)/ `EngineAgentRegistryReading` / `EngineAgentAbsenceReapRow`(三位:`absence?` / `viewing?` / `registry?`)/ `EngineAgentRowTally` / `EngineAgentAbsenceReclaim` / `EngineAgentAbsenceReap` / `EngineAgentAbsenceReapOptions`。根公面 +14。接入文档 **§102**。
|
|
70
|
+
|
|
71
|
+
### Changed
|
|
72
|
+
|
|
73
|
+
- **停泊审批行重开的卡带上出身词 `origin`**(CC-196):`surfaceFsApprovalAndDecide` 从待决列表行重铸卡请求时,此前逐键透传了 `requiresRealApproval` / `mandated` / `hasBidiControls`,却丢了行顶层的 `origin`;现在与活卡帧腿逐字同形:非空串才上卡、原文字节、词表不校、缺席不编词(绝不折成 `policy`)。两腿与 `askSurvivesPosture` 共用一把读法,只认自有键 —— 活卡帧腿此前按原型链读,JSON 帧零变化。🔴 今天的服务端待决列表行还不带这一位,本处在行上出现时即生效、端零改动。接入文档 **§102a O-1**。
|
|
74
|
+
- **接入文档改口:缺席行那一句与回收时机**:§59 S-1 / §60 S-5 让端渲的「engine no longer reports this agent · outcome unknown」改为本版常量 `ENGINE_AGENT_ABSENT_ROW_TEXT`(后半句 `it may still be running`)—— 缺席行在登记确认离场之前不会被删、可能在屏上挂很久,而引擎不再上报的最常见原因恰恰是它还在跑,「结局不知道」容易被读成「大概已经结束了」。§60 S-5「置可回收标(等价 `notified` + `evictAfter = now + grace`)」作废:grace 窗不回收,回收以 §102 为准。零代码行为变化(本包此前不回收任何行)。
|
|
75
|
+
- **`approvalIsMandated` 也认 `mandate` 词**(CC-195):闭集内的 `mandate` 词在场 ⇒ 答 `true`。上游契约是「词在场 ⇔ 那个词就是强制位的全部理由」⇒ 有词必有位;只有词没有位是矛盾形,按强制读(收紧方向:多问人)。`{ mandated: true }` / `{ ruleOffersAbsence: 'mandated' }` / `{}` 三形答案与 0.83.1 相同,`mandatedApprovalDetail()` 逐字不变;今天服务端不带 `mandate`,零变化。接入文档 **§102a M-1 / 102a′**。 入参型改为具名导出型 `ApprovalMandatedFacts`(`mandated?` / `ruleOffersAbsence?` / `mandate?` 三位都可缺;只带 `mandate` 的最小形在 TypeScript 下也能直接传)。
|
|
76
|
+
|
|
77
|
+
### Gates
|
|
78
|
+
|
|
79
|
+
- 新增常驻门 `run-ask-survives-posture-test.mjs`:十一个出身词逐词对独立期望表(宿主没报 / 报了没命中各一遍;期望表词集与 `ASK_ORIGIN_WORDS` 双向等值 —— 上游加词、本包跟上之后当天红,逼一次归类)、两路组合表、坏形、射程显形、三条真卡口(活卡帧 / 停泊行 / 悬挂 ask)交出去的卡请求端到端、返回形闭集冻结。
|
|
80
|
+
- 扩门:`run-durable-card-display-keys-test.mjs`(停泊腿的 `origin` 与 `mandate`;原型链上的 `origin` 不算;只有词 ⇒ `approvalIsMandated` 答 `true`)、`run-hitl-gate-honesty-test.mjs`(活卡帧腿的 `mandate` 与判据口认词;原型链上的 `origin` 不上卡)、`run-gate-vocabulary-test.mjs`(第六张表:对实装引擎真字节逐词逐序等值、闭集、读口、六句措辞门、编译期双向钉见证、出包面 —— dist 全树 `.js` / `.d.ts` —— 零引擎引用,附植入负控证明扫描会响)、`run-approval-frame-keys-test.mjs`(`mandate` 领先 SDK 锚与服务端发布版的两条带退出条件登记 + 引擎型面按版本号硬判)、`run-engine-notice-catalog-test.mjs`(七十三码 + 两枚新码的位置与 audience)。
|
|
81
|
+
- `mandate` 六词对引擎的 `PersistedRuleMandate` 型做**编译期双向钉**(本包常量自己的元素型;加 / 删一词构建即红);只用引擎的型,出包的 `.js` / `.d.ts` 不引用引擎包。
|
|
82
|
+
- 新门 `run-engine-agent-absence-projection-test.mjs`(102 格):公面与型面(AST)/ 措辞逐字、零终态词、不暗示已结束 / 单一谓词 / 落标与撤标 / 停表 / 在跑数与分格 / sweep 豁免 / 硬 TTL × 登记读数四形回收与留痕(`unknown` 与没给读数永不删、`running` 不删、只有 `ended` / `not_listed` 才删、观看中(`viewing`)的行只在登记确认离场时单列、旧名 `retained` 不读)/ 回收后终态句展示名与状态词两处消毒 / 真投影 → 真缺席订阅口的端到端段。
|
|
83
|
+
|
|
84
|
+
### Known limits(本版新增)
|
|
85
|
+
|
|
86
|
+
- 姿态谓词只读出身词与宿主的 settings 读数两路;另几位更强的卡上事实由宿主既有检查负责。祖先标记、粗粒度 shell 门、外联 / 不可逆 / 写保护标、部署策略这几类出身词不拦(出身词本身不足以判,由宿主既有检查先判;祖先标记要祖先原问题的事实,本包没有这一读位);规则店读不出的「这一次调用对不上规则」一形与读不出整店同按必问收;`hook` 出身一律必须问,CC 在 bypass 下对 hook ask 是有条件保留;宿主不报 settings 读数时除七个必问出身词外一律答 `unknown`,不认识的出身词即使宿主报了没命中也答 `unknown`;settings ask 规则的匹配在各端,本包只收一个布尔。
|
|
87
|
+
- 今天的服务端待决列表行不带 `origin`,停泊卡重开时仍缺出身词;`mandate` 要等服务端发布版带上才到卡。
|
|
88
|
+
- 缺席行在后台代理登记读口到来之前一行都不回收(宁可多挂,不可把仍在跑的子代删掉);缺席事件不带离场原词,「这一行回收过」的记账留在端;在跑数把 `pending` 算在跑、不区分前台行;回收句的分钟数四舍五入,`ttlMs` 小于 30 秒时说 `after 0 min`(只在测试旋钮缩短 TTL 时出现)。
|
|
89
|
+
- 桌面端经座位契约收审批请求,`ToolPermissionRequest` 上没有 `mandated` / `mandate` 位:桌面端的强制卡拿不到这两件事(存量缺口,本版不补)。
|
|
90
|
+
- 审批 feed 的订阅回调只在内容变化时发:端把一枚 feed 扇给多个会话时,新会话接入的同一拍按拉面 `reading()` / `snapshot()` 补读一次(配方见接入文档 §102f);显式刷新 `refresh()` 在回体整段读不懂时先发「不知道」再 reject,调它的端请接住。
|
|
91
|
+
- 完整台账见接入文档 §102 末行「包侧缺口」。
|
|
92
|
+
|
|
93
|
+
## 0.83.1(2026-09-26)
|
|
94
|
+
|
|
95
|
+
> 主题:patch —— 三件公共逻辑归包,只增公面、另有几处行为订正。① 展示层安全出口 `displayUntrusted` 三端单源,本包合成的终局行与结果帧 `errors[]` 在铸点洗掉凭据(CC-112 / CC-187);② 会话规则记录的无损判定 `sessionPolicyDeliverable`(CC-117);③ 插件 hook 逐条判「投给引擎 / 本客户端执行 / 如实不跑」,开机「不会执行」清单与 `/hooks` 标注措辞单源,同批不再把带参数的设置来源 command hook 条目送上请求、传了计划时也不再送 `mcp_tool` 条目(CC-174)。根公面运行期导出 1238 → **1254**(+16),公面类型 +26,`SettingsPort` +1 可选成员,`hooksForWire` +1 可选参;peer sdk 地板 `>=11.3.0` 不动。
|
|
96
|
+
|
|
97
|
+
### Added
|
|
98
|
+
|
|
99
|
+
- **`displayUntrusted(text, opts?)`**(CC-112):wire 派生文本上屏前的合成出口,三端一只。五张网默认全开 —— 凭据 URL 结构面(userinfo 整段、query 的每个值、fragment、路径段参数值、以已知凭据前缀打头的路径段)、凭据词级面(`Authorization: Bearer <值>`、`Authorization: token <值>` 一类非 bearer / basic 方案(方案词留、凭据段换记号;参数表形方案 —— Digest、签名算法形、OAuth 1.0 —— 参数名一律保留,`response` / `signature` / `oauth_signature` 的值换记号)、`api_key=<值>` / JSON 引号形 / `OPENAI_API_KEY=<值>` 这类键值对的值,分隔符两侧的空白不设上限;散文里不带标签的已知凭据字面形 —— `sk-…` / `ghp_…` / `AKIA…` / JWT 三段形 —— 整词换记号(随机段不足 16 位的 `sk-video` 这类名字不算);PEM 私钥块(含 PGP 私钥块)的主体换成一枚记号,头尾两行与它们旁边的换行保留,没有 END 行时遮到块尾,凭据标签后面紧跟私钥块时同样整块处理;诊断词与 `max_tokens` 一族计量单位不洗,方案词后跟的是一张窄词表里的常见英文词(`Invalid bearer token`;表外的词照遮)、反引号里点名的凭据变量名、裸复数 `tokens:` 后的纯整数计数、裸复数 `keys:` 后的键名清单也不洗)、控制符(C0 / DEL / C1 / 孤代理项)、双向与格式字符(格式类整类,ZWNJ / ZWJ 除外;含行 / 段分隔符)、行折平;顺序为凭据 → 字符 →(字符面改动过字节时)凭据 → 字符 … 到不动点。凭据两网在一份「判别视图」上匹配:ANSI 转义序列(按 ECMA-48 整族认:CSI、`ESC ( B` 一类字符集指定、`ESC 7` / `ESC M` 一类单字功能、已结尾的 OSC / DCS 控制串)、格式字符、控制符(以及本出口自己铸的 `\uXXXX` 转义)夹在标签、分隔符与值之间照样认得出,遮盖映回原文 —— 序列留在记号两侧,不会留下半截转义加记号的误导形;已结尾控制串的内容(窗口标题、超链接地址)另作一段文本洗;序列的最后一个字符恰是凭据词的首字母时(`<ESC>Bearer <值>`),按「终端吞掉这个字符」与「这个字符是词首」两种读法各判一遍,任一种认出就遮。整段文本线性处理(十万字符量级的标签密排 / 控制串引导符密排文本在百毫秒量级完成)。RGI 地区旗(黑旗加标签字符的整串)原样保留。参数表形方案的参数名与等号之间、等号与值之间带空白(`username = "bob"`)照认,头值折行续写照认(Negotiate / NTLM / Basic / Bearer 一类的首段照旧整段换记号);无 scheme 的 `user:pass@host` 与凭据标签后的转义反斜杠都不设长度上限(此前口令超过 256 字符 —— 例如把 JWT 当口令 —— 或 JSON 转义套三层以上时整段认不出、原样上屏)。选项:五张网各自可关(`credentialUrls` / `credentialWords` / `controls` / `bidi` / `foldLines`)、`keepLayout`(多行正文保 `\t` `\n`)、`mark`(危险字符呈现为可见转义 `\uXXXX`〔astral 为 `\u{…}`〕/ 点 / 空格,默认可见转义)、`max`(按输出封长,截点不劈转义记号、不劈代理对;不是有限数时不封)。幂等(带 `max` 时同样);干净文本原样返回(默认形开着行折平:换行、连续空白与首尾空白仍会折平);只管呈现、不参与任何判定。新增公面类型 `DisplayUntrustedOptions` / `DisplayMark`。接入文档 **§101a D-1 / 101a-1**。
|
|
100
|
+
- **会话规则记录的无损判定 `sessionPolicyDeliverable(behavior, rules)`**(CC-117):给一批同一 behavior 的规则串,答「能不能原样写进这条会话的规则记录」。只有一类能无损对上 —— 不带限定、恰好是一个工具名的 deny,写进 `toolDeny`(逐字、序不变、重复保留);其余每一条带一个成因词,闭集 `SESSION_POLICY_WITHHELD_WHY` 五词:`ask_no_bucket` / `qualified_deny` / `peer_wide` / `wildcard` / `not_a_tool_name`。🔴 **整批可送才送**:批里只要有一条送不了,`deliverable` 就是 `{}`,绝不给子集。坏形入参(表外 behavior、不是数组、读的时候抛错)不抛、一条都不送。每个成因词一句用户面话,单源 `sessionPolicyWithheldNotice(why)`,只收成因词、结构上回显不了规则串。判据与整批语义与此前宿主侧那一份相同,只有三处更严(都是此前判能写、而会话规则记录按原字节比一条都拦不住的形,本版扣下):① 首尾带空白的名字(如 ` Read`,`not_a_tool_name`);② 串里任何位置带 `*` 的规则(如 `*` / `Bash*` / `Web*` / `mcp__*__get_user` / `mcp__srv__get_*`,此前只扣下 MCP 工具段恰为 `*` 那一形;`wildcard`)—— 带 `*` 的规则在 CC 的规则语义里是通配(终端现行的匹配器只认 MCP 工具段恰为 `*` 那一形),而会话规则记录按精确名比、一条都拦不住;没有哪只工具的名字带 `*`,扣下的代价只是晚一拍生效。带括号限定的规则(如 `Bash(git push:*)`)仍报 `qualified_deny`:括号里的 `*` 是参数样式,不是工具名通配。③ 覆盖一个服务器或对端全部工具的规则不只认 MCP:引擎认得的每个协议命名空间(今天是 `mcp__` 与 `a2a__`)的 `<命名空间>__<服务器或对端>` 一律扣下(`peer_wide`)—— 此前 `a2a__payroll` 这类被判能写,而引擎把它当作覆盖该对端全部工具的规则,会话规则记录按精确名比一条都拦不住。成因词的用户面话 `sessionPolicyWithheldNotice(why)` 对任何入参都不抛(包括转成字符串时会抛错的值),表外的值一律回通用句。整批读不懂的入参(表外 behavior、不是数组、读的时候抛错)结果带 `unreadable: true`,与空批可分。新增公面类型 `SessionPolicyDeliverability` / `SessionPolicyWithheldRule` / `SessionPolicyWithheldWhy`。接入文档 **§101a P-1 / P-2 / 101a-3**。
|
|
101
|
+
- **插件 hook 每轮计划 `hooksWirePlan(opts?)`**(CC-174):设置来源投影 + 每一只插件 hook 的判定(`engine` / `shell` / `not_run` 加原因)+ 被治理筛掉的插件 hook + 被拿掉的设置来源 hook + 这一轮请求体 `settings.hooks` 的确切内容(`plan.wire`)。入参是宿主报的三条读数:`baseUrl`(读引擎的 `taskSettings.pluginHooks` 能力位)、`engineOwnedByThisShell`(引擎是不是本机由本客户端自起的)、`shellHookEvents`(本客户端自己的本地执行器为插件 hook 触发哪些事件)。宿主没报 `engineOwnedByThisShell`、或引擎能力还没探到时,判定原因与计划上的引擎事实都如实写「不知道」(原因 `engine_locality_unknown` / `engine_capability_unknown`,`plan.engine` 上对应位为 `null`),不说成「不是本机起的」「不支持」。传 `null` 与不传同处置。宿主交来的插件数据在读的时候抛错(会抛的 getter、已撤销的代理)与读口抛错同处置:这一轮 `pluginReader: 'failed'`、零插件判定与插件条目,设置来源照发,不外抛。🔴 每轮算一次、同一份用到底:先按它决定本地跳过哪些插件 hook,再把同一份交给 `hooksForWire({ plan })` 组请求体。接入文档 **§101a H-2 / 101a-4 / 101a-5**。
|
|
102
|
+
- **`SettingsPort.enabledPluginHooks?(): PluginHooksReading`**(可选,CC-174):已启用插件的 hooks(`hooks/hooks.json` 与 manifest `hooks` 按宿主加载后的形)、插件 id / 名 / 根目录 / 数据目录 / 是否由 managed 设置启用 / 声明的选项(敏感选项**只报声明,型面上没有值位**),加安全·bare 模式位。不实现 ⇒ 插件条目照旧不投,每个已装的端口告警一次。接入文档 **§101a H-1 / 101b**。
|
|
103
|
+
- **`hooksForWire(opts?)` 新增可选 `{ plan }`**:给了就原样返回 `plan.wire`(同一个对象);不给时不读插件口、不投插件条目。传 `null`(或 `{ plan: null }`)与不传同处置,不抛;`plan` 位上不是计划的值(空对象、数、串等)按不传处置 —— 设置来源照发,不会整份变空;把计划直接当第一个参数传入(`hooksForWire(plan)`)认得出,按 `{ plan }` 处置。接入文档 **§101a H-2**。
|
|
104
|
+
- **插件 hook 的查询口、措辞单源与闭集**(CC-174):查询口 `pluginHookVerdictOf(plan, pluginId, event, groupIndex, hookIndex)`(本地跳过同一只用);开机一次性「不会执行」清单 `hooksNotRunNotice(plan)`(按插件 × 原因一行,只含插件名、事件名与固定短句;名字过 `displayUntrusted` 的字符面,零宽 / 标签字符 / 软连字符渲成可见转义,引号转义成 `\"`,超过 120 个字符截断加省略号,转义与封长单遍完成)、`/hooks` 执行方标注 `pluginHookExecutorLabel(verdict)`、原因短句 `pluginHookReasonText(reason)` / `settingsHookDropText(reason)`;事件判定口 `isEngineFiredHookEvent(event)`;闭集 `ENGINE_FIRED_HOOK_EVENTS` / `PLUGIN_HOOK_DISPOSITIONS` / `PLUGIN_HOOK_REASONS` / `PLUGIN_HOOK_EXCLUSION_REASONS` / `SETTINGS_HOOK_DROP_REASONS` 与对应类型。判定原因闭集 `PLUGIN_HOOK_REASONS` 共 13 词,其中「引擎不是本机由本客户端起的」(`engine_not_local`)与「引擎不支持插件 hook」(`engine_capability_absent`)只在读数确实这么说时用;宿主没报 / 能力没探到走「不知道」两词(`engine_locality_unknown` / `engine_capability_unknown`)。接入文档 **§101a H-3 / H-4 / H-8 / 101a-7**。
|
|
105
|
+
- **插件 hook 投给引擎的投影形**(CC-174):本机自起的引擎报 `taskSettings.pluginHooks === true`、条目是 command、插件没声明敏感选项、并进去不超引擎每事件上限时,条目带 `plugin: { name, root, dataDir, options? }` 投出(`options` 只带已存的非敏感值)。🔴 截至本版,引擎尚未报出这一能力位,`plugin` / `args` 两键在引擎契约上也还没有定形 ⇒ 实际零投;判定、告知与本地跳过今天就生效。能力位与契约形在引擎侧同版出现时,本包在那一版核对键名与值形之后再放开;形与这里不同则按引擎的改。接入文档 **§101a-6**。
|
|
106
|
+
|
|
107
|
+
### Changed
|
|
108
|
+
|
|
109
|
+
- 🔴 **结果帧错误信封 `errors[]` 洗掉凭据 —— wire 可见的行为变化**(CC-187):`errors[]` 逐条过凭据两网,凭据位换成闭形记号(`«redacted:userinfo»` / `«redacted:query»` / `«redacted:fragment»` / `«redacted:secret»`),条数、顺序与其余字节不变。方向只会更安全(原来原样带出的凭据值不再带出);按 `errors[0]` 渲失败原因的端零改动即得净文本。射程按载体划:被挡原因不论是引擎、模型(经报告被挡的工具)还是 hook 反馈写的,进了合成终局行与 `errors[]` 就照洗;assistant 正文行、成功臂的 `result` 与错误信封的 `_sema_salvaged_result` 一个字节不碰。`errors[]` 里的非字符串元素原样放回。要在错误文本里抠 URL / 键值的消费方请按记号处理。接入文档 **§101a D-3 / 101a-2**。
|
|
110
|
+
- **本包合成的终局行正文在铸点洗掉凭据**(CC-187):终态错误行(`API Error:` / `Run stopped:` / `Run blocked…` / `Model output error:` 各形与「会话被占」那一句)与 `Outcome unknown:` 行,正文里的凭据位换成同一组记号;主机、端口、路径、query 键名、行首身份与其余文字逐字节不动。print 车道与交互车道是同一份产物;身份判定仍按引擎原话判完再洗,洗消只改呈现字节,不做字符面(那归呈现边界)。此前本包不洗,洗消只在终端自己的出口做,且只认 0.83.0 已改名的旧行类旗。`Outcome unknown:` 行与 `errors[0]` 洗后仍是同一句。接入文档 **§101a D-2 / 101a-2**。
|
|
111
|
+
- **设置来源 hook 条目上手写的 `plugin` 键在发出前剥掉**(CC-174):值为 `undefined` 也剥,条目其余部分照发;设置里写的 hook 不是插件,不许借这一键冒充插件上下文,请求体上的 `plugin` 位只由本包按插件读口交来的身份铸。宿主交来的设置文档本身不改,剥了会留一行 debug。接入文档 **§101a H-7 / 101a-6**。
|
|
112
|
+
- **字符面三只旧口改由同一只引擎实现,输出逐字节不变**(CC-112):`escapeDisplayControlChars` / `collapseLabel` / `capForDisplay`、同伴消息署名规范化与 hook 故障横幅共用 `displayUntrusted` 的字符面引擎,字符集保持各自旧集(双向族按枚举,不含零宽 / 软连字符 / 标签字符;署名规范化只折 C0 / DEL / 行段分隔符),全部 BMP 码元与代理组合对拍逐字节相同。它们窄于 `displayUntrusted` 的默认,新代码请用新口。接入文档 **§101a D-5**。
|
|
113
|
+
|
|
114
|
+
- **强制审批卡的那一句说明改口**(`mandatedApprovalDetail()`):此前那一句说「…so it is asked every time」,与引擎契约不符 —— 强制卡的约束是「存下的 allow 规则与记住的回答都清不掉,每一次调用要各自被回答」,而这一次的回答者可以是人,也可以是 hook 或部署运行的自动裁决(全域放行席、自动模式分类器、沙箱准入),所以这张卡未必每次都摆到人面前。新句:`no saved allow rule and no remembered answer can retire this question; each call is settled on its own — by you here, by a hook, or by an automatic check the deployment runs — and an answer covers only that call`。仍是零参数,仍不指人去写规则。按旧句逐字断言的测试改锚。接入文档 **§101a G-1**。
|
|
115
|
+
|
|
116
|
+
### Fixed
|
|
117
|
+
|
|
118
|
+
- **`failed` 帧上不是字符串的 `errorMessage` / `errorCode` 让流在合成终局行这一步抛错**:现在按缺席处理(回落到下一位,两位都缺席时正文为 `run failed`),流照常收尾。接入文档 **§101a D-2**。
|
|
119
|
+
- **交互车道重建的合成终局行丢了行类旗**(CC-187 同批):适配器把整条 assistant 消息重建成转录行时,上游帧带 `_sema_api_error_message: true` 的,重建行此前把它丢掉(合成行在交互车道上只剩 `model: '<synthetic>'` 可认),现在同样带上;模型行不新增任何键。接入文档 **§101a D-4**。
|
|
120
|
+
- **审批回决备注与子代续跑收据放过了双向 / 格式字符**(CC-112 残留收编):`readDecisionNoteAudit(ack).note` 与 `decisionNoteAuditLine(...)` 引用的备注,此前清掉控制字符却放过双向重排 / 格式字符(U+202E、U+2066、零宽、标签字符)、行 / 段分隔符与孤代理项,能把一条拒绝理由在屏上重排成另一句,现在一并折成空格(清洗后为空的备注与纯空白同处置:不算回显正文,不再渲一对空引号);`resumeSettledSubagent(...)` 成功时的 `receipt` 与失败时的调试日志行,此前只清 C0 / DEL,现在 C1(如 U+0085)、双向 / 格式字符与孤代理项一并换成 `.`,收据封长时不再截出半个代理对。除这几类字符外输出逐字节不变。接入文档 **§101a D-6 / D-7**。
|
|
121
|
+
- **设置来源的 `mcp_tool` hook 条目让整个请求被拒**(CC-174):引擎的 hooks 契约不认这一型,原样发出时整份 hooks 解析失败、整个请求被拒。现在:① 按计划组请求体(`hooksForWire({ plan })`)时不发,逐条记进 `plan.settingsDropped`,因此哪都不跑的进「不会执行」清单;② 不传计划的 `hooksForWire()` 在引擎点亮的事件上**照旧发出** —— 这类宿主看不到清单,整个请求被拒是它们唯一看得见的信号(被静默拿掉的守卫 hook = 用户以为在生效、其实没跑),有意保留;③ 引擎不点亮的事件上一律不发(引擎本来不跑这些事件,发出去只换来一次整份被拒)。接入文档 **§101a H-5**。
|
|
122
|
+
- **设置来源带参数的 command hook 条目被引擎只拿可执行名去跑**(CC-174):`type` 为 `command` 且带参数的条目 —— 非空数组(exec 形),或串 / 对象 / 数这些非数组值(上游本身不认,但照发同样会被剥)—— 引擎报出 `taskSettings.pluginHooks` 之前会静默丢掉 `args`、把 `command` 交给 shell 去跑(参数丢了,`command: "bash"` 这类会把 hook 的输入当脚本执行);现在能力位到货之前不发这类条目(不传计划时按未报判),到货之后原样发(非数组形由引擎响亮拒)。prompt / http 等条目上的 `args`、值为 `null` 的 `args` 不算,照旧原样发出;空数组 `args: []` 也不算带参数、照旧原样发出 —— 没有参数可丢,引擎按 shell 形跑的就是同一个可执行文件(如托管设置里的 `{ command: "/opt/guard/deny-dangerous", args: [] }`);只有命令串含空白、引号或 shell 特殊字符(shell 会拆词、展开或串接,与直接执行不等价)时才按带参数拿掉,判定只认由字母、数字与 `_ . / : + -` 组成的命令串为等价;其余条目(含形状不对的)照旧原样发出,由引擎响亮拒。拿掉原因与用户面话同一句(`exec_form_unsupported`)。不传计划的 `hooksForWire()` 拿掉这类条目、且那个事件由引擎点亮时,经日志口告警 —— 每个 settings 端口 × 事件 × 原因恰一次。接入文档 **§101a H-6**。
|
|
123
|
+
|
|
124
|
+
### Gates
|
|
125
|
+
|
|
126
|
+
- 新增常驻门三道:`run-display-untrusted-projection-test.mjs`(凭据两网的判据与反例样本、合成出口、旧口收编逐字节回归、残留收编、两车道合成行洗消、单源普查)、`run-session-policy-deliverable-test.mjs`(向量表、成因闭集、整批语义、坏形不抛、措辞纪律;在场时与宿主侧那一份逐例差分,分歧只许是首尾空白、带 `*` 的通配与非 MCP 命名空间整对端三类(逐类计数);另对实装引擎包的协议命名空间表双向对账)、`run-plugin-hooks-projection-test.mjs`(真端口夹具 + 真能力缓存进真 dist 的计划、请求体与措辞口)。`gates-manifest.json` 137 → 140,README「Guards」表同批 +3 行。
|
|
127
|
+
|
|
128
|
+
### Known limits(本版新增)
|
|
129
|
+
|
|
130
|
+
- 展示层:fragment 被整段换成记号之后,值里未编码的停字符(`|` `"` `<` 等)右边那一截落在洗法外(`#api_key=AB|<尾>` ⇒ `#«redacted:fragment»|<尾>`;真实 URL 进文本前已百分号编码);旧三口不跟随 `displayUntrusted` 的宽字符集(放宽是另一次行为变更);服务端逐串脱敏后投出的透传文本(引擎通告、审批卡正文与入参、回决备注、续跑收据)本包不再洗第二遍凭据;凭据面认不出的形 —— 驼峰名标签(`secretAccessKey`)、无前缀的 40 位 AWS 秘钥串、字面反斜杠写法的转义(`\x1B[33m`)夹在标签与值之间、括号或 YAML 块标记包住的值(块标记那一形里真值在下一行原样可见)、口令里含未编码 `/` `#` 的 userinfo、`Authorization` 方案词后换行再跟的普通值、标签与分隔符之间隔着空白又紧贴在上一只值后面的内层标签;仍会多遮的形 —— 标签后的普通词(`key: model`、`Missing key: ANTHROPIC_API_KEY`、`x-ratelimit-reset-tokens: 6m0s`)、方案词后的表外英文词(`basic subscription`)、引号里的键名、文档地址的 query / fragment 值、标签在行尾隔空行后的下一段首词(功能词 / 诊断词开头的除外)、`Bearer realm="…"` 挑战形里的 `realm=`;机读面(`errors[]` / 合成终局行)在记号两侧保留原有的控制符 / 格式字符(字符面归呈现边界)。与终端此前自带的那一只相比,凭据面上本包几乎只在更严一侧有差(控制符 / 格式字符 / 着色序列夹带、分隔符后长空白、`Authorization` 的非 bearer / basic 方案、散文里的已知凭据字面形),随机语料差分里更松的个例均为对方靠记号里的冒号误吃,或剥掉不可见字符后两边同样不遮。
|
|
131
|
+
- 会话规则:CC 旧工具名(如 `Task`)与带转义括号 / 反斜杠的名字会被判「能写」而系统里没有一层拦得住(与此前宿主侧同答);`mcp__<server>`、以及带 `*` 的规则(`mcp__<server>__*` / `mcp__<server>__get_*` / `Bash*` 等)在较新的引擎上有的其实能按集合命中,但 wire 上没有位说出对面是哪一代引擎,照旧不写(代价是晚一拍生效)。
|
|
132
|
+
- 插件 hook:引擎侧能力位与执行器尚未到货,插件 hook 在引擎腿上仍然不跑,本版做到的是如实说与判定就绪;守卫类插件 hook 投不出去时只如实告知,不改成逐次询问;插件 hook 模块(`register(on)` 形)与 skill / frontmatter hook 不在判定范围;exec 形里 `${user_config.KEY}` 由本包先替换、路径占位由执行方后替换,与上游次序相反;由引擎执行的插件 hook 没有进度与成功输出的展示面;设置来源的 hooks 在安全模式下的处置本包仍没有读口;不传计划的宿主若在引擎点亮的事件上配了 `mcp_tool` hook,整个请求照旧被拒(有意保留的响亮失败,改按计划组请求体即可);设置来源 command 条目上值为 `null` 的 `args` 照旧原样发出(与不写 `args` 同跑,上游本身不认这一形);不传计划的宿主上,被拿掉的 exec 形设置来源 hook(包括守卫类)只经日志口告警一次 —— 宿主若把日志只写进调试记录,用户看不到这一句,换钉前请改按计划组请求体并渲「不会执行」清单(见接入文档 §101y);空参数数组的 command 条目只在命令串由字母、数字与 `_ . / : + -` 组成时照发,引擎在 Windows 上改用 PowerShell 执行时这一等价判据未实测。
|
|
133
|
+
- 完整台账见接入文档 §101 末行「包侧缺口」。
|
|
134
|
+
|
|
52
135
|
## 0.83.0(2026-09-26)
|
|
53
136
|
|
|
54
137
|
> 主题:🔴 **minor**(行为面与型面都有 BREAKING)—— CC 形消息上 24 项非 CC 键按裁定 C-R103([8200] / [8232])改 `_sema_` 名、删除或改走 chrome 臂,转录 id 改成 UUID 形,0.82.7 标过渡的三只联网搜索旧读口到期删除,公面类型 −4;同版另有结果帧 CC 键 `terminal_reason`、用户层 `disableAllHooks` 在引擎腿上生效、退化审批卡三形拒收改写、注入件自愈句整族重写与引擎通告码册 +2;根公面运行期导出 1240 → **1238**,peer sdk 地板 `>=11.3.0` 不动。
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.83.
|
|
38
|
+
**Version:** 0.83.2
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -293,7 +293,7 @@ guard still cross-checks the table by name).
|
|
|
293
293
|
| `scripts/run-client-core-portability-test.mjs` | Kernel / A-layer / index import closures, the runtime-dependency equality gate, barrel reachability, and a real esbuild `--platform=browser` bundle |
|
|
294
294
|
| `scripts/run-client-core-diff-test.mjs` | Differential equivalence against the CLI reference bridge + replay-id invariant + ledger round-trip |
|
|
295
295
|
| `scripts/run-seat-contract-keys-test.mjs` | The seat IPC contract: verb list ↔ SPEC ↔ types, element-wise |
|
|
296
|
-
| `scripts/run-approval-frame-keys-test.mjs` | The tool-approval frame key mirror, element-wise against the SDK's runtime anchor (one carve-out: AHEAD_OF_ANCHOR entries — keys the server already emits but the SDK anchor has not caught up to — may lead by one generation; the gate turns red the day the SDK catches up, forcing the entry's removal — the register is occupied again — this time by the bit that says a saved allow rule cannot retire a given approval card, carrying both the release that minted it and the byte coordinates that prove it, so the lead is a dated record rather than an exemption; its predecessor left the register the other way, by being retired upstream rather than by the anchor catching up). Beside the key mirror it now guards three further faces of that bit: the closed word table the durable leg reads it through must be **the very array object the SDK exports**, not a same-looking copy — reference identity, because an equal-contents check still permits a second table that diverges the day upstream adds a member; the one predicate a client is meant to call answers over both legs — the live card's bit and the parked row's absence word, which is all the row carries, since the row has no such bit at all — and answers `false` for a malformed value exactly as its presence-only siblings do, a strictness the upstream mint shares; and the one sentence minted for it must never point the reader at writing a rule, since a rule written in answer to a mandated question can never take effect where it was written. The same bit's key also has to reach the card request itself, which the SDK's card anchor does not list — a fact that arrives at the package boundary and stops there is the shape of defect this file's guards exist to catch |
|
|
296
|
+
| `scripts/run-approval-frame-keys-test.mjs` | The tool-approval frame key mirror, element-wise against the SDK's runtime anchor (one carve-out: AHEAD_OF_ANCHOR entries — keys the server already emits but the SDK anchor has not caught up to — may lead by one generation; the gate turns red the day the SDK catches up, forcing the entry's removal — the register is occupied again — this time by the bit that says a saved allow rule cannot retire a given approval card, carrying both the release that minted it and the byte coordinates that prove it, so the lead is a dated record rather than an exemption; its predecessor left the register the other way, by being retired upstream rather than by the anchor catching up). Beside the key mirror it now guards three further faces of that bit: the closed word table the durable leg reads it through must be **the very array object the SDK exports**, not a same-looking copy — reference identity, because an equal-contents check still permits a second table that diverges the day upstream adds a member; the one predicate a client is meant to call answers over both legs — the live card's bit and the parked row's absence word, which is all the row carries, since the row has no such bit at all — and answers `false` for a malformed value exactly as its presence-only siblings do, a strictness the upstream mint shares; and the one sentence minted for it must never point the reader at writing a rule, since a rule written in answer to a mandated question can never take effect where it was written. The same bit's key also has to reach the card request itself, which the SDK's card anchor does not list — a fact that arrives at the package boundary and stops there is the shape of defect this file's guards exist to catch Since 0.83.2 the frame also mirrors `mandate` (the word a mandated question stands on) ahead of the SDK anchor and of the released server fixture, each allowance carrying its own exit condition; the installed engine must declare the member with the six-word closed type, and the key must reach the card request. |
|
|
297
297
|
| `scripts/run-segment-authority-single-source-test.mjs` | The authoritative-segment replacement verdict, single-sourced. `text_end.content` and the `text_delta` stream stopped being byte-identical the day the engine started redacting the former through the same filter as the result, so every consumer now has to decide six ways what to do with the segment it has half-emitted — and until this release that decision existed **twice**: once here for the transcript lane, once in the shell for the print lane, hot-fixed a version apart. The verdict is now one pure function both lanes call, and the guard pins it on the quantity that actually decides the outcome: whether the authoritative text still *starts with* the bytes that already left, not whether a flush has happened — the latter is a precondition, and anchoring on it withholds a perfectly ordinary answer. Each of the six forms is checked with its counter-case, the prefix length is pinned to UTF-16 code units against a non-ASCII sample whose UTF-8 byte count differs (slicing by bytes leaves the very thing being redacted on screen), and the withheld-segment ledger is compared by normalised equality rather than substring, because a short redaction marker quoted in an unrelated later answer would otherwise suppress that answer entirely. The same file pins the session-level memory-capture declaration to one mint point — the wire value is a single-member closed set, and a consumer that spells it wrong gets a loud refusal rather than a silently dropped privacy request — and pins the SDK URL/health transit to be the **same function reference**, since wrapping it would discard the one guarantee the transit exists for. A last section strips comments with the TypeScript parser and asserts the second expression has not grown back |
|
|
298
298
|
| `scripts/run-print-bash-iserror-test.mjs` | The print lane's Bash `is_error` authority (structured over regex). A second section pins where the denial classification word lands on this lane: on the message envelope, never inside the tool-result block, because that block is forwarded verbatim to the provider on compaction and a self-minted key there is the shape of an old, real defect. A word outside the upstream table — or an empty string, a non-string, or nothing at all — mints no key rather than a guess, and the word never moves the error flag, because attribution does not decide anything |
|
|
299
299
|
| `scripts/run-bash-benign-exit-interpretation-test.mjs` | Benign non-zero Bash exits (`returnCodeInterpretation`) stay non-errors across all three derivation arms, and the annotation transits to the card |
|
|
@@ -306,7 +306,7 @@ guard still cross-checks the table by name).
|
|
|
306
306
|
| `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails |
|
|
307
307
|
| `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools. The same roster now decides the tool list on the first line of a non-interactive run: the host holds that line until the roster arrives and lists exactly what the engine mounted at the start of the run, in mount order. The guard runs a real assembly frame through the projection into the decision, and pins that the host falls back to the estimate only once the roster is known not to be coming — a manifest without one, an unreadable one, model output or the run's end arriving first — rather than on a timer alone (model activity counts, including a model call that is still waiting or retrying; an error line the stream synthesizes when a run fails before assembly counts as the run ending), that a sub-run's manifest is never mistaken for the run's own, that an empty roster is taken as the engine's answer rather than as silence, and that the wait bound covers both sequential default budgets the engine gives an external tool server to connect and list its tools. The holding logic itself lives in the package as a small per-run gate — buffer, decide once, release the held messages in arrival order, then pass through — and the guard drives real stream output through it to pin that the release happens exactly once, at the manifest, releasing exactly the held prefix. The ordering itself also lives in the package as a stream wrapper, and the guard checks the final output a consumer reads: the first line is always the tool-list line, a message that arrives while that line is still being built comes after it, a timer firing races nothing out of order, a source that ends or fails before the decision still gets its first line and held messages out before the error, and an early exit closes the source |
|
|
308
308
|
| `scripts/run-permission-rule-issue-codes-test.mjs` | The rule-lint refusal codes an engine reports when it will not compile a permission rule. The SDK publishes neither a schema nor a type for them, so the package mints the table from the engine's own bytes and the guard pays the cost of that copy instead of leaving it to somebody remembering: it parses the codes the engine actually mints and reconciles them against the table in both directions, so a code added upstream (the user would see a bare code) and a code only the package believes in (a branch that can never fire) both fail. It also reconciles the table plus a small retired ledger against the engine's declared union, which is deliberately not the same set — one member was renamed and its old name is still declared — so reviving a code the engine will never mint again is impossible and a future stale member shows up immediately. Sentences are pinned one per code, mutually distinct, and split by family: a rule that is wrong and a rule that is legal but unsupported on this lane are different next steps and may not share a sentence. The engine's own message rides along as prose — sanitized and capped after escaping, never matched on |
|
|
309
|
-
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, nine words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because the sets differ in kind: one is genuinely closed on the wire (an out-of-set record is withheld by the engine, so reading one means the record is damaged) while the other is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine) Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. |
|
|
309
|
+
| `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, nine words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because the sets differ in kind: one is genuinely closed on the wire (an out-of-set record is withheld by the engine, so reading one means the record is damaged) while the other is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine) Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer; it must not say the question is asked every time — an answer for this one call may come from the person, a hook or an automatic check the deployment runs — and its wording is checked against the engine package's own description of the mandate A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. Since 0.83.2 a sixth list covers the word a mandated question stands on (`APPROVAL_MANDATE_WORDS`, six words): it must equal the engine's own list word for word and in order, membership is exact, the card reader `readApprovalMandate` answers only for an own key holding one of the six words, and each word has one fixed sentence explaining why the question must be confirmed — six distinct sentences that never point the reader at writing a rule, never promise a question every time, never claim only a person may answer, and repeat no other sentence the package mints. The list is also pinned against the engine's type at compile time in both directions, while the published build references no engine package at all: every `.js` and `.d.ts` file in the build is scanned, and the same scan is first shown to fire on references planted in a scratch directory. |
|
|
310
310
|
| `scripts/run-engine-identity-test.mjs` | The engine generation anchors on `/health` (`pid`, `instanceId`, `startedAt`; engine >=7.67.0). `/health` is the one unauthenticated door and its heartbeat is always green, so "another host restarted the shared engine" used to be discoverable only by having some authenticated request hit a 401 first — a path that misreads a restart as a network fault. The reader narrows each anchor independently (one malformed field never hides the other two) and always hands back a reading object rather than an absence, because the caller is asking which anchors answered, not whether there was a response. The comparison is a three-word verdict, not a boolean: `unknown` when the two readings share no comparable anchor at all — an empty intersection means nothing could be compared, never that nothing changed — and the boolean convenience is pinned so that only `true` is an assertion. Any comparable anchor differing decides `changed`, so a reading whose `startedAt` matches while its `instanceId` does not cannot be waved through as the same life; precedence only decides which anchor gets named in the diagnosis |
|
|
311
311
|
| `scripts/run-posture-knob-projection-test.mjs` | The three deployment knobs on the operator face (`serverGates.durableApproval` / `streamAskWindowMs` / `sessionAutoTitle`, engine >=7.67.0), each read as a value **plus who set it plus one operator-facing pointer** rather than a bare value — a bare boolean cannot answer why this particular machine is on this setting or how to pin it back, and a default that flips with the deployment shape is invisible without that. A worker too old to report readings still sends a bare boolean; the reader folds it into the same shell so consumers keep one branch, but raises a `legacy` bit, answers `undefined` from the machine-readable source accessor, and mints a sentence that contains no source word at all — claiming a source nobody reported is worse than admitting the worker cannot say. The other two knobs are honestly absent on such a worker rather than defaulted, a malformed side knob drops only itself while the anchor knob drops the whole reading, and the four sentences are pinned literally distinct so an operator can tell "not observed" from "not reported" from a real value. The last leg reads the installed SDK's `openapi.yaml` and `types.d.ts` directly, including a pin that exactly one knob on this face is numeric — the premise the millisecond-to-prose rendering rests on |
|
|
312
312
|
| `scripts/run-terminal-facts-projection-test.mjs` | The four unconsumed terminal-receipt facts: `TaskResult.effectiveReasoning` / `effectiveMemoryScopes` are narrowed into `_sema_effective_reasoning` / `_sema_effective_memory_scopes` on the CC-shaped `result` (success and error envelopes alike; a malformed value mints nothing, never a default tier), the resume **reopen** family (`resume.env_failed` / `tool_unavailable` / `tool_contract_mismatch`) is a frozen closed set with a reader and three-sentence copy that is disjoint from the refusal and retry-later sets, and `routePairingVerdict` reads `ModelInfo.routePairing` as ok / broken / unknown without policing the open set. A fifth section pins the structured-output key on the success result: the CC-spelled `structured_output` is the only home for the value the wire calls `structuredOutput`. The camelCase spelling this package used to mint on its own — a misspelling of the CC field, not an additive field of our own — rode alongside it for exactly one release (0.79.1) and is **absent from 0.80.0 on**, pinned both by own-key and by `in`, so a consumer still reading the old name sees `undefined` rather than a stale copy. The wire position is read exactly once, so a value-changing accessor is only ever asked for its first answer; absence stays absence; a wire key that is present but `undefined` mints nothing, since a key whose value is `undefined` makes a consumer that tests presence read "the engine produced nothing" as "the engine produced an empty result"; falsy-but-present values such as `null`, `0`, `""` and `false` are still minted, and so are shapes that are not records at all — an empty array, a populated array, a string, a number, a boolean — each carried through by the same reference, because the shape of that value is decided by the caller's own schema and the package does not get to filter it; and the error envelope carries no such key, because the CC error arm has no such field. Which spelling CC itself declares is witnessed from the mirror's own syntax tree rather than a constant copied into the guard, so the day that field is renamed upstream the guard says so. |
|
|
@@ -351,7 +351,7 @@ guard still cross-checks the table by name).
|
|
|
351
351
|
| `scripts/run-usage-verbatim-channel-test.mjs` | The two complementary usage disciplines (core 3.0.0 metering semantics): the CC `ModelUsage` mirror stays pure (five pinned keys, `totalInputTokens` has no seat), while the sema-owned channel forwards the engine `turn_end.usage` object **verbatim** (six keys, incl. `totalInputTokens`) via `last_turn_usage.engineUsage` / `handle.latestEngineUsage` — honest absence on pre-3.0.0 engines, no fabricated zeros |
|
|
352
352
|
| `scripts/run-plan-review-decide-verify-test.mjs` | `decidePlanReview`'s post-decide honesty ([2315]/[2316], engine RB-471 family): a 2xx from the decide endpoint is **not** a terminal — the wire re-pulls the task status and words the outcome by the real shape (still-locked / legal new gate / genuinely left park / unverified), never claiming success it hasn't earned; when the engine answers that the session's stored resume context cannot be read, the outcome names the unreadable row and says the decision was not applied. Driven against a real fake-engine HTTP server through the shipped dist. The outcome queue item also carries a machine-readable `_sema_planReviewOutcome` (task id, a package-minted dispatch number, decision, effect) so a host can tell which in-flight decision an outcome belongs to without searching the prose; the prose is byte-identical, the number is minted only for a decision the in-flight latch admits, and a caller that passes no metadata gets no key. |
|
|
353
353
|
| `scripts/run-shell-gate-durable-allow-test.mjs` | #110: the durable approval leg for **shell** gates. The tool_end HOLD/REJECT predicate must cover Bash the same way park detection already does (otherwise the park poison frame `Operation aborted` hits the transcript, `endedCalls` swallows the real replayed result, and the user who pressed Yes watches a command that really ran be reported as aborted); a replayed, already-decided park must resume reading the stream instead of being reported as a failed turn; `lastEventId` must track numeric `seq` too. Mutation-proven: each of the three fixes reverted turns the gate red |
|
|
354
|
-
| `scripts/run-hitl-gate-honesty-test.mjs` | [2393] the four HITL disciplines that a passing type-check cannot see. (1) The park predicate and the `tool_end` predicate must cover the **same** set — the park side admits a first-class `kind:'tool_approval'` gate for *any* tool name, and a `tool_end` frame carries no `kind`, so the frame-level judge falls back to the engine's exact abort marker; otherwise the poison frame hits the transcript and `markEnded` swallows the real replayed result (the #110 disease, reopened on kind-only gates). (2) The already-decided identity criterion is **one-shot**: its two inputs are monotonic, so without consumption one successful decide makes every later park failure — including a real `approvals.list` outage — read as "already resolved" until the 24-hop budget runs out and reports a cause that has nothing to do with what happened. (3) A `plan_review` card dismissed without an answer must be re-presentable: the idempotent re-arm short-circuit re-publishes the still-armed card, and a stale armed id (responder gone) re-arms from scratch rather than presenting a card nobody can answer. (4) `HitlSafetyError` is a safety signal — the `remember` fallback arm must re-raise it instead of auto-retrying the decide, while a plain unknown-key 400 still falls back. (5) The polling leg reschedules after an escaping throw and flips `mode()` to `idle` once it consistently fails, so the honesty surface stops reporting a dead feed as live. (6) The live-frame leg carries the fact behind "you are being asked because the auto-mode classifier could not run" all the way to the card port. Transit narrows on SHAPE only — a non-empty cause string is taken verbatim, an open set, because the word table's owner is the engine and re-checking a closed table at the package boundary would drop a legal value the day a new cause word appears, which is exactly the information worth keeping. A malformed carrier degrades to absence rather than half-minting, and absence stays absence: it covers "the classifier answered", "this ask never qualified" and "this deployment has no classifier" at once, so nothing may render it as reassurance. The guard also pins the division of labour that makes the open set safe — the same word that transits is judged again by the public display reader, which narrows to the availability axis, so a word the engine says it never stamps on this fact renders no sentence while still being visible on the card for triage A later section pins the split this release introduced on the deny close-out frame. Until now every denied tool call was stamped with the same sentence — the one that says *the user* does not want to proceed — including the calls denied automatically on a lane that has no approval surface at all, where nobody was ever asked. The guard drives all three shapes (a person pressed No, a rule settled it, nobody said which) through both close-out arms and the durable park leg, and pins that the third shape is byte-identical to the previous release: an attribution nobody supplied is not evidence for either answer. The rule-settled shape carries the shell's own reason on a second line when there is one and stands alone when there is not, because a blank line where a reason should be reads worse than no line at all. The attribution is read from own data properties only, so neither a polluted prototype nor a getter can make an automatic denial claim a person made it — and the getter case is pinned to never run at all. The transcript classification word is minted only on the two paths where the upstream transcript format really carries one; the three classifier words and the two abort words are left absent, with the abort words pinned against the strings this package actually normalises interruptions to, which are different strings. A final section pins the decide-operation observer: one `start` in the same tick as the first request and exactly one `end` after the last attempt has settled, across success, retried timeouts, exhausted transient failures, semantic refusal, binding mismatch and both kinds of caller abort, with nothing between retries; an observer that throws or rejects — even when logging that fault fails — never changes what is sent or returned, and the stream-level dependency reaches both durable park legs. A package-internal re-delivery of the same decision (plain approve after an older server rejects the session-scope flag, or a re-send without the attribution key) is reported as one operation with a single start and end. An operation handle only groups sends for the same session and bound call — a send for another gate through the same handle is its own operation — and closing a handle while a send is still in flight defers the end until that send settles. When a person picks "allow for this session" on a parked card and the grant is known not to have been stored — the server answers so, which newer servers do for the gates they can recognise from the parked row as needing a person each time, or the server refuses the session-wide grant with one of the refusal codes that are fixed by the row or the deployment and the package falls back to a plain approval — the parked path now says so with the same line the live path uses, and the receipt carries the server's bit for hosts that call that path directly; an answer without the bit, or any other failure — including a conflict that an internal retry can hit after an earlier send already stored the grant — is treated as unknown and says nothing, a plain approval that never asked for the grant says nothing, and a host logger that throws after a successful decision, on either card path or in the bridge's retry step, can no longer turn it into a second send or a failure. From 0.83.0 it also pins approval cards whose arguments are not the tool's real input: a live frame whose arguments were omitted over the size cap or never sent (the card holds at most a path recovered from the question), and a parked approval whose stored input is missing or replaced by the upstream size marker. On every such card an edited approval sends nothing at all — no respond, no decide, no session-grant attempt — ends as edit-refused, raises the same notice once and flags the card so hosts do not offer editing; both production entry points are driven end to end, and the parked path neither mistakes it for an already-decided gate nor reconnects. The explanatory line is pinned word for word in its four forms: a recovered path is the only thing it claims to have recovered, an unrecoverable one says nothing was recovered, neither claims a reconstructed diff, and the parked forms each say which of the two gaps it is, with near-miss shapes of the size marker still rendered as ordinary input. Real input on the stream, the frame or the record keeps edits flowing, and plain approvals and denials are untouched. |
|
|
354
|
+
| `scripts/run-hitl-gate-honesty-test.mjs` | [2393] the four HITL disciplines that a passing type-check cannot see. (1) The park predicate and the `tool_end` predicate must cover the **same** set — the park side admits a first-class `kind:'tool_approval'` gate for *any* tool name, and a `tool_end` frame carries no `kind`, so the frame-level judge falls back to the engine's exact abort marker; otherwise the poison frame hits the transcript and `markEnded` swallows the real replayed result (the #110 disease, reopened on kind-only gates). (2) The already-decided identity criterion is **one-shot**: its two inputs are monotonic, so without consumption one successful decide makes every later park failure — including a real `approvals.list` outage — read as "already resolved" until the 24-hop budget runs out and reports a cause that has nothing to do with what happened. (3) A `plan_review` card dismissed without an answer must be re-presentable: the idempotent re-arm short-circuit re-publishes the still-armed card, and a stale armed id (responder gone) re-arms from scratch rather than presenting a card nobody can answer. (4) `HitlSafetyError` is a safety signal — the `remember` fallback arm must re-raise it instead of auto-retrying the decide, while a plain unknown-key 400 still falls back. (5) The polling leg reschedules after an escaping throw and flips `mode()` to `idle` once it consistently fails, so the honesty surface stops reporting a dead feed as live. (6) The live-frame leg carries the fact behind "you are being asked because the auto-mode classifier could not run" all the way to the card port. Transit narrows on SHAPE only — a non-empty cause string is taken verbatim, an open set, because the word table's owner is the engine and re-checking a closed table at the package boundary would drop a legal value the day a new cause word appears, which is exactly the information worth keeping. A malformed carrier degrades to absence rather than half-minting, and absence stays absence: it covers "the classifier answered", "this ask never qualified" and "this deployment has no classifier" at once, so nothing may render it as reassurance. The guard also pins the division of labour that makes the open set safe — the same word that transits is judged again by the public display reader, which narrows to the availability axis, so a word the engine says it never stamps on this fact renders no sentence while still being visible on the card for triage A later section pins the split this release introduced on the deny close-out frame. Until now every denied tool call was stamped with the same sentence — the one that says *the user* does not want to proceed — including the calls denied automatically on a lane that has no approval surface at all, where nobody was ever asked. The guard drives all three shapes (a person pressed No, a rule settled it, nobody said which) through both close-out arms and the durable park leg, and pins that the third shape is byte-identical to the previous release: an attribution nobody supplied is not evidence for either answer. The rule-settled shape carries the shell's own reason on a second line when there is one and stands alone when there is not, because a blank line where a reason should be reads worse than no line at all. The attribution is read from own data properties only, so neither a polluted prototype nor a getter can make an automatic denial claim a person made it — and the getter case is pinned to never run at all. The transcript classification word is minted only on the two paths where the upstream transcript format really carries one; the three classifier words and the two abort words are left absent, with the abort words pinned against the strings this package actually normalises interruptions to, which are different strings. A final section pins the decide-operation observer: one `start` in the same tick as the first request and exactly one `end` after the last attempt has settled, across success, retried timeouts, exhausted transient failures, semantic refusal, binding mismatch and both kinds of caller abort, with nothing between retries; an observer that throws or rejects — even when logging that fault fails — never changes what is sent or returned, and the stream-level dependency reaches both durable park legs. A package-internal re-delivery of the same decision (plain approve after an older server rejects the session-scope flag, or a re-send without the attribution key) is reported as one operation with a single start and end. An operation handle only groups sends for the same session and bound call — a send for another gate through the same handle is its own operation — and closing a handle while a send is still in flight defers the end until that send settles. When a person picks "allow for this session" on a parked card and the grant is known not to have been stored — the server answers so, which newer servers do for the gates they can recognise from the parked row as needing a person each time, or the server refuses the session-wide grant with one of the refusal codes that are fixed by the row or the deployment and the package falls back to a plain approval — the parked path now says so with the same line the live path uses, and the receipt carries the server's bit for hosts that call that path directly; an answer without the bit, or any other failure — including a conflict that an internal retry can hit after an earlier send already stored the grant — is treated as unknown and says nothing, a plain approval that never asked for the grant says nothing, and a host logger that throws after a successful decision, on either card path or in the bridge's retry step, can no longer turn it into a second send or a failure. From 0.83.0 it also pins approval cards whose arguments are not the tool's real input: a live frame whose arguments were omitted over the size cap or never sent (the card holds at most a path recovered from the question), and a parked approval whose stored input is missing or replaced by the upstream size marker. On every such card an edited approval sends nothing at all — no respond, no decide, no session-grant attempt — ends as edit-refused, raises the same notice once and flags the card so hosts do not offer editing; both production entry points are driven end to end, and the parked path neither mistakes it for an already-decided gate nor reconnects. The explanatory line is pinned word for word in its four forms: a recovered path is the only thing it claims to have recovered, an unrecoverable one says nothing was recovered, neither claims a reconstructed diff, and the parked forms each say which of the two gaps it is, with near-miss shapes of the size marker still rendered as ordinary input. Real input on the stream, the frame or the record keeps edits flowing, and plain approvals and denials are untouched. Since 0.83.2 the live frame's `mandate` word reaches the card only when it is one of the six known words, read back identically by `readApprovalMandate`; it never adds the separate mandated key to the card, but a known word on its own makes `approvalIsMandated` answer true, because the engine only sends the word as the whole reason for that bit. The frame's `origin` is read from its own keys only, so a value inherited through the prototype chain never reaches the card. |
|
|
355
355
|
| `scripts/run-park-hop-progress-test.mjs` | L-80: the park re-attach loop budgets **stalled** rounds, not parks. A turn where the model keeps hitting gates and every one of them is really decided (a card was answered, the engine really moved on) must never be cut off by the hop budget — the budget counts consecutive rounds that produced no progress, and "the engine revived and immediately parked again on the same coordinates" is not progress. The three non-progress arms (already-resolved, decide-transport-exhausted, and a re-scan that was adopted but led nowhere) share one same-cause limit instead of one arm having a limit and the others having none, and every non-progress re-attach is announced once through the host callback rather than only to the debug log. When the limit is spent the resolver reads the approval queue once more and puts whatever is decidable in front of the user before it gives up; only when there is genuinely nothing to show does it fail soft, and the terminal message then carries the real cause and a real way out instead of a sentence about a budget. On the self-heal side, a reopen verdict that reports `decidedWithoutCard` — the chain settled the gate by rule, so there was no card to present — is progress, not a reopen failure, and the user is not told their message was NOT sent. Negative control: a genuinely empty queue with a run that never moves still fails soft |
|
|
356
356
|
| `scripts/run-notif-fleet-honesty-test.mjs` | [2393] the five notification/fleet disciplines a green type-check cannot see, each proven by reverting the fix. (1) The workflow-side dedup `return` keeps a count and a trace — without it "suppressed by design" and "a real completion swallowed because the runId minting changed" are the same observation. (2) `seq` normalisation has exactly one mint point, so a 0-based or fractional wire `seq` cannot make the watcher lane and the frame lane key the same completion differently (which would feed the model twice). (3) The TTL sweep defers to a probe arm that is still inside its own deadline — an entry recorded as "abandoned" must not be delivered a moment later — while an arm that has outlived its deadline never blocks the sweep, so the headless exit gate keeps its liveness. (4) The reset hook really clears every ledger it claims to (the sticky `prompt` ledger leaked across cases). (5) The fleet ledger counts all three drop paths (malformed / unknown frame type / isolation drop), and the panel projection's settled recycling is anchored on the settle instant and skips still-present rows, so the dedup token is never carried off with the entry (which would re-emit `end`) |
|
|
357
357
|
| `scripts/run-public-surface-test.mjs` | The outward promises: the npm export surface baseline (an **exact set**, both directions — a new export that never entered the baseline is one nobody watched leave, and deleting it later would not be red), the peer floor witness, and this README's claims |
|
|
@@ -362,7 +362,7 @@ guard still cross-checks the table by name).
|
|
|
362
362
|
| `scripts/run-client-core-singleton-test.mjs` | Module-level singletons ⇄ `docs/refactor/p1-scan/singleton-manifest.json`, **both directions**: an unregistered singleton is red (registering it forces someone to answer "what if this got duplicated"), a stale entry is red, and the `dupRisk: high` count only goes down |
|
|
363
363
|
| `scripts/run-catalog-loader-gates-test.mjs` | The model-catalog candidate chain (`loadCatalogWithSources`) and the provider device-code seam: offline ⇒ `bundled` with an honest `online.reason`, a good source ⇒ `online` plus a cache write, a second offline run ⇒ `cacheHit`; the three hostile source shapes (malformed JSON, `schemaVersion: 99`, off-domain `http`) each fall through to the bundled table, and an off-allowlist target is **never dialled** — including a `302` to another host, proven by a real loopback server's hit counter staying at zero; a one-byte edit to `catalog.sha256` drops that source while an unavailable sidecar only warns; and the device-code poller's `pending → ok` / `expired` arms run against a real loopback HTTP server with an injected clock |
|
|
364
364
|
| `scripts/run-abortable-sleep-test.mjs` | The shared `abortableSleep(ms, signal)` leaf (consumed by `workflowClient.ts` and `agentSession/backgroundView.ts`'s poll backoff): normal timeout resolution, immediate wake-up on `abort` mid-wait, `clearTimeout` really firing on that path, and a post-resolve late abort staying a no-op |
|
|
365
|
-
| `scripts/run-durable-card-display-keys-test.mjs` | The durable approval row's two display keys survive the row→card recast in `surfaceFsApprovalAndDecide`: `governanceForced` stamps on strict `true` only (absence is "no evidence", never `false`), `ruleSuggestions` passes through the same shape-narrowing reader as the live-frame leg and lands on the **read-only** card key — plus a standing pin that the durable leg never stamps the redeemable `ruleSuggestions` card position (the `/decide` body has no rule slot; offering a "don't ask again" option there would be an affordance nothing can honour), and a section for the parked twin of the classifier-unavailable fact: the upstream declares that key on the parked action itself, verbatim and under the same name as the synchronous ask, so this leg reads it rather than guessing a carrier name the way the deliberately unprojected keys must. The guard drives both legs with the same cause and asserts the card ends up byte-identical either way — the observable consequence of one reader serving two key paths, and the thing that silently diverges the day someone writes a second copy. Its own reach is printed rather than implied: what is proven is the package-boundary promise "on the row ⇒ on the card", not that today's engine flattens that key onto the pending row. A further section covers the two display facts the recast had been dropping for far longer. One of them the row has carried all along under a DIFFERENT NAME than the live frame uses — the frame puts it at the top level, the row nests it under the risk descriptor — and that difference in name is exactly why it went unnoticed; unlike the keys this leg deliberately refuses to project, its carrier is witnessed in the engine's own artefact rather than guessed. Neither is decoration: the shell's stand-aside arm reads them, so a call that matched a remembered allow rule which could NOT silence it looked like an ordinary ask on the durable path and was auto-approved with no card at all. Both land on the SAME card slot the live leg uses (one shape for the ends), verbatim bytes, present only when non-blank, never folded into an empty string — and the guard pins the discipline in both directions, including that a top-level key the upstream row does not actually have must still not grow this position. The security-class approval bit (`requiresRealApproval`) rides a parked row's card when the row carries it at top level, on strict `true` only, while look-alike nested carriers are ignored; this is pinned with a constructed row, because today's pending list does not carry the bit yet. A second, separate bit (`irreversibleParkGate`) marks a parked card whose row sits on the irreversible-ask gate kind — a gate-kind fact that covers asks the engine flagged for real approval at the first decision plus safety-tightened gates, with a known engine gap for approval demands raised only on a storage recheck — on the park path only, on the exact gate word only, and never in place of the real bit. |
|
|
365
|
+
| `scripts/run-durable-card-display-keys-test.mjs` | The durable approval row's two display keys survive the row→card recast in `surfaceFsApprovalAndDecide`: `governanceForced` stamps on strict `true` only (absence is "no evidence", never `false`), `ruleSuggestions` passes through the same shape-narrowing reader as the live-frame leg and lands on the **read-only** card key — plus a standing pin that the durable leg never stamps the redeemable `ruleSuggestions` card position (the `/decide` body has no rule slot; offering a "don't ask again" option there would be an affordance nothing can honour), and a section for the parked twin of the classifier-unavailable fact: the upstream declares that key on the parked action itself, verbatim and under the same name as the synchronous ask, so this leg reads it rather than guessing a carrier name the way the deliberately unprojected keys must. The guard drives both legs with the same cause and asserts the card ends up byte-identical either way — the observable consequence of one reader serving two key paths, and the thing that silently diverges the day someone writes a second copy. Its own reach is printed rather than implied: what is proven is the package-boundary promise "on the row ⇒ on the card", not that today's engine flattens that key onto the pending row. A further section covers the two display facts the recast had been dropping for far longer. One of them the row has carried all along under a DIFFERENT NAME than the live frame uses — the frame puts it at the top level, the row nests it under the risk descriptor — and that difference in name is exactly why it went unnoticed; unlike the keys this leg deliberately refuses to project, its carrier is witnessed in the engine's own artefact rather than guessed. Neither is decoration: the shell's stand-aside arm reads them, so a call that matched a remembered allow rule which could NOT silence it looked like an ordinary ask on the durable path and was auto-approved with no card at all. Both land on the SAME card slot the live leg uses (one shape for the ends), verbatim bytes, present only when non-blank, never folded into an empty string — and the guard pins the discipline in both directions, including that a top-level key the upstream row does not actually have must still not grow this position. The security-class approval bit (`requiresRealApproval`) rides a parked row's card when the row carries it at top level, on strict `true` only, while look-alike nested carriers are ignored; this is pinned with a constructed row, because today's pending list does not carry the bit yet. A second, separate bit (`irreversibleParkGate`) marks a parked card whose row sits on the irreversible-ask gate kind — a gate-kind fact that covers asks the engine flagged for real approval at the first decision plus safety-tightened gates, with a known engine gap for approval demands raised only on a storage recheck — on the park path only, on the exact gate word only, and never in place of the real bit. Since 0.83.2 the recast also carries the row's ask origin (`origin`) exactly as the live-frame leg does — a non-empty string, verbatim, open vocabulary, never invented or defaulted — and the word a mandated question stands on (`mandate`), accepted only when it is one of the six known words and read back through `readApprovalMandate`; a word outside that set, or a malformed value, leaves the card without it. The word never adds the separate mandated key to the card, but a known word on its own makes `approvalIsMandated` answer true, because the engine only sends the word as the whole reason for that bit. An `origin` inherited through the row's prototype chain is not read. |
|
|
366
366
|
| `scripts/run-session-memory-status-test.mjs` | The session **memory-status** read face (S-53): the two judgements three clients would otherwise each get wrong. First, *same status, different code* — this route's 404 carries two unrelated meanings (`not_found.session` = unknown or non-owned session; `not_found.route` = a pre-7.53 server that has no such route at all), so dispatching on the **status** would report "your deployment lacks this surface" as "your session does not exist". The verdict is anchored on `errorCode`, the two 404s are pinned to **different** verdicts, and — the load-bearing negative control — a 404 carrying **no** code falls to `failed` rather than guessing either way, since a wrong guess in either direction is a false statement a user would act on. 501 is allowed a codeless fallback because both of its arms mean the same thing here, and `capability.*` stays split from `feature.*` because those two share a status while their dispositions are opposite. Second, *absence means something different per key*: `optOutSource` and `lastCaptureAt` are legitimately absent on a **healthy** session (a zero-history session really is `{captureOptedOut:false, committedCount:0, foldedCount:0}` with no degradation at all), so reading absence as "off/none/0" asserts something unprovable. Two combined readers are pinned: capture opt-out is read from **both** its keys (a record-store fault yields `indeterminate`, never `active` — the difference between "your conversation is being remembered" and "nobody knows"), and last-capture is a **three-state** read whose discriminator is the *other* key, because `lastCaptureAt`'s absence alone covers both "ledger unreadable" and "genuinely no contributions" and therefore decides nothing; the two shapes are pinned to different verdicts so a single-key read turns red. The thin wrapper is the only IO: it never throws, drops malformed keys to absence rather than trusting them (an unreadable value must answer "don't know", never render as truth), refuses to spend a request on an empty `sessionId`, and passes `signal` through untouched |
|
|
367
367
|
| `scripts/run-crash-converged-projection-test.mjs` | The `crashConverged` read face on `GET /v1/approvals` (L-38): what the *previous life* of a crashed local engine left behind, projected for every client. Three judgements are pinned. First, **absence is not an empty list** — a missing key (an older server, deps not present, or a carrier that is not an array at all) returns `undefined`, and the client renders nothing; an empty array returns a present zero-count object, which is the server actually saying "none". Folding the first into `{total:0}` would have the client assert "nothing was left behind" on a surface a person uses to decide whether it is safe to re-run something — the worst possible direction for a false statement — so the two cases are pinned to different **return shapes** and a test asserts the two verdicts are unequal. Second, bucketing is a **four-term conjunction**: `orphanState === 'pending'` *and* `resumeSafe === true` *and* both approval-evidence keys (`originalDecision`, `decidedAtMs`) absent. A fifth term rejects any row carrying an **accessor**, and accessors are never invoked at all — reading one means synchronously running someone else's code, and `catch` catches throwing, not *never returning*, so a looping getter would pin the startup thread forever (the row cap does nothing against that shape). The same rule covers the three untrusted reads outside the row as well — the envelope's `crashConverged` key, the carrier's `length`, and every numeric index are read as own property *descriptors* and only data descriptors are used, so accessors and prototype entries read as absent and are never invoked. Such a key is treated as absent: if it was a required field the row is counted as dropped, if it was optional or additive the row survives without it. That also closes the ordering attack, since spreading runs getters in property order and an earlier one could `delete` the approval evidence before it is ever copied (measured before the fix: such a row reached the resume-safe bucket), and the check therefore moves ahead of the read, onto the property descriptors — from which the snapshot is then built directly, because checking descriptors and *then* spreading is two independent observations of the same row, and a non-throwing proxy can make the two `ownKeys` calls disagree (first showing `originalDecision: 'approve'` so the row reads as plain data, then omitting that configurable key so the snapshot loses the evidence; measured before the fix: the dangerous row reached the resume-safe bucket after exactly two enumerations, and after it, one). Keys are written with `Object.defineProperty` rather than plain assignment, because `'__proto__'` is a legal own enumerable key and `o['__proto__'] = x` does not store a value — it calls the prototype setter, letting a row whose own properties are all plain data (so the accessor gate never fires) inject a prototype whose `sessionId` getter deletes the approval evidence from the snapshot during validation; `defineProperty` fires no setter, so the key survives as ordinary additive data and the snapshot keeps `Object.prototype`. A row that simply arrives with a custom prototype is treated the same way, since the snapshot only enumerates own properties: approval evidence sitting on the prototype would never reach it, and a perfectly ordinary object with no proxy and no accessors could otherwise be called safe to re-run — real bodies come from `JSON.parse` and always carry `Object.prototype`, so nothing genuine trips it). Validation itself runs on a **null-prototype** dictionary and the bucketing verdict is carried out of that same pass rather than re-read from the delivered row, because every property lookup on an ordinary `{}` reaches `Object.prototype`: a polluted `sessionId` getter there would delete the approval evidence from the snapshot mid-validation and send the row to the safe bucket (measured before the fix). The row handed to the client is still an ordinary object — the null prototype is an implementation detail of the check, not of the value) — real JSON bodies are all data properties, so only a middle-layer-synthesised payload ever trips it, and it too lands in the human bucket rather than being dropped. The `decided` arm means the human had already approved and side effects may be half-landed, so it always goes to the human bucket, as does `resumeSafe === false` and — the last two terms — any row whose own fields contradict each other, since `pending` claims nothing ran while that evidence says somebody pressed approve. Deciding "not safe" costs one extra question (recoverable); deciding "safe" wrongly has somebody re-run work that already partly happened (not). A 2x2 truth table pins that exactly one cell is resume-safe, so reading either key alone turns red, and the contradictory rows are routed to the human bucket rather than dropped — they are real orphans, and the ones most worth showing. Third, unreadable rows are **dropped and counted**, never thrown and never passed through: the product is declared as `CrashConvergedRow`, so letting a row missing a required field — or carrying one of the wrong type — past would be a lie at the type level, and the closed literal discriminators (`decision` / `cause` / `orphanState`) decide family membership rather than being an open vocabulary. The measuring stick stops at the **type** floor, though: degenerate-but-well-typed values (`ts: NaN`, an empty `toolName`) are kept, because swallowing a real orphan over a decorative field is the worse direction, and the one deliberate exception is `approvalId`, which must be non-empty to be a row identity at all. `dropped` is kept separate from `total` so unreadable rows never inflate "N approvals were affected"; each row is a **one-shot snapshot** — every own enumerable key is read exactly once, and validation, bucketing and the handed-back value all read that same snapshot, so additive upstream keys survive while a **non-idempotent** getter (one that never throws, just answers differently on a second read) can no longer erase the approval evidence between the check and the bucketing (measured before the fix: such a row landed in the resume-safe bucket while its checked value was `"approve"`). Hostile carriers are counted rather than allowed to reject: **every** touch of the carrier is guarded — envelope property reads, `Array.isArray` itself (it throws on a revoked proxy), the `length` read, each indexed read and each row's property reads — and a traversal that dies halfway returns absence rather than a half-counted total. A row that cannot be read never takes the batch with it: its own shape check is inside its own guard, so one revoked-proxy row costs a `dropped` tick rather than collapsing the whole projection to absence — which a client would have read as "this deployment does not offer the surface". Traversal goes by **numeric index, never the carrier's own iterator protocol**, because `for...of` hands the carrier the question of which rows exist: an array carrying an overridden `Symbol.iterator` can yield nothing (measured before the fix: a real orphan became `{total:0}`, which a client reads as "the server said there are none") or swap a dangerous `decided` row for a safe-looking one (measured: `fake-safe` was returned in place of `real-danger`). Row count is capped at 100000 and the cap is checked **before** the walk: requiring only a non-negative integer `length` does not stop a proxy trap reporting a billion, and this surface runs on the startup / `--resume` path, where a synchronous spin freezes the thread (measured before the cap: twenty million rows took 18.3 seconds and twenty million index reads; a billion does not come back). The honest boundary is stated rather than overclaimed — a proxy can still lie in its `length` or index traps, which is the same thing as a host injecting a lying transport — and the widening of `ApprovalsResourceLike.list()` is proven **additive** by really running tsc over a legacy `{ pending }` mock *and* over the real `AgentClient` path — the projector takes `unknown` precisely because a parameter shaped as "an object with an optional `crashConverged`" is a TypeScript weak type that the installed SDK's own `list()` return shape shares no property with, which only a real-client compile would have caught — with a known-red control so a clean run means the checker spoke |
|
|
368
368
|
| `scripts/run-self-orchestration-denial-test.mjs` | The three judgements behind a **denied self-orchestration request** (server 7.57.0), each of which all three clients would otherwise get wrong on their own. First, whether to retry at all is a **conjunction that may not be loosened**: HTTP 501 *and* an `errorCode` that is **exactly** `capability.self_orchestration_required`. That code shares its shape with every other `capability.*` 501, so dispatching on the prefix would drag "some other capability is not wired up" into the retry arm — those requests do not become acceptable once the two keys are gone, so the client would spend a request and then tell the user the wrong reason. Negative controls cover all four directions: a sibling `capability.*` code, a truncated or suffixed variant of the right one, a codeless 501 (it decides nothing, so it decides nothing — no guessing), and the right code under 500 / 400 / 503 or a string `"501"`. The classifier reads structurally rather than by `instanceof` (a host may inject its own transport; across realms or duplicate SDK instances an understandable error would read as unreadable), so a class instance, a bare `{status, errorCode}` literal and an error carrying those fields on its **prototype** all reach the same verdict — and a hostile proxy or a throwing getter yields `null` instead of throwing, because this classifier runs inside a `catch` block where anything it throws escapes the caller's own guard. Second, removing the intent is a **structural** operation, not wording: `selfOrchestration` sits at the top level while `ultracode` sits under `settings` — two different stamping legs — and a client hand-writing `delete` will miss the second one, which costs the user the same failure twice. The single stripper is pinned to touch exactly those two: other `settings` sub-keys and their values survive byte for byte, `deferTools` is left alone (pulling `Workflow` out would be a behaviour change, not a removal of intent), additive unknown keys survive at both levels, the input object is never mutated, `settings` is only dropped entirely when `ultracode` was really there and nothing else remains (an already-empty one is left as is), a non-object `settings` is not touched at all, an `ultracode` that only exists on the prototype does not count, and the whole thing is idempotent. The end-to-end leg runs a real `buildTaskRequest` product through it and asserts the stripped body still passes the registration gate key by key. Third, on the capabilities body, **absence is not "switched off"**: a pre-7.57 server has no `workflowsGate` key at all, so reading absence as "the engine says no" asserts something the server never said, and the mirror-image disease is folding an **unrecognised** `denial` into `null`, which would have the client render "nothing was denied" when the truth is "denied, for a reason I do not recognise". Five shapes are pinned — caps unreadable, gate absent, closed-set member, unknown value, accessor — with the unknown arm carrying the raw token (or an empty one when the value is not even a string) and never collapsing to `null`. All four untrusted reads go through own **data descriptors** only, and the guard pins the getter invocation count at zero, since `catch` catches throwing but not *never returning*; a descriptor trap that throws and a revoked proxy both yield honest absence rather than an exception — though *what* absence means differs by field, and the guard pins that split rather than a blanket rule: an accessor on `workflows`, `workflowsGate` or `engineCan` reads as absent, while an accessor on `denial` reads as `{unknown:''}`, because a key that is **not there** is the gate saying "nothing was denied" whereas a key that is there but cannot be read is "denied, and I could not read why" — folding the second into the first is exactly the false statement this face exists to prevent. Two further pins came out of an adversarial review. The exported retry list is **frozen at runtime**, not merely `as const`: the verdict hands out that same reference, so any consumer splicing it once would poison every later verdict in the process — the guard asserts `Object.isFrozen`, that four different mutation attempts leave it byte-identical, and that a verdict issued *after* those attempts still carries the original two entries. And the classifier reads `denial` only **after** both criteria have passed, since it is not a criterion but an extra field on the verdict: the guard pins the getter invocation count at zero for any error that does not match and at most one for an error that does. The scope line is drawn explicitly rather than overclaimed — "no getter ever runs" holds for `projectWorkflowsGate`, which reads **wire JSON** where every field is an own data property by definition, but not for the classifier, which reads a **thrown value** that may well be an SDK `APIError` class instance carrying `status` and `errorCode` on its prototype; insisting on own data descriptors there would report a perfectly readable error as unreadable, so that side promises only that it never throws. A final pin covers the **integration document's own worked example** rather than the library: the shipped SDK's `tasks.stream()` is an `async` generator, so calling it issues no request at all — the POST happens inside `streamRaw` on the first iteration, and a `try` wrapped around the `stream(...)` call itself can never catch the 501. A client following a submit-shaped recipe on the streaming leg would never run the classifier, and the whole strip-and-retry path would silently do nothing. The guard drives the **real** `TasksResource` against a fake transport, offline, and pins both halves: the synchronous leg is in flight the moment it is called, the streaming leg has issued zero requests after the call and raises on the first `next()` — and it does so through the **real** error path, with `openStream` returning an actual 501 `Response` that the SDK's own `errorFromResponse` turns into the typed error, pinning the `openStream`→`errorFrom` call order so a transport that stops minting `errorCode` cannot pass. The documented recipe is then **executed** rather than keyword-counted: exactly one retry, a second body that really lost both keys while every other setting survives byte for byte, the caller's own request object left untouched, one disclosure and only one, a second 501 propagating with the request count still at two, and — after the first 501 — an abort leaving the count at one with nothing disclosed. A last leg is type-level: `stripSelfOrchestrationIntent` carries an SDK `TaskRequest` overload, because the wide `Record<string, unknown>` form erases the caller's type and the document's "strip and resubmit" line would not compile without an unsafe cast; a real tsc run over a virtual file proves both the narrow and the wide path, with a known-red control — and it compiles the document's two recipes **verbatim**, extracted from the section itself, because a recipe that does not compile is a recipe that was never given: `{ transientOk: true, signal }` is a TS2379 under `exactOptionalPropertyTypes`, which no amount of prose review had caught. The last thing pinned is the one that would have been quietest of all: the SDK's `stream()` returns only on a `done` or `failed` frame, so a stream truncated mid-run — or yielding nothing at all — ends the `for await` just as normally as a completed one. The documented `runOnce` therefore tracks whether it ever saw a terminal frame and raises when it did not, the guard's success fixture emits a real terminal and asserts the handler received it, and a truncated-stream control asserts that shape is reported as a failure with no retry and nothing disclosed. That terminal-frame rule then needed one more turn of its own: the underlying reader returns *normally* when the signal is aborted, so the check as first written rewrote a user's cancellation into a generic stream fault — a client keying off `AbortError` to suppress the error would instead have shown a failure, or resubmitted. Cancellation is therefore checked first, a real-SDK case aborts from inside the handler and asserts the original `AbortError` survives with no retry and nothing disclosed, and the document is checked for that ordering. The harness runs the documented `handle` and `transcript.note` as real spies rather than pushing frames itself, the drive loop rethrows exactly as the document does, and the disclosure ledger is proven to be the caller's own array by a positive identity assertion — without which the cancellation leg's "nothing disclosed" would have been vacuously true. Each recipe is compiled **on its own**, with a preamble that declares only what a host supplies and injects no library symbol, since compiling them together let the second one borrow the first one's imports, and the preamble's own types are decoupled from what the recipes import so the "remove the imports and it must fail" control fails for the right reason — which is checked by attribution, not merely by redness. Ordering is the last thing to get right: the cancellation check must come before the truncation error but **both** must sit behind the terminal-frame test, because a cancellation that lands after the run already reported `done` would otherwise overwrite a real outcome — one that may have already had effects — with "cancelled", and a person reading that will run it again. Aborting from inside `handle(done)` and `handle(failed)` are both pinned to still report success, and the ordering assertion is anchored inside the streaming `runOnce` body rather than the section, since the section's first `throwIfAborted` belongs to the synchronous recipe and would have made a reversed streaming recipe pass — and that ordering check is now anchored on the TypeScript AST rather than on text, since a comment reproducing the two statements in the right order let a genuinely reversed body pass. One more timing fact had to be written into the recipe: a single SSE read buffers several frames and the SDK yields them back to back, so checking the signal only after the loop lets a cancelled run keep consuming the rest of the chunk — measured, an abort inside `handle(turn_start)` still swallowed the `done` that followed and reported success. The recipe therefore re-checks after every non-terminal frame. Finally, the behavioural matrix is no longer run against a copy of the recipe: both recipes are extracted from the document, transpiled, and **executed** with injected host objects, so the disclosure assertion really exercises the document's own `transcript.note(disclose(...))` line, and the synchronous leg gets the same full matrix the streaming one does |
|
|
@@ -426,6 +426,11 @@ guard still cross-checks the table by name).
|
|
|
426
426
|
| `scripts/run-memory-saved-projection-test.mjs` | Engine memory writes (a successful `Remember` tool call) moved off the transcript onto the additive `memory_saved` chrome event, driven through the real pipeline: zero transcript rows for the write (the transcript is byte-identical to the same frames with the write reported as not successful), exactly one event whose `notes` carry the note text verbatim (notes, not file paths) and whose key set is exactly kind / laneProof / id / notes; no event for a missing, empty or non-string note, a non-`true` `ok`, a tool error, a missing result or another tool name; two writes give two events in order with distinct ids; a sub-agent write rides the sub-agent lane and an empty parent id emits nothing rather than falling back to the main lane; the `id` is derived from the write's wire key (the tool-end event id, else the tool-start event id, else the call id; empty ids count as absent), so projecting the same wire events twice gives the same id, and it never collides with the tool result row of the same or another call; the event sits right after the tool result row; the arm is registered as required. |
|
|
427
427
|
| `scripts/run-result-frame-projection-test.mjs` | Result frames and the synthesized terminal rows. The CC key `terminal_reason` is minted on result frames only where it follows from what the engine reported: `completed` on success, `max_turns`, `budget_exhausted` and `structured_output_retry_exhausted` for the three matching engine codes, on both the done-frame path and the failed-event path. Every other outcome leaves the key absent as an own property rather than present with an undefined value: wall-clock and token-budget limits, the classifier denial limit, cancellation, unknown codes, blocked, paused, unreadable or missing terminal records, and the busy-session refusal. The public reader `terminalReasonForResult` shares the minting predicate and is checked to agree with the minted key on every frame the gate produces. Both the minted key and the reader derive the word from the frame's CC subtype (success with `is_error` strictly false, and the three limit subtypes), not from the error code, so a replayed row whose status is paused, blocked or unrecognised never carries a word that contradicts its subtype. The four words are checked against the mirrored CC union, and the minting file is checked to hold no hand-copied code literals. The renamed superset keys (`_sema_error_code`, `_sema_salvaged_result`, `_sema_model_degraded`, `_sema_selected_model`, and the row flag `_sema_api_error_message`) are driven through the real stream pipeline. Each must be present under its new name, the old name must be absent, and every frame the gate saw is swept for old names. The selected model appears on error envelopes whenever the terminal record carries it, and never on a failed event, which has no record. It stays separate from the provider-reported model name. The two in-package readers still work: the interactive result arm reads the salvaged text under its new name (and old-shape frames under the old one), and the print init gate treats the renamed flag as the run having ended. |
|
|
428
428
|
| `scripts/run-layering-shadow-export-test.mjs` | Same-name shadows across the first-party clients that consume this package (terminal, desktop, web and the admin console). Each client's product sources are read at the local clone's `origin/main` (or its HEAD when there is no such ref), without fetching, and parsed with the TypeScript parser; every top-level runtime export the client declares itself is compared with this package's public runtime exports. The guard prints which ref, commit and commit date it read for each client, and warns (without failing) when that commit is more than seven days old, because the result then only describes that older snapshot. A client-side declaration carrying the name of a package export means a piece of shared logic now lives in two places and can drift apart. It fails the guard unless it is listed in `scripts/layering-shadow-exemptions.json`, and a listed row must carry a retire-by version no more than three minor lines ahead (it fails once the package reaches it). It also fails once the client has removed the shadow and the row still stands. Re-exports of this package's own exports are the intended form and never count. A client tree that is not present is reported as a skipped section, not as a pass. The ruler proves itself on an in-memory fake client (planted shadows must be caught, legal forms must not), on a throwaway repository (a missing `origin/main` falls back to HEAD, a broken one is a fault rather than a silent fallback), and refuses to report zero on a client whose scan surface is empty. |
|
|
429
|
+
| `scripts/run-session-policy-deliverable-test.mjs` | Which of a batch of user-written permission rules can be written into a session’s own rule record without changing their meaning, and why each of the others cannot. The record holds whole tool names and command names only, so exactly one class maps across losslessly: a deny rule that names one tool with no qualifier. Everything else is withheld with one word from a closed five-word list — an ask rule (the record has no ask tier), a deny rule with a parenthesised qualifier (recording just the name could block more), a rule covering every tool of one server or agent peer (for every protocol namespace the engine knows, checked against the engine package's own table) or containing a wildcard (*) anywhere (an engine that compares exact names would block nothing), and an entry that is not a tool name — and each word has one sentence, which never echoes the rule itself; asking for the sentence never throws, even with a value that throws when turned into a string. A name with leading or trailing whitespace counts as not a tool name: the record compares exact bytes, so it would block nothing. The guard pins the batch semantics: the deliverable part is either the whole batch or empty, never a subset, so a caller cannot send half a change and report it as saved. It also checks that malformed input never throws and never delivers anything (non-arrays, non-string entries, holes, a polluted array prototype, a length or index that throws, a changing index read once), that a batch which cannot be read at all is marked `unreadable: true` while an empty batch is not, so the two stay tellable apart, that each word is produced by some vector and nothing outside the list is produced, and — when a checkout of the previous in-client implementation is present — that this function gives the same answer on every recorded vector and on tens of thousands of generated rules and pairs, except for three deliberately stricter classes (whitespace-padded names; rules with a wildcard anywhere, which the previous implementation sent as exact names unless the wildcard was the whole tool part of a server rule; and peer-wide rules outside the MCP namespace, which it did not recognise), whose disagreements are counted per class and must match an independent count exactly. |
|
|
430
|
+
| `scripts/run-plugin-hooks-projection-test.mjs` | Plugin hooks: each command hook an enabled plugin declares is decided one by one as running in the engine, running in this client, or not running at all, and the page of hooks sent with a request is built from the same per-turn plan the client uses to skip its own copies, so one hook never runs in two places. Governance is judged first and always wins — a managed hooks switch-off, an untrusted workspace, safe or bare mode, or a governance read that fails sends no plugin hook and does not list it as a gap; managed-hooks-only (set directly, or through a merged non-managed hooks switch-off) keeps only managed plugins; the plugin-only customization lock does not touch plugin hooks. A hook reaches the engine only when this client started the engine on this machine, the engine reports plugin-hook support, the entry is a command, the plugin declares no sensitive option, and the event still fits the engine's per-event limits; the gate walks that matrix cell by cell, including the limit boundaries and a session goal hook counting toward them. A fact that was never read is reported as not known rather than as a fact: a host that does not say where the engine runs gets a "not known whether this client started the engine" reason, an engine whose capabilities have not been read yet gets a "not known yet whether it supports plugin hooks" reason, and the plan's two engine facts are null in those cases, not false. Events the engine never fires run only if the client says it fires them itself, and hooks the upstream behaviour itself refuses (option references in a shell-form command, an unset option in exec form, malformed entries) run nowhere. Exec-form arguments are passed element by element with only saved non-sensitive option references filled in; path placeholders are left for the executor. Sensitive option values never reach the request: with a host that wrongly supplies one, every string in the plan, the request body, the notice, the labels and the log is searched for it across eight cells. A host without the plugin reader keeps the previous request body and gets exactly one warning per settings port; plugin data that throws while it is being read (a throwing getter, a revoked proxy) is treated like a failing reader — no plugin hooks this turn, settings hooks still sent, nothing thrown; the not-running notice names the plugin and events, never a command or an option value, and escapes control characters in names. Command hooks from settings that carry arguments (a non-empty `args` array, which is the exec form, or any other non-null value) are removed from the request until the engine reports support for arguments, because the engine would otherwise drop the arguments and run the bare command through a shell; an empty `args` array is not treated as carrying arguments when the command is made only of letters, digits and `_ . / : + -` (the shell runs the same executable), so such a guard still reaches the engine, while an empty array on a command with spaces or shell characters is removed; `args` on a prompt or http entry, a null `args`, or an entry with no type is left alone, and those go out unchanged. MCP tool hooks, which the engine cannot parse, are removed only from a request built from a plan, whose not-running notice the host shows; a request built without a plan still carries them on engine-fired events, so the engine rejects the whole request loudly instead of a guard hook silently not running — the gate checks both request bodies against the engine's own hooks schema. Without a plan, every removed hook of that kind on an engine-fired event produces one warning per settings port, event and reason. Malformed entries still pass through for the engine to reject loudly, and passing null where the options object goes behaves like passing nothing; a `plan` option that is not a plan is ignored rather than turning the whole page into nothing, and a plan passed directly in place of the options object is recognised and used. A `plugin` key written by hand on a settings hook is stripped before sending (even when its value is undefined), because only hooks that come from the plugin reader may carry plugin context; the settings document itself is left untouched and a debug line records the count. The two hand-copied tables, the engine-fired event list and the engine limits, are checked against their owners. |
|
|
431
|
+
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. |
|
|
432
|
+
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. |
|
|
433
|
+
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. |
|
|
429
434
|
|
|
430
435
|
Each suite carries a floor that only moves up — a refactor that stops executing a group of
|
|
431
436
|
assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
|
package/dist/adapt/arms.js
CHANGED
|
@@ -119,6 +119,7 @@ const assistantArm = function* (m, { ctx, idOf, text, cards, inst }) {
|
|
|
119
119
|
uuid: idOf(m),
|
|
120
120
|
session_id: ctx.sessionId,
|
|
121
121
|
parent_tool_use_id: null,
|
|
122
|
+
...(m._sema_api_error_message === true ? { _sema_api_error_message: true } : {}),
|
|
122
123
|
}, ctx.now());
|
|
123
124
|
text.markEmittedText(textBlock.text);
|
|
124
125
|
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { isCcToolDenialKind, isCcToolDenialKindADenial, isGateDeniedByWord } from '../../gateVocabulary.js';
|
|
2
2
|
import { stamp } from '../types.js';
|
|
3
3
|
import { putOwnKey } from '../../ownKey.js';
|
|
4
|
+
import { REDACTION_MARKER_PREFIX, washCredentials } from '../../displayUntrusted.js';
|
|
4
5
|
import { readEffectiveReasoning, readEffectiveMemoryScopes } from '../../effectiveFacts.js';
|
|
5
6
|
import { isReviewPark, readRunTerminal, runTerminalCode, runTerminalGateToolName, } from '../../runTerminal.js';
|
|
6
7
|
import { toCcModelUsage } from './turnUsageToModelUsage.js';
|
|
@@ -30,7 +31,6 @@ function nonEmptyStr(v) {
|
|
|
30
31
|
return typeof v === 'string' && v.length > 0 ? v : undefined;
|
|
31
32
|
}
|
|
32
33
|
export const TOOL_INPUT_JOIN_MAX_CALLS = 4096;
|
|
33
|
-
const REDACTION_TOKEN_PREFIX = '\u00abredacted';
|
|
34
34
|
function isTransportPlaceholderLeaf(v) {
|
|
35
35
|
return v === '[circular]' || v === '[depth-limit]';
|
|
36
36
|
}
|
|
@@ -46,7 +46,7 @@ function joinableToolInput(args) {
|
|
|
46
46
|
if (++nodes > TOOL_INPUT_SCAN_MAX_NODES || depth > TOOL_INPUT_SCAN_MAX_DEPTH)
|
|
47
47
|
return undefined;
|
|
48
48
|
if (typeof v === 'string') {
|
|
49
|
-
if (v.includes(
|
|
49
|
+
if (v.includes(REDACTION_MARKER_PREFIX) || isTransportPlaceholderLeaf(v))
|
|
50
50
|
return undefined;
|
|
51
51
|
continue;
|
|
52
52
|
}
|
|
@@ -483,7 +483,7 @@ function errorResult(ctx, parts) {
|
|
|
483
483
|
...costFactParts(parts.stats, parts.observed),
|
|
484
484
|
...reemitEffectiveFacts(parts.facts),
|
|
485
485
|
...nestedUsageByTaskParts(parts.stats, parts.observed?.nestedUsageByTask),
|
|
486
|
-
errors:
|
|
486
|
+
errors: parts.errors.map(washCredentials),
|
|
487
487
|
...terminalReasonParts(parts.subtype, true),
|
|
488
488
|
...(parts.errorCode !== undefined && parts.errorCode.length > 0 ? { _sema_error_code: parts.errorCode } : {}),
|
|
489
489
|
...(parts.degraded !== undefined ? { _sema_model_degraded: parts.degraded } : {}),
|
|
@@ -5,6 +5,7 @@ import { readRunCostFacts, terminalToSdkResult, TOOL_INPUT_JOIN_MAX_CALLS } from
|
|
|
5
5
|
import { coerceOutput, publishSubagentContentEvent } from '../subagentContentStore.js';
|
|
6
6
|
import { ACTIVE_RUN_BUSY_ERROR_CODE, OUTPUT_INVALID, isLimitsExceededCode } from '../engineErrorCodes.js';
|
|
7
7
|
import { isReviewPark, readRunTerminal, runTerminalCode } from '../runTerminal.js';
|
|
8
|
+
import { washCredentials } from '../displayUntrusted.js';
|
|
8
9
|
import { ccToolDenialKindForToolEnd, gateDeniedBy, gateOutcomeOf } from '../gateOutcome.js';
|
|
9
10
|
import { isCcToolDenialKind, isCcToolDenialKindADenial, isGateDeniedByWord } from '../gateVocabulary.js';
|
|
10
11
|
const mainLane = () => ({ lane: 'main' });
|
|
@@ -101,7 +102,7 @@ function syntheticTerminalRow(ctx, text) {
|
|
|
101
102
|
uuid: undefined,
|
|
102
103
|
session_id: undefined,
|
|
103
104
|
type: 'assistant',
|
|
104
|
-
message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text }] },
|
|
105
|
+
message: { role: 'assistant', model: '<synthetic>', content: [{ type: 'text', text: washCredentials(text) }] },
|
|
105
106
|
parent_tool_use_id: null,
|
|
106
107
|
_sema_api_error_message: true,
|
|
107
108
|
});
|
|
@@ -158,6 +159,21 @@ export function _resetDroppedFrameReportForTest() {
|
|
|
158
159
|
export function _droppedFrameMemoSizeForTest() {
|
|
159
160
|
return reportedDroppedTypes.size;
|
|
160
161
|
}
|
|
162
|
+
function withStringFailureFields(ev) {
|
|
163
|
+
if (ev.type !== 'failed')
|
|
164
|
+
return ev;
|
|
165
|
+
const got = ev;
|
|
166
|
+
const badMessage = got.errorMessage !== undefined && typeof got.errorMessage !== 'string';
|
|
167
|
+
const badCode = got.errorCode !== undefined && typeof got.errorCode !== 'string';
|
|
168
|
+
if (!badMessage && !badCode)
|
|
169
|
+
return ev;
|
|
170
|
+
const { errorMessage, errorCode, ...rest } = ev;
|
|
171
|
+
return {
|
|
172
|
+
...rest,
|
|
173
|
+
...(badMessage ? {} : { errorMessage }),
|
|
174
|
+
...(badCode ? {} : { errorCode }),
|
|
175
|
+
};
|
|
176
|
+
}
|
|
161
177
|
let inFlightTurns = 0;
|
|
162
178
|
export function isRunStreamActive() {
|
|
163
179
|
return inFlightTurns > 0;
|
|
@@ -199,7 +215,8 @@ async function* runStreamInner(events, ctx, handle = {}) {
|
|
|
199
215
|
return false;
|
|
200
216
|
}
|
|
201
217
|
};
|
|
202
|
-
for await (const
|
|
218
|
+
for await (const rawEv of events) {
|
|
219
|
+
const ev = withStringFailureFields(rawEv);
|
|
203
220
|
if (ev.type === 'tool_end' && ev.isError !== true)
|
|
204
221
|
successfulToolEndObserved = true;
|
|
205
222
|
const seq = eventSeq(ev);
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
export declare const REDACTION_MARKER_PREFIX = "\u00ABredacted";
|
|
2
|
+
export declare function looksLikeKnownSecret(s: string): boolean;
|
|
3
|
+
export declare function washCredentials(text: string): string;
|
|
4
|
+
export type DisplayMark = 'escape' | 'dot' | 'space';
|
|
5
|
+
export interface DisplayFace {
|
|
6
|
+
readonly unsafe: RegExp | null;
|
|
7
|
+
readonly fold: boolean;
|
|
8
|
+
readonly mark: DisplayMark;
|
|
9
|
+
}
|
|
10
|
+
export declare const LEGACY_ESCAPE_FACE: DisplayFace;
|
|
11
|
+
export declare const LEGACY_LABEL_FACE: DisplayFace;
|
|
12
|
+
export declare const LEGACY_NAME_FACE: DisplayFace;
|
|
13
|
+
export declare function applyDisplayFace(text: string, face: DisplayFace): string;
|
|
14
|
+
export interface DisplayUntrustedOptions {
|
|
15
|
+
readonly credentialUrls?: boolean;
|
|
16
|
+
readonly credentialWords?: boolean;
|
|
17
|
+
readonly controls?: boolean;
|
|
18
|
+
readonly bidi?: boolean;
|
|
19
|
+
readonly foldLines?: boolean;
|
|
20
|
+
readonly keepLayout?: boolean;
|
|
21
|
+
readonly mark?: DisplayMark;
|
|
22
|
+
readonly max?: number;
|
|
23
|
+
}
|
|
24
|
+
export declare function displayUntrusted(text: string, opts?: DisplayUntrustedOptions): string;
|