@sema-agent/client-core 0.81.1 → 0.82.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/README.md +12 -8
  3. package/dist/adapt/toolCards.d.ts +15 -0
  4. package/dist/adapt/toolCards.js +34 -19
  5. package/dist/adapter/activeRunSelfHeal.d.ts +3 -0
  6. package/dist/adapter/activeRunSelfHeal.js +42 -9
  7. package/dist/adapter/downstream/eventToSdkMessage.js +7 -4
  8. package/dist/adapter/downstream/terminalToSdkResult.d.ts +2 -0
  9. package/dist/adapter/downstream/terminalToSdkResult.js +44 -5
  10. package/dist/adapter/runStream.js +33 -4
  11. package/dist/adapter/types.d.ts +2 -0
  12. package/dist/agentsWireCaps.d.ts +1 -0
  13. package/dist/agentsWireCaps.js +110 -29
  14. package/dist/decideFailureNote.d.ts +1 -0
  15. package/dist/decideFailureNote.js +8 -0
  16. package/dist/engineErrorCodes.d.ts +1 -0
  17. package/dist/engineErrorCodes.js +1 -0
  18. package/dist/engineNoticeCodes.js +2 -0
  19. package/dist/executionLaneCapability.js +3 -0
  20. package/dist/fileHistoryCaptureCapability.d.ts +18 -0
  21. package/dist/fileHistoryCaptureCapability.js +58 -0
  22. package/dist/gateOutcome.js +22 -2
  23. package/dist/hitl/askGateWire.d.ts +1 -1
  24. package/dist/hitl/askGateWire.js +1 -1
  25. package/dist/hitl/hitlHostSurface.js +6 -4
  26. package/dist/hitl/parkResolver.d.ts +1 -0
  27. package/dist/hitl/parkResolver.js +11 -2
  28. package/dist/hitl/planReviewWire.js +14 -1
  29. package/dist/hitl/toolApprovalWire.js +3 -2
  30. package/dist/index.d.ts +4 -0
  31. package/dist/index.js +4 -0
  32. package/dist/modelIdentityResolvability.d.ts +31 -0
  33. package/dist/modelIdentityResolvability.js +145 -0
  34. package/dist/printToolResultFrame.js +15 -1
  35. package/dist/registryQuotaUsage.d.ts +44 -0
  36. package/dist/registryQuotaUsage.js +111 -0
  37. package/dist/resumeRefusalCopy.d.ts +12 -0
  38. package/dist/resumeRefusalCopy.js +45 -1
  39. package/dist/rewindArchiveCapability.d.ts +24 -0
  40. package/dist/rewindArchiveCapability.js +125 -0
  41. package/dist/seam.d.ts +2 -0
  42. package/dist/toolResult.d.ts +2 -2
  43. package/dist/toolResult.js +7 -6
  44. package/dist/toolRoster.d.ts +12 -0
  45. package/dist/toolRoster.js +35 -0
  46. package/dist/wireErrorTriage.d.ts +4 -0
  47. package/dist/wireErrorTriage.js +59 -5
  48. package/dist/wireFailureShape.js +27 -22
  49. package/docs/INTEGRATION-CLIENTS.md +192 -13
  50. package/package.json +3 -3
package/CHANGELOG.md CHANGED
@@ -49,6 +49,46 @@
49
49
  > 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
50
50
  > commit 漏转在发布前就红,不再靠人记。
51
51
 
52
+ ## 0.82.1(2026-09-24)
53
+
54
+ ### Added
55
+ - **模型身份可解析判据**(CC-151):新增 `modelIdentityResolvability` / `modelIdentityDetail` / `modelSetupDecision` / `modelSetupNotice` / `observedEngineAnswer`。端按拓扑自报读数(本地引擎的五条来路各 `complete` / `partial` / `absent` / `unreadable`,远程引擎的应答观测;本地模型目录文件那一路在引擎启动之后才读,型上没有 `complete`,运行期报了按判不出处置),包答 `resolvable` / `not_resolvable` / `unknown`。有网关与凭证却没有模型名的来路不算可解析;没报或读不出的来路让答案成为 `unknown` 而不是 `not_resolvable`;远程引擎从不判 `not_resolvable`,也不拿本地配置替它作答。首启决策由「可解析三态」与「本端能不能配模型」两位决定,`unknown` 一律不弹配置向导。来路的读法留在各端。接入文档 **§93 S-11**。
56
+ - **每一轮与终局带上供应商自己报的模型名**(CC-141):引擎在供应商回答时报出的名字与请求所用 id 不同时,本包逐轮在用量那一路(`turn_usage._sema_response_model`,chrome `last_turn_usage` / `subagent_turn_usage` 的 `responseModel`)、终局在带终局记录的结果上(`_sema_response_model`,成功与失败都带;引擎只发失败事件、没有终局记录的那一形不带)原样给出;名字相同或没报时这一位缺席。`result.model` 不变,仍是引擎选定的那个 id。通知码册同批加 `route.response_model_mismatch`。接入文档 **§93 S-8**。
57
+ - **文件历史捕获的读口**(CC-163):新增 `projectFileHistoryCaptureCapability` / `noteEngineCapsForFileHistoryCapture` / `observedFileHistoryCapture` / `forgetFileHistoryCaptureReading`、判据 `fileHistoryCaptureMode` 与措辞单源 `fileHistoryCaptureDoctorDetail`,读引擎在能力表里报的 `fileHistoryCapture`(`off` / `on-always`)。键缺席 ⇒ 判不出(老引擎),不折成 `off`;不认识的新词原样收、判据答 `unknown`。`off` 那句只说「不捕获」,能不能回退代码仍由 `/rewind` 那组读数回答。接入文档 **§93 S-4**。
58
+ - **续跑被「准入政策」拒时的专属一句**(CC-154):新增 `resumePolicyBlockFromError(err)` 与 `resumePolicyBlockContent(detail)`,读引擎把续跑拒收折成的 `resume_blocked_by_policy`,并把折叠时带上的原码分三态 —— 给了原码 / 这次拒绝没带原码 / 带了但读不出。措辞说清「引擎在准入检查处拒了续跑」并带出原码,但**不**断言此前有没有东西已被消费、也**不**断言重试会不会过(这两件本读数都不知道),指路先看任务状态、反复出现找运维。按原码有专属说法的拒绝(如场景不在指派里)照旧由那一口认领,本句是它们都不认领时的兜底。接入文档 **§93**。
59
+
60
+ ### Fixed
61
+ - **没有执行车道的部署,诊断行不再说「工具在另一台机器上」**(CC-153):引擎在未配置执行车道时会报 `in-process`,此前诊断行把它和「工具在远端机器」说成一回事。现在单独说明本部署没有执行车道、引擎自带的改文件与跑命令工具不挂载;某一次运行到底有没有工具,仍看工具名册。接入文档 **§93 S-3**。
62
+ - **计划复核的决断还在路上时,不再给同一道门重开一张卡**(CC-160 ②):同一任务的计划复核决断在飞(请求还没落地,或去键重发在途)期间,`reopenPlanReviewCard` 现在不重开、答 `{reopened:false, _sema_decisionInFlight:true}`(同步形与回执形都是;宿主经 `deliverDecision` 自带投递口时,这道闸看不见那一发在飞);修前它不看这件事,于是用户刚作答、决断还在路上时,自愈链能再弹一张同题的卡,用户会对同一道门答第二次。决断落地之后的重开照常,没有冷却。接入文档 **§93**。
63
+ - **自愈提示不再教刚作答的用户再决一次**:会话被占、自愈链去重开那张门卡时,重开口若因「这道门的决断还在路上」拒开(判决带 `_sema_decisionInFlight:true`;本包计划复核重开口现在这样答,端侧自己的重开口铸同一位即可),结局仍是 `plan-review-reopen-failed` / `ask-reopen-failed`,另带 `decisionInFlight:true`;提示改说「你的答案还在送往引擎的路上,没有再弹卡,稍后重发」,不再说「没能重开」,也不再给直接决断的入口(那会是第二次决断)。这种情况下也不会进入会取消运行的陈旧停泊处理。steer 排队回执与系统注入件两处读同一份判决,措辞同步。接入文档 **§93 S-15**。
64
+ - **决断失败时,被「准入政策」拒的那一种说成人话**(CC-154):审批卡的批准 / 拒绝、提问卡作答、中断时撤卡的警告、计划复核的结局,这几处在引擎把续跑拒收折成 `resume_blocked_by_policy` 时,此前只给一行机器形原句;现在补上 `resumePolicyBlockContent` 那一句并点名原码。别的失败文字逐字不变。接入文档 **§93 S-5**。
65
+ - **提问卡 / 文件审批停泊腿不再把「被政策拒」「记录读不出」误判成「这道门早已决过」**:这两种失败(外层码 `resume_blocked_by_policy` / `corrupt`)门其实还在;修前失败文字里的原句或行坐标恰好含 `not found` / `resolved` / `already` 时,停泊腿会空转一轮重连,再以一句与真因无关的「坐标对不上」收场。现在这两个码不再按文字判。端侧若自己用 `isAlreadyResolvedGateReason` 扫失败文字,请在那之前调新增的 `isGateStandingErrorCode(errorCode)`(真 ⇒ 门还在,不是已决),包内停泊腿用的是同一只。接入文档 **§93 S-6**。
66
+ - **非交互(`-p`)车道:Bash 良性非零退出不再被标成失败**(CC-168):结构化结果整只缺席时(大输出会让它被丢弃),交互转录早已按模型面首行的退出码说明判良性,`-p` 车道的 tool_result 却只认结构化那一位,于是同一次良性退出在 `-p` 被标 `is_error:true`。现在两个车道用同一只解析器,标准形上给同一判定(结构化结果不带类型、只带退出码这类非标准形上,两车道的判定仍各自实现,可能分叉);文本不是完整的 Bash 框架时照旧按退出码判。接入文档 **§93 S-7**。
67
+ - **非交互(`-p`)车道的工具结果帧补上结构化结果**(CC-149):`toolEndResultToUserFrame` 此前只铸 `tool_result` block,信封上的 `tool_use_result`(CC SDK 面的结构化工具结果)与降级标记 `_sema_degraded` 整段缺席,于是同一次工具结果在交互转录里有、在 `-p` 流里没有,读流的一方只能从给模型看的正文里去猜。现在值与交互转录出自同一只铸口;两位只上信封,block 的正文与 `is_error` 不变。过渡名 `toolUseResult` 不上 `-p` 帧(0.83.0 退役;需要时从 `tool_use_result` 同值起别名)。待办清单的富结果要开卡时的入参,`-p` 帧拿不到,这一形如实回落正文体。子代的工具结果(`parent_tool_use_id` 非空)值照算,但不写主会话的通知台账(在飞 workflow、在飞后台任务、已通知);`wireOutputToBody` / `structuredToToolUseResult` 为此多一个可选末位 `leaderLedger`,缺省与修前相同。接入文档另附一张 `tool_end_result` 臂键逐键去向表(臂 → `-p` 帧 → 交互记录,含缺席语义)。接入文档 **§93 S-9 / 93d**。
68
+ - **被拒分类词在两条车道上同判**(CC-172):门记录的结算段在场但读不出(或分类器成因 / 分类归属在场但读不出)时,`-p` 帧与读器在原始帧上都如实缺席,交互转录却铸出一个自信的类别(「规则拒」或「分类器拒」),会把人引去找一条不存在的规则。现在类别只在原始帧上判一次:判不出时内部臂带 `_sema_denial_unclassified:true`,`ccToolDenialKindForToolEnd` 见它答缺席,两条车道同一只帧同在同缺、同词;对原始帧的判定不变,直接喂原始形臂的自建管线行为不变。接入文档 **§93 S-10**。
69
+ - **非交互车道能看到子代跑过的工具调用了**(CC-158):宿主在 `EmitContext.lane` 声明 `'print'` 后,子代的工具调用与结果按 CC 形进入主输出:`assistant` 帧与工具结果帧都带 `parent_tool_use_id`,值为发起子代的那次调用。子代的正文与推理仍不进主输出;子代 `assistant` 帧不带 `message.model`(本包不知道子代用的是哪个模型)。不声明车道时 `runStream` 的行为不变;直接调用 `eventToSdkMessage` 的自建管线,子代 `tool_start` 的产出随本版变化(`parent_tool_use_id` 由 `null` 变为父调用 id,不再带 `message.model`)。修前这类调用在非交互输出与转录里完全缺失。接入文档 **§93 S-12**。
70
+ - **终帧被拒清单不再漏掉没问过人的拒绝**(CC-159):清单此前只从人审账本派生,规则、hook、分类器、写保护这类不问人的拒绝都不进那本账,于是混合的一发里少列,还声称清单完整。现在终帧会对账本流见过的被拒调用,补上账本里没有的那些:三键齐全的进 `permission_denials`,缺字段的只进 `_sema_permission_denials` 并标记清单不完整。账本缺席时照样对账。人审账本里没有、由流内补上的那一次调用,若既收到拒绝、又被本地中止(打断或连坐取消),按合并后的类别判为中止,不补这一行,两种到达顺序同判;两只收口帧的归因互相矛盾而类别读不出时照补,人审账本里已有的拒绝行也照列(见接入文档 KL-52)。不向本包交被拒调用表的自建管线,行为不变。接入文档 **§93 S-13**。
71
+ - **投影输入变化后的第一轮不再带旧的自定义代理载荷**(CC-161):新增 `invalidateTaskAgentsWire()`。端在投影会读的输入变化时(例如关掉自动记忆、切换工作目录、重载代理定义)调用它:缓存载荷当场作废并按新输入重投,既有就绪门 `awaitTaskAgentsWire` 会等这次重投落定(上限不变;超时则那一轮不带自定义代理,而不是带旧的)。修前同步读面交的是上一次缓存的投影,变化后的第一次提交里,代理的系统提示仍按旧输入拼成,与同一请求里的新设置互相矛盾(下一轮自愈)。更早一次重投晚到时,不再覆盖更新的结果。能力探测还没答复时作废,重投会先等这台引擎答「收不收」再写;换引擎时上一台的判定当场作废。重投挂住一次后,本代任何一次结果落定(包括读面踢起的后台刷新)即放行,不再每轮等满上限;挂住的一代被新一代取代时,正在等的就绪门改等新一代;引擎确证不支持时就绪门不等。`taskAgentsField()` 的同步形不变。接入文档 **§93 S-14**。
72
+ - **投影闭包同步抛错时,自定义代理的同步读面不再抛出**:`taskAgentsField()` 在后台刷新时直接调用投影闭包,闭包若同步抛错(而不是返回被拒的 promise),此前会从读面抛出,导致请求组装整个失败;现在只算这次刷新失败,读面照常返回缓存。接入文档 **§93 S-14**。
73
+
74
+ ## 0.82.0(2026-09-23)
75
+
76
+ ### 🔴 删键预告改期(本版不删,0.83.0 删)
77
+ - 0.81.0 预告「0.82.0 删 23 个键 + 1 个子型」(CC 形消息上本包自铸或从 CC 内部转录面搬来的顶层键,含过渡键 `toolUseResult`)**改到 0.83.0**,按维护方裁定。本版这些键**照旧在场、与 0.81.x 逐字同形**,读它们的端在 0.82.0 上零改动;逐键终态(删 / 改 `_sema_` 前缀 / 映到 SDK 面同义位)与处置清单在 0.83.0 之前单独发布,读点多的键会逐一点名。接入文档 §92y ㉙。
78
+
79
+ ### Added
80
+ - **`/rewind` 能不能回退代码,改由引擎自己的能力位回答**(CC-146):新增只读读面 `observedRewindArchive` / `noteEngineCapsForRewindArchive` / `forgetRewindArchiveReading`(`/v1/capabilities` 的 `resumeAt`=能不能按消息分叉对话、`rewindFiles`=分叉同步回退跟踪集、`rewindFilesTo`=只回退代码不分叉、`restoreFiles`=服务端认不认当前请求键拼法四位),判据单源 `rewindCodeArchiveAvailability(reading, 'both' | 'code')` 答 `available` / `unavailable` / `unknown`,以及措辞单源 `rewindArchiveDoctorDetail` / `rewindRestoreFilesEpochDetail`。消费端此前用「本地备份表非空」判这件事,而那张表只由消费端自己进程内的工具填 —— 引擎执行工具的会话上它恒空,于是两个回退代码的档**从不出现**,而引擎那边文件历史一直在记。「回退对话并连同代码」这一档要**两件都在**(对话锚与文件历史),上游只给「只回退代码」那一档合了锚条件、没给这一档合取位,所以合取在判据里做:任一明说没有 ⇒ 不能,全部明说有 ⇒ 能,其余判不出 —— 只看文件历史会在「有文件历史、没有锚存储」的部署上把这一档说成可用,而请求必败。每位四态:引擎说有 / 说没有 / **没报**(老引擎)/ **报了读不出**,后两者一律 `unknown`,既不折成「不能」也不折成「能」;一位读不出不连坐其余几位;`unknown` 的处置是照常给档、不承诺结果。键名由编译期不变量绑在上游 `Capabilities` 的**显式声明**成员上(剥掉索引签名 —— 不剥则该不变量对任何串恒真)。`restoreFiles` 位只进诊断句、**不**改请求键拼法(旧拼法在现役引擎上被直接拒)。同票追加「只回退对话」那一档的判据 `rewindConversationAvailability(reading)` 与措辞单源 `rewindConversationDoctorDetail(reading)`:只看对话锚位,三态同上;锚明说没有时两个回退到消息的档一起不可用,锚有也不代表「连同代码」那一档可用。「连同代码」那一档在「没有文件历史」时的诊断句不再顺带断言「对话还能回退」—— 对话锚判不出时那句话没有依据,对话能否回退由新判据回答。接入文档 **§92**。
81
+
82
+ - **续跑被拒时,判断按「原本是什么错」而不是「被折成了什么错」**(CC-154):引擎自 7.95.0 起把几条**续跑路径**(审批卡作答后的续跑、唤醒、计划复核等,都从提交时保存的上下文重建请求)上的受理拒绝**统一折**成一个「受理策略变了」的错误,**原本的错误码被放进另一个字段**,附加材料原样随体。新增取码单源:外层正是那只折叠码且里层带着原码时**答原码**,否则答外层码 —— 折叠在与不在**同一份读法**,上游把折叠收窄回去那天本包零改动跟随;外层不是那只折叠码时,里层同名字段**不夺话语权**。受它影响的现有读口有一处:「场景不在你的指派列表」那句专属文案 —— 挂起与续跑之间收紧了场景指派时,它此前在续跑路径上认不出、用户只拿到一句兜底错误,现在经取码单源读得出原码与可用列表;豁免只给折叠形,非折叠的同状态错误照旧不认。**射程**:主对话提交不经这只折叠,那条路径上「恢复点失效就去掉它自动重发一次」的行为不受影响。接入文档 **§92**。
83
+ - **装配期就能知道「这一 run 有没有手」**(CC-152):新增 `handsMountedFromManifest` / `handsMountedDetail`,并给工具名册的行加一格 `mountedBy`(「是哪个挂载条件让它上车」;上游闭集十一词,当前 SDK 未声明这一位、按结构读、认词不收窄)。装配清单里**一行带这一条件的工具都没有**,意味着**这一条腿**上模型手里**没有引擎自带的**改文件 / 跑命令工具(清单按腿铸:委派出去的子腿可能挂了这类工具,由别的条件挂上来的外接工具也可能写盘 —— 这一读数不覆盖它们,也不断言改盘不可能);而这件事在**终态层说不出口**:引擎的终态词汇说的是「这趟跑完了」,不回答「做成没做成」,所以等到终局再判的编排器无从判起。读数三态,两个方向都不折:名册**读得出**且这样的行恰零条,是上游在陈述一个事实;而**根本没有名册**不是那个事实 —— 清单的静态半场从不带名册,更老的引擎给名册但不报挂载条件,那里的「零条」是关于读者的陈述而不是关于这趟跑的。四种判不出各有各的说法,每一句都不会说成「这趟没有工具」。接入文档 **§92**。
84
+ - **云控制面额度读数的三分**(CC-150):新增 `projectRegistryQuotaUsage` / `quotaAmountText` / `registryQuotaDoctorDetail`。`GET /api/v1/quota/usage`(principal 视图)的回体里 `null` 有**两种正面事实**并且互不相同 —— `tokenQuota: null` = 本实例没给这个主体配额度、`limit` / `remaining: null` = 此窗不设限;而**键缺席或类型不对**才是判不出。三者在读数上分得开,**没有一档折成 `0`**(措辞里 `no cap` 与 `unknown` 两句零个数字):把「不设限」渲成 `0` 会让用户以为一滴不剩,把「没配额度」渲成判不出会让这条真话消失 —— 而本票的起因正是命令行**看不到**中心配好的额度。`used` 没有「不设限」这一档(上游恒给数),那一格上的 `null` 只能判不出;`resetAt: null` 只有在上游明说「没耗尽」时才是「没有恢复时刻」这条事实,否则只是判不出(耗尽与否与有没有恢复时刻是两个判断,不共用一位)。取回体那一步仍在调用方(子路径入口是零加工转口面,不放带判断的读器)。回体里的 `budget` 与 `latestRequest` 本版不投影(缺席语义未读到铸点)。接入文档 **§92**。
85
+
86
+ ### Fixed
87
+ - **几只错误读口在「附加键」读不出时不再抛出**:从错误对象上取附加字段(场景拒绝的允许列表、续跑拒绝里的等待秒数与运行 id、折叠码背后的原码、读不出的那一行坐标)的那一只共用读法,此前在对象被撤销的代理上会把异常抛进调用方的错误处理路径;现在读不出一律当缺席。
88
+ - **会话存档的续跑上下文读不出时,不再丢掉引擎点名的那一行**(CC-156):引擎此时回答「这一会话存下的续跑上下文读不出」并在 `where` 里给出读不出的那一行,原句还叫人去看 `where` —— 而计划复核的决断、耐久审批的批准 / 拒绝、提问卡的作答,用户中断时撤掉挂起卡片的那一次拒绝,以及忙碌会话里选择「插话」而那一轮恰好已经结束的那一次,这几条路径此前只转述那句话、把坐标丢了;turn 错误判决的 `http` 臂也只投状态、码与原句。新增读口 `corruptStoredRowWhere(err)` 与措辞单源 `corruptStoredRowContent(where)`(坐标先转义控制字符再封长),turn 错误判决的 `http` 臂多一格 `storedRowWhere`(出现即非空,值为原文,渲前需转义);上述几条路径的失败说明(中断那一条是上屏的警告与日志)现在都带出那一行;读不出坐标时逐字同旧。计划复核这一形同时改判为「没有生效」(引擎在读到上下文之前不会落定这次决断),不再说「无法确认」。已知限制:审批卡与提问卡在同步车道上仍以原终帧收尾,这时坐标只写进宿主日志,屏上看不到(KNOWN-LIMITS KL-40)。接入文档 **§92 S-12**。
89
+ - **自动模式分类器拒掉的工具调用,不再被标成「规则拒」**(CC-155,订正 0.81.1):0.81.1 给「没问过就拒」的调用补上被拒类别时,把分类器那一层也并进了 `permission-rule`。现按门记录里拒的**形**分三词:分类器自己的裁决 ⇒ `automode-blocked`,分类器没跑成而 fail-closed ⇒ `automode-unavailable`,分类器答了但读不出裁决 ⇒ `automode-parsing-error`;由上游策略链带着分类器成因拒的,同按分类器词;成因词不认识或读不出 ⇒ 整键缺席不猜(只在分类器与策略链两层上;其余层照旧按层名);这些只在「帧上没有本地类别、门记录里没有结算」时适用,结算词与本地类别照旧优先;策略链的拒不带成因、却带着分类归属时,记录分不开「分类器自己判的」与「分类器放行后别的策略拒的」⇒ 整键缺席;其余各层照旧 `permission-rule`(已知限制见 KNOWN-LIMITS KL-39)。user 工具结果记录与终帧被拒清单同一只读数,交互与 `--print` 两条车道同批受益。在 0.81.1 上这类拒绝会被说成一条规则拦的,用户会去找一条不存在的规则;更早的版本上它是缺席。接入文档 **§92 S-10 / ㉘**。
90
+ - **文档订正:「工具结果两键同缺席」这一读法作废**(§92y ㉗;§90b 第一条与 §90c U-G1 后半):结果两键只在有值时铸,而回落链在空输出上产出空串 ⇒ **两键恒在场**。空输出有两种值形 —— wire 既无结构化输出又无正文 ⇒ 值 `''` 且带降级标记;正文是空串 ⇒ 值是该工具的退化结构体、不带降级标记。按「键在不在」判「这张卡有没有结果」两种都会误判;新判据看**值**与降级标记。本版无代码改动,判据与文档改口。
91
+
52
92
  ## 0.81.1(2026-09-22)
53
93
 
54
94
  ### Fixed
package/README.md CHANGED
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
35
35
 
36
36
  ## Scope
37
37
 
38
- **Version:** 0.81.1
38
+ **Version:** 0.82.1
39
39
 
40
40
  - **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
41
41
  B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
@@ -303,14 +303,14 @@ guard still cross-checks the table by name).
303
303
  | `scripts/run-terminal-cause-projection-test.mjs` | The `7.64.0` wire reshape, projected. A run's ending stopped being eight parallel flat keys and became **one tagged cause** (`completed \| failed \| blocked \| paused`), and a tool call's gate stopped being four orthogonal words and became **one record** (`disposition` / `settlement?` / `origin?`). Both are read in exactly one place in this package, and this guard pins them at **two levels**, because the dangerous seam is "the reader was updated, the consumer was not": each terminal arm is checked on the reader *and* on the `subtype` / `is_error` / `errors[]` the projector actually emits. Two properties carry most of the weight. First, a terminal word this reader does not know is **never** laundered into an empty success — it lands on an `unknown` arm carrying the word verbatim, while a payload with no terminal word at all (the mock lane) keeps the success arm exactly as before, which is the one and only case the reader answers `null`. Second, the three window words (`approval_window_expired`, `denial_limit_window_expired`, `park_sla_expired`) must each be told apart by a different predicate: the previous generation collapsed all three onto one `timeout`, and re-merging them would throw away the discrimination this reshape just restored. Two byte generations are read by one reader, keyed on the discriminator upstream nailed (`"terminal" in result`): the current cause form, and the **flat** form that a current engine still emits on two lanes — replayed persisted bytes, which the service passes through verbatim rather than back-filling, and the service's own rejection envelope. A cause-form payload that also carries stale flat keys must ignore them entirely: keeping one compatibility read is what gives a single fact two sources. The same file also pins the MCP delivery verdict and HTTP status riding the wiring manifest, the four-state write-protection reading (where three of the four states mean *cannot tell*, and none of them may be printed as "there is no table"), and the park-reopen fetch identity: that predicate is asserted through the **real entry point**, since the defect being fixed was precisely a call site wired to a different predicate than the one that routed the row there From 0.80.0 one of those three boundaries flips: the key naming **who settled a refusal** stopped being a dead byte and became part of the wire, so the check stopped scanning the build output for the word and started reading the request bodies the two decision legs actually send. A refusal attributed to the deployment's own policy carries the word; one attributed to a person, one with no attribution at all, and one carrying a word the vocabulary does not hold carry nothing — the wire has no slot for “a person decided this” other than the key's absence, so inventing one would be minting a word upstream does not have. The allow family never carries it on any of its routes, because that combination is refused before the approval is judged while the side effects of allowing have already landed, and the three refusals nobody was asked about (a card that failed, a user who walked away, an interruption) carry nothing either. A deployment that signs the bodies it accepts does not sign that word, and there is no capability bit to ask beforehand, so a refusal on exactly that ground is answered by re-sending the same decision once with that one key removed — byte-for-byte the same otherwise — rather than letting an optional note take the whole denial down with it. The guard measures that along three axes: the decision still lands and is reported as decided with the attribution handed back and a separate flag saying it never reached the wire; a caller who aborted in between gets no second request; every other refusal code, and every decision that never carried the key, send exactly once. The classification of a second failure is made from what the second body actually carried, not from what the card asked for. |
304
304
  | `scripts/run-auto-mode-unavailable-test.mjs` | The fact behind "you are being asked because the auto-mode classifier could not run", and the one place its sentence is minted. The cause table is a **copy**, reconciled word for word in both directions against the installed engine's own bytes — it narrowed upstream, and the guard follows rather than keeping the old shape: a table checked against something nobody ships any more is the oldest way for a guard to be green and wrong. The retirement is held from both sides — the removed table must really be gone upstream, and the removed reader and word must really be gone here — while the word that left keeps arriving cleanly from an older engine, because the reader takes the cause as an **open set**: the vocabulary belongs upstream, so a copied list here would discard a legal value the day one is added, and the value discarded is precisely "this outage is a NEW kind". The reader's one exclusion is the word the engine says it never stamps here — the classifier did run and did answer, just outside its contract, so reading it as a failure would invent an event the engine denies. That exclusion used to be derived from a second table which no longer exists; the reason for it never lived in that table, so it is now stated where it actually comes from, pinned as a **named** set (a magic literal scattered through the reader reds) and cross-checked against the engine's own verdict declaration and against the reader having exactly one such comparison. One reader serves both the live ask and its durable parked twin, since the two carry the same key path and a second copy is how two ledgers drift apart. Absence is pinned as absence — most asks never consulted a classifier at all — and the sentences are checked mutually distinct, prototype-safe, and walked end to end: an unknown word reaches the sentence a person reads (the fallback that names it verbatim) and the status reading (unavailable for this round, never a fallback to "available"), with counter-controls proving neither assertion is vacuous |
305
305
  | `scripts/run-engine-notice-catalog-test.mjs` | The engine-notice catalog and its audience table. Whether a notice deserves a person's attention is not decided by whether this end happens to have a phrasing for it — that drifts with each client's build order — but by whether the engine minted the code into its own written catalog; the audience row answers the separate question of *who* the fact is for, since an operations fact pushed at an end user is noise and a user-facing fact buried in an operator log is something withheld from the person who could act on it. Both tables are reconciled against the installed engine's own artefacts in both directions and pinned in lockstep with each other, unknown codes fall back to the conservative operator side, and catalog membership is tested on the raw value so a code carrying control characters cannot impersonate a registered one after sanitizing. The reader for a dropped MCP injection keys on its own code alone and treats a missing session, server or reason as absence rather than throwing at a read site. A reverse pin enforces the upstream's single-mint contract: the engine composes those sentences from the host's facts, so a copy of them appearing in this package's source or build is a second source that would drift, and fails |
306
- | `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever |
306
+ | `scripts/run-tool-roster-projection-test.mjs` | The leg's tool roster — what the engine says it actually mounted and what face each tool wears — replacing three word lists that were only ever an estimate taken from one traffic capture against one pinned engine. The reader copies the engine's own all-or-nothing discipline: a roster whose row cannot be read, or whose declared count disagrees with the rows, is dropped whole rather than handed over short, because a consumer reading a short roster concludes the missing tools are not mounted — the upstream says in as many words that this is worse than sending nothing. A malformed *face* on a row (path target, render hints) drops only that face, since a face is not an identity. Shims are built strictly from roster rows and never guessed from a tool's name, and an axis that cannot be read stays absent rather than defaulting to `false` or `never`, which would render "unknown" as "safe". For run-time changes the guard pins the one hard rule in the contract: a digest that does not match is **not** a rejection — the carried roster is the new state regardless and only the summary becomes unusable, because refusing the swap would leave the consumer holding a stale roster forever. One reading here answers a question that the terminal state structurally cannot: whether this run was assembled with any file-and-shell tools at all. The engine's terminal vocabulary says a run finished, not whether the work got done, so an orchestrator that waits for the end and then guesses has nothing to guess from — while the assembly manifest already said it at the start, one row per mounted instance with the single condition that mounted it. The reading is three-state and both folds are refused: a roster that is readable and carries no such row is the engine stating a fact, while no roster at all is not that fact — the static half of a manifest never carries one, and an older engine reports rosters without naming the mount condition at all, where an empty count would be a statement about the reader rather than about the run. Those two are kept apart in the reason the reading carries, and the wording for every unknown case is checked never to claim the run had no tools |
307
307
  | `scripts/run-permission-rule-issue-codes-test.mjs` | The rule-lint refusal codes an engine reports when it will not compile a permission rule. The SDK publishes neither a schema nor a type for them, so the package mints the table from the engine's own bytes and the guard pays the cost of that copy instead of leaving it to somebody remembering: it parses the codes the engine actually mints and reconciles them against the table in both directions, so a code added upstream (the user would see a bare code) and a code only the package believes in (a branch that can never fire) both fail. It also reconciles the table plus a small retired ledger against the engine's declared union, which is deliberately not the same set — one member was renamed and its old name is still declared — so reviving a code the engine will never mint again is impossible and a future stale member shows up immediately. Sentences are pinned one per code, mutually distinct, and split by family: a rule that is wrong and a rule that is legal but unsupported on this lane are different next steps and may not share a sentence. The engine's own message rides along as prose — sanitized and capped after escaping, never matched on |
308
308
  | `scripts/run-gate-vocabulary-test.mjs` | The two gate vocabularies — who denied a call (`DeniedBy`, nine words) and who asked about it (`AskOrigin`, eleven) — together with the one place their sentences are minted, so the same denial does not read three different ways across three clients. The tables are copies, not opinions: the gate parses the members straight out of the installed SDK's declarations and reconciles them against the package's tables in both directions, so a word added upstream (nobody renders it, the user sees a bare code) and a word only the package believes in (a branch that can never fire) both fail. Every word must carry its own literal sentence and no two may collide, including the sibling pairs the upstream deliberately split apart — an organization store and a personal rule store being unreadable send you to different people, and the two tighten origins exist precisely to name which layer of engine logic asked. The two fallbacks are pinned distinct because the sets differ in kind: one is genuinely closed on the wire (an out-of-set record is withheld by the engine, so reading one means the record is damaged) while the other is genuinely open (the server only checks for a non-empty string, so an unknown word just means the client is older than the engine) Alongside them sits an **uplift anchor** rather than a third table: the reason a call was decided the way it was is a distinct semantic face from who denied it and who asked, one upstream has not mirrored into the SDK at all, and one whose newest member — a shell command allowed because it only reads — has no sentence anywhere yet. Minting the union here would create the second drifting source the day upstream publishes it, so the guard instead asserts the **absence** from both ends: the SDK declarations carry no such union near that word, and the installed engine’s own list does not carry the word either. The engine end fires first, on the batch that raises the dependency, which is exactly when the ownership question should be answered; the SDK end fires when the mirror lands. Either red is the work order to mint the sentence, never a reason to delete the anchor. A fourth mint now sits beside the three tables and is not a table at all: a single presence-only fact — that no saved rule and no standing posture can retire this question — earns one sentence, taking no argument precisely so a caller cannot mistake it for a second kind of mandate, pinned distinct from every sentence the tables mint, pinned never to point at rule-writing, and pinned not to overclaim the stronger neighbouring demand that a person rather than a configuration must answer A fifth table joins them from 0.80.0: the thirteen words for **how a wait ended**, mirrored in both directions from the engine's own declarations — the table's owner — with the wire SDK's copy held alongside as a second witness that must match it word for word and in order, so the day the SDK falls a generation behind, that is what turns red rather than the mirror silently following the wrong source. The newest of them says a deployment's own policy answered the card — not a person, and not “nobody could be asked” — so the guard pins it apart from both neighbours by behaviour, feeding every one of the thirteen words through all five named predicates and checking which word makes which one speak, rather than what any predicate returns. Two of the thirteen also decide how a refusal is filed in the session transcript; that mapping is minted once and reused by both of the package's own entry points, and anything outside those two words yields nothing rather than a guess. |
309
309
  | `scripts/run-engine-identity-test.mjs` | The engine generation anchors on `/health` (`pid`, `instanceId`, `startedAt`; engine >=7.67.0). `/health` is the one unauthenticated door and its heartbeat is always green, so "another host restarted the shared engine" used to be discoverable only by having some authenticated request hit a 401 first — a path that misreads a restart as a network fault. The reader narrows each anchor independently (one malformed field never hides the other two) and always hands back a reading object rather than an absence, because the caller is asking which anchors answered, not whether there was a response. The comparison is a three-word verdict, not a boolean: `unknown` when the two readings share no comparable anchor at all — an empty intersection means nothing could be compared, never that nothing changed — and the boolean convenience is pinned so that only `true` is an assertion. Any comparable anchor differing decides `changed`, so a reading whose `startedAt` matches while its `instanceId` does not cannot be waved through as the same life; precedence only decides which anchor gets named in the diagnosis |
310
310
  | `scripts/run-posture-knob-projection-test.mjs` | The three deployment knobs on the operator face (`serverGates.durableApproval` / `streamAskWindowMs` / `sessionAutoTitle`, engine >=7.67.0), each read as a value **plus who set it plus one operator-facing pointer** rather than a bare value — a bare boolean cannot answer why this particular machine is on this setting or how to pin it back, and a default that flips with the deployment shape is invisible without that. A worker too old to report readings still sends a bare boolean; the reader folds it into the same shell so consumers keep one branch, but raises a `legacy` bit, answers `undefined` from the machine-readable source accessor, and mints a sentence that contains no source word at all — claiming a source nobody reported is worse than admitting the worker cannot say. The other two knobs are honestly absent on such a worker rather than defaulted, a malformed side knob drops only itself while the anchor knob drops the whole reading, and the four sentences are pinned literally distinct so an operator can tell "not observed" from "not reported" from a real value. The last leg reads the installed SDK's `openapi.yaml` and `types.d.ts` directly, including a pin that exactly one knob on this face is numeric — the premise the millisecond-to-prose rendering rests on |
311
311
  | `scripts/run-terminal-facts-projection-test.mjs` | The four unconsumed terminal-receipt facts: `TaskResult.effectiveReasoning` / `effectiveMemoryScopes` are narrowed into `_sema_effective_reasoning` / `_sema_effective_memory_scopes` on the CC-shaped `result` (success and error envelopes alike; a malformed value mints nothing, never a default tier), the resume **reopen** family (`resume.env_failed` / `tool_unavailable` / `tool_contract_mismatch`) is a frozen closed set with a reader and three-sentence copy that is disjoint from the refusal and retry-later sets, and `routePairingVerdict` reads `ModelInfo.routePairing` as ok / broken / unknown without policing the open set. A fifth section pins the structured-output key on the success result: the CC-spelled `structured_output` is the only home for the value the wire calls `structuredOutput`. The camelCase spelling this package used to mint on its own — a misspelling of the CC field, not an additive field of our own — rode alongside it for exactly one release (0.79.1) and is **absent from 0.80.0 on**, pinned both by own-key and by `in`, so a consumer still reading the old name sees `undefined` rather than a stale copy. The wire position is read exactly once, so a value-changing accessor is only ever asked for its first answer; absence stays absence; a wire key that is present but `undefined` mints nothing, since a key whose value is `undefined` makes a consumer that tests presence read "the engine produced nothing" as "the engine produced an empty result"; falsy-but-present values such as `null`, `0`, `""` and `false` are still minted, and so are shapes that are not records at all — an empty array, a populated array, a string, a number, a boolean — each carried through by the same reference, because the shape of that value is decided by the caller's own schema and the package does not get to filter it; and the error envelope carries no such key, because the CC error arm has no such field. Which spelling CC itself declares is witnessed from the mirror's own syntax tree rather than a constant copied into the guard, so the day that field is renamed upstream the guard says so. |
312
312
  | `scripts/run-export-liveness-test.mjs` | Every runtime export in the public baseline must be **alive**: referenced by some gate, or explicitly registered in `scripts/export-liveness.json` as `contract` (consumed by a client with no gate yet), `internal` (an internal helper amplified onto the public surface by `export *`), `candidate` (with ticket + retire-by) or `retire` (dead; retire-by version). Registration is accounting, not exemption: a row for a name a gate already references is stale and must go, a row for a name no longer exported is red, `retire`/`candidate` rows go red the moment `package.json` reaches their retire-by version, and the row count only ratchets down. When the sibling client trees are on disk the consumption evidence is checked by name — a `contract` row's claimed consumers must equal the real set, and a `retire` name must not be imported by any client. Names that have already left the surface are kept in a per-version `removed` ledger: they must never reappear in the baseline or the registry, and the ledger's versions must not run ahead of the changelog. |
313
- | `scripts/run-wire-refusal-copy-test.mjs` | Two wire refusals read the same way on every client: a cancel's 409 carries one of two codes with opposite dispositions (`conflict.approval_settled` — someone else already decided, go read the result; `conflict.run_not_running` — nothing changed, send the cancel again), an unrecognised or codeless 409 is reported as such rather than guessed, and the submit-side 429 `usage.window_exhausted` is read as a waitable refusal whose wait is stated only when the engine supplied one. `ControlRouter.cancel` raises a distinct safety code for the retry-directly case. A third family covers the refusals that a deny's *settler note* can draw: a deployment that signs the decisions it accepts but does not sign that note, a body whose decision and note contradict each other, and a word the engine does not recognise. All three are refused **before** the approval is judged, so each sentence states plainly that nothing was decided and the approval is still waiting, and each names a different next step — none of them “send it again unchanged”, which would simply be refused again. Two of the three codes already mean something else in this package: one is shared with the plan-review leg, so the reader refuses to claim it unless the caller states that the note really was sent, and the other is split by the field the engine names, because without that field the same code means a review outcome was rejected for its content. The third sentence deliberately does not say the word was misspelled: upstream mints that same field for at least four different reasons, so the only thing it proves is that the engine would not take the settlement details and named which part — which is what the sentence says, and where it points. The three sentences are pinned verbatim rather than by keyword, because a keyword check passes a sentence that tells the reader to send the same body again unchanged, which is the one next step that is certainly wrong. The field itself is read from the bag the SDK keeps additional response keys in, not off the top of the error, since only the hand-built shapes a test would write carry it there. Both decision legs hand the reading back on their outcome, and a leg that never sent the note claims nothing. |
313
+ | `scripts/run-wire-refusal-copy-test.mjs` | Two wire refusals read the same way on every client: a cancel's 409 carries one of two codes with opposite dispositions (`conflict.approval_settled` — someone else already decided, go read the result; `conflict.run_not_running` — nothing changed, send the cancel again), an unrecognised or codeless 409 is reported as such rather than guessed, and the submit-side 429 `usage.window_exhausted` is read as a waitable refusal whose wait is stated only when the engine supplied one. `ControlRouter.cancel` raises a distinct safety code for the retry-directly case. A third family covers the refusals that a deny's *settler note* can draw: a deployment that signs the decisions it accepts but does not sign that note, a body whose decision and note contradict each other, and a word the engine does not recognise. All three are refused **before** the approval is judged, so each sentence states plainly that nothing was decided and the approval is still waiting, and each names a different next step — none of them “send it again unchanged”, which would simply be refused again. Two of the three codes already mean something else in this package: one is shared with the plan-review leg, so the reader refuses to claim it unless the caller states that the note really was sent, and the other is split by the field the engine names, because without that field the same code means a review outcome was rejected for its content. The third sentence deliberately does not say the word was misspelled: upstream mints that same field for at least four different reasons, so the only thing it proves is that the engine would not take the settlement details and named which part — which is what the sentence says, and where it points. The three sentences are pinned verbatim rather than by keyword, because a keyword check passes a sentence that tells the reader to send the same body again unchanged, which is the one next step that is certainly wrong. The field itself is read from the bag the SDK keeps additional response keys in, not off the top of the error, since only the hand-built shapes a test would write carry it there. Both decision legs hand the reading back on their outcome, and a leg that never sent the note claims nothing. When a decision fails because the engine cannot read a stored row, the row the engine names is read (from the error's top level, or from the SDK's bag of additional response keys where the real SDK puts it) and carried into the failure reason on all three decision legs, escaped for display. |
314
314
  | `scripts/run-tool-disclosure-progress-projection-test.mjs` | The two wire arms sdk 9.6.0 adds — `tool_disclosure` (name-only tool census: open-set `policy`, `thresholdPercent` absent ≠ default, `deferred`/`activated` full snapshots) and `tool_progress` (one frame, two beats: Bash ticks carry an output tail with `totalLines`/`totalBytes` that come and go together; other tools carry only `elapsedSeconds`) — project to neutral internal arms plus chrome arms. Required keys missing ⇒ `malformed`; bad optional keys drop only themselves; the sub-flow three-key gate keeps child frames off the leader lane; both arms are `required: false` in the arm table with duties stated (the output tail is untrusted raw and must never be fed back to the model). |
315
315
  | `scripts/run-mcp-panel-projection-test.mjs` | The `GET /v1/sessions/:id/mcp` panel reader (`projectMcpPanel`; server >=7.77.0 adds the optional `lastLegMcp` key) and the single wording mint for its "last leg" line. Absence of `lastLegMcp` is one literal sentence that never blames the engine version (a new session, a leg outside the retention window, a leg without a manifest and an older engine all look the same on the wire); a key that is present but unreadable is a different sentence plus a `lastLegMcpUnreadable: true` mark, never folded into absence. The `mcp[]` roster goes through the same reader as the live `wiring_manifest` third section, so a replayed roster and a live one have one shape. The two faces of the panel (`servers[]` and the last-leg roster) may legitimately differ, so the view carries no agreement flag and none of the five sentences mentions `servers`. Required keys are pinned to the SDK `openapi.yaml` component bytes **0.69.0:** `fetchMcpPanel` fetches the panel through the SDK client's own `sessions.mcp` call (same transport and auth as every other read) and projects it; transport failure, an unreadable body and an empty session id all come back as `undefined`, never as a fabricated empty panel 0.71.0 adds section K: `mcpEngineLegPresence(view)` — the engine-side MCP presence tri-state read only off the panel view (`unknown` when the view could not be read, never rendered as "no MCP configured") |
316
316
  | `scripts/run-absence-fold-census-test.mjs` | A package-wide census of the "absence folded into a positive outcome" defect shape, so that fixing the six sites this release does not merely move the shape somewhere else. The defect is defined by position, not syntax: a fallback position (the unconditional tail return, the `default:` arm, the literal minted when there is nothing to pass on, the value returned from an error path) may only say `unknown` or stay absent, never a positive word. Detection walks the syntax tree of every source file, so comments, strings and multi-line spellings cannot hide or fake a hit, and covers five forms: the right arm of `??` / `\|\|`, the else arm of a ternary, the first return of an explicit `default:`, a `catch` block or `.catch(() => …)` arrow returning a healthy value, and a function whose last statement returns a positive word after other returns. Every remaining hit must be registered with a written reason, an unregistered hit fails the gate naming the file and line, the registered count must equal the real count so a cleared site cannot leave a spare allowance behind, and the gate proves its own teeth behind a fence (a failed self-proof refuses to report any count): each form injected into an in-memory copy must add exactly one hit, two correct spellings are pinned as non-hits, and samples inside comments or strings do not count. It also pins the headline site: the fleet panel projection no longer mints an `end` with `isError: false` on absence |
@@ -338,7 +338,7 @@ guard still cross-checks the table by name).
338
338
  | `scripts/run-display-cap-order-test.mjs` | The order in which untrusted text is sanitised and length-capped, across every mint point that puts an engine- or database-supplied string on a screen. The sanitiser rewrites each invisible character as a six-character escape, so capping the **raw** string first and escaping afterwards hands the screen six times the width that was budgeted — a forty-character allowance becomes two hundred and forty. The guard does not hardcode that allowance, because each mint point wraps its field in different fixed prose and the prose moves: it anchors on the deciding quantity instead, feeding one benign and one control-character input of the same length through the same mint and requiring the second not to come out longer. That criterion is immune to wording changes and stays sensitive to the expansion, and it is `<=` rather than `==` on purpose — a correct escape-then-cap backs the cut off a partially-consumed escape token, so the control-character line is legitimately the shorter of the two, and demanding equality would score that avoidance as a regression. Each mint is bracketed by two positive controls (the input really reaches the screen; the cap really engages) and the expansion predicate is shown to turn red against a deliberately cap-then-escape reference, so an all-green run cannot mean the guard simply measured nothing. The shared mint point is checked directly for the two avoidances it owes — never splitting an escape token in half, which would leave something on screen that looks like the beginning of a complete answer, and never splitting a legal surrogate pair, which would manufacture the very lone surrogate the sanitiser exists to catch |
339
339
  | `scripts/run-seat-task-request-origin-test.mjs` | Where every field of the seat lane's send-message payload comes from, and whether it actually lands anywhere. The seat payload is a closed interface this package mints itself, and most of its fields are meant to ride verbatim onto the engine's request body — two facts nothing used to connect, so both directions could drift in silence. A seat field could be named after a request position that does not exist, in which case a client writes to it, the wire carries it, the engine ignores the whole key, and the screen shows a switch that does nothing; conversely a new request position could arrive with no seat to sit in, which is **structural** absence — the closed set *is* the carrier, so a decision missing from it has nowhere to be put at all, the same shape logged when the effort dial had no seat. The guard turns each field's origin into data: either it names the request position it forwards to, or it is declared seat-local with a written reason, and the two are mutually exclusive. Forwarding claims are then checked against the **installed** SDK's type declarations, parsed rather than restated — a hand-copied list of position names would only ever prove that two transcriptions agree. The parser is held to reading top-level positions only, since a nested option object's inner keys would otherwise be mistaken for positions of the request itself, and it proves that discrimination on synthetic input before any verdict is given. The two subagent fields carry a standing regression pin, and the retention window's inner keys are read from the declaration the same way, so a seat that offers a tunable window cannot offer one the wire has no room for |
340
340
  | `scripts/run-wire-auth-source-test.mjs` | **When** the outbound credential is read. A literal string is consumed at construction — the transport captures it in a closure and every later request reuses that one copy — so once the engine is replaced by another session and the credential rotates, a long-lived client keeps presenting the old one and the only way out is to rebuild the client along with everything hanging off it. The credential position now also accepts a getter that is called **once per outbound request**. The guard anchors on the deciding quantity, which is not "was the getter called" — reading once at construction and reusing the result would satisfy that too, and is exactly the shape being removed — but *which read produced the value on the wire*: it changes the getter's answer between two requests through the same client and requires the second request to carry the new one, and it requires construction to read the getter **zero** times. The three-state credential semantics are replayed per request rather than assumed: on loopback an unavailable credential sends **no** authorization header at all rather than a fabricated one, off loopback it sends the fail-closed anonymous identity so the deployment answers with an honest 401, and the guard shows a single client moving between those states across successive requests. A getter that throws is fail-soft — the request still goes out under the no-credential branch, because a broken credential port should not take the whole wire down, and the exception may itself carry credential material. The same-origin relay form is checked to stay out of the getter path entirely, and every request is checked to keep the credential in the authorization header only — never in the URL, never in another header |
341
- | `scripts/run-subagent-durable-divert-test.mjs` | The side-channel that keeps a **sub-agent's** content out of the leader's transcript, on the replay leg. A content frame stamped with a parent tool-call id belongs to a child, and rendering a child's tokens as the leader's own text is the pollution this divert exists to prevent — but the predicate only listed the four **live** frame shapes, while the durable leg replays the same segment in its **aggregated** form. Those frames fell straight through onto the main projection path, which is how a reconnect or a resumed session ended up with the child's answer printed as the leader's. The anchor is unchanged and shared: the parent tool-call id is what says whose frame this is, and whether the frame is an increment or a whole segment has nothing to do with whose it is — judging the two shapes separately is exactly how one of them got missed. Folding the aggregate into a synthetic increment would have been the smaller diff and the wrong one: an increment means *append*, so a segment that already streamed live and then replays whole would be counted **twice**. The two are kept distinct and the aggregate absorbs instead — a whole segment whose prefix is what the buffer already holds replaces it, which also makes a redelivery of the same frame idempotent, and a prefix that does not match falls back to appending both rather than deciding on the engine's behalf which version counts. Segment boundaries stay with the tool frames rather than moving into the aggregate arm, since closing there would turn a second replay of one segment into a second entry, and the increment arm is pinned to keep appending so a token run that happens to be a prefix of the next does not silently lose characters |
341
+ | `scripts/run-subagent-durable-divert-test.mjs` | The side-channel that keeps a **sub-agent's** content out of the leader's transcript, on the replay leg. A content frame stamped with a parent tool-call id belongs to a child, and rendering a child's tokens as the leader's own text is the pollution this divert exists to prevent — but the predicate only listed the four **live** frame shapes, while the durable leg replays the same segment in its **aggregated** form. Those frames fell straight through onto the main projection path, which is how a reconnect or a resumed session ended up with the child's answer printed as the leader's. The anchor is unchanged and shared: the parent tool-call id is what says whose frame this is, and whether the frame is an increment or a whole segment has nothing to do with whose it is — judging the two shapes separately is exactly how one of them got missed. Folding the aggregate into a synthetic increment would have been the smaller diff and the wrong one: an increment means *append*, so a segment that already streamed live and then replays whole would be counted **twice**. The two are kept distinct and the aggregate absorbs instead — a whole segment whose prefix is what the buffer already holds replaces it, which also makes a redelivery of the same frame idempotent, and a prefix that does not match falls back to appending both rather than deciding on the engine's behalf which version counts. Segment boundaries stay with the tool frames rather than moving into the aggregate arm, since closing there would turn a second replay of one segment into a second entry, and the increment arm is pinned to keep appending so a token run that happens to be a prefix of the next does not silently lose characters When the host declares the non-interactive lane, a sub-agent's tool calls and results are also forwarded into the main output with their parent tool-use id (the sub-agent's text and thinking still stay out, as in the reference CLI); without that declaration the output is unchanged. |
342
342
  | `scripts/run-subagent-content-budget-test.mjs` | The **byte** budget on the sub-agent transcript ledger. It used to be bounded only by *counts* — so many entries per child, so many children — and a count is not a budget when a single entry has no ceiling of its own: one tool result carrying an inlined attachment, or one long model answer, and a single slot sits on tens of megabytes. The guard anchors on how many bytes are **still held** after over-filling, not on whether truncation fired, because an implementation that flags the overflow without actually dropping anything satisfies the second and not the first. Dropping is required to leave a record — how much went and where the retained content now starts — and that record has to reach the render plan, because content that vanishes with no marker gives the reader a transcript shorter than what happened with nothing to say so; the record is one per child, updated in place, pinned to the front, and excluded from the budget it describes. Order matters and is checked: oldest entries go first and the live tail is trimmed only as a last resort, since taking the text the user is watching stream while older history survives is the wrong end. The total budget evicts a whole least-recently-used child rather than shaving every child, and the configuration surface is fail-loud on zero, negatives, non-finite and non-integer values — a silently ignored budget is the exact failure this exists to remove — with the rejection proven atomic so a bad second field cannot leave half a configuration behind. The defaults are checked to be a magnitude that can really be reached, since a number too large to hit is a field rather than a budget |
343
343
  | `scripts/run-subagent-usage-projection-test.mjs` | Per-subagent usage, split by task. The engine's final accounting carries the delegated spend as **one total** — tokens, turns, task count — and no per-task breakdown, while every sub-flow turn on the stream carries its own usage. This package used to fold that away at the leader/sub-flow divide (a child's output tokens must never reconcile the leader's response length), so a client showing a subagent's detail pane had nothing to print. The split table can therefore only be accumulated from the stream, and this guard pins what that costs. The two existing leader-only arms stay **byte-for-byte unchanged** — the new arm is additive and always carries the sub-flow's own lane proof, so a host cannot mistake a child's numbers for the session window. Attribution is by the engine's own originating-task id — deliberately not a second `taskId`, which the event identity does not carry and whose absence would silently collapse every child under one parent call — falling back to the parent call id; a turn that answers neither is dropped rather than filed under an invented row, because merging two children's ledgers is worse than missing one. Cache-read tokens are read from the **engine's own shape** rather than the mirrored one, since the mirror fills that member with zero when the wire omits it and reading it there would erase the difference between *not reported* and *no cache hit*. A turn that reported no usage at all still counts as a turn and still adds its zeros — the numbers are a lower bound, and dropping the round would make the bound less true, so the honesty bit rides on the row instead and is never spelled `false`; such a round still emits its live arm, because the frame that says "this round has no account" is the one a real-time consumer most needs and the easiest one to drop. The same honesty bit also survives a terminal that carries no statistics at all: what the stream observed is unioned with what the final record says, so a run that already reported an unmeasured round cannot come out the other end looking like an exact zero. Finally the table says whether it is **partial**, and that verdict is anchored on the quantity that actually decides it: the engine's own totals. Turn count and row count must both reconcile before the table claims to cover the whole run; anything else — including totals that cannot be read — marks it partial, so the failure direction is always the safe one (a complete table called partial, never the reverse). The two accounts are kept separate and are never added together or used to correct each other. One more thing the totals cannot settle: the row key has **two namespaces** — the originating-task id and the parent call id it falls back to — and nothing upstream promises they are disjoint, so the same literal can name one child's identity and another child's parent call. Accumulation therefore keys on the origin as well as the id; the delivered table still keys on the bare id, and a cross-namespace clash is merged into one row that says so, with the partial verdict forced, because a row count and a turn count can both reconcile while the attribution behind them is wrong. The table itself is likewise a **per-stream snapshot** handed to the terminal projector by value rather than left on the caller's context: the three terminal projectors are public, so a host may drive one run through the stream and project another's terminal directly on the same context, and a table left behind would be attributed to whoever projects next — silently called complete whenever that run's own totals happen to match. Without a snapshot, both table keys are simply absent |
344
344
  | `scripts/run-result-text-backfill-test.mjs` | What happens when the terminal frame's answer text and the text already on screen do not match. A turn's answer normally streams in and the terminal frame carries the same words again, so the two agree — but when the connection drops mid-answer and the reconnect brings the finished version, "this turn already produced assistant text" is true, the terminal fallback is skipped entirely, and the screen stays permanently short of whatever arrived while the stream was down, with nothing to say so. Four cases are pinned. Nothing on screen yet: render the terminal text whole, byte for byte the previous behaviour. On-screen text is a **prefix** of the terminal text: emit only the missing tail, and the guard measures the deciding quantity — the total bytes that reached the screen must equal the terminal text, which fails both for a missing tail and for a re-render that would print the first half twice; when the two are already equal, nothing is emitted at all. Terminal text is a prefix of what is on screen (an engine-side trim): touch nothing, since there is nothing missing and overwriting with the shorter version would erase what the reader already saw. Neither is a prefix of the other: emit **nothing** and raise a fact instead — which version counts is the engine's to say, and appending the terminal version after the streamed one composes a passage nobody ever wrote. That fact carries lengths rather than text, so a renderer is not handed a third version to choose from, and its declared duty is to *reword* the transcript line, never to render more. A cross-segment case proves the comparison reads the whole committed answer rather than the last segment, and the whole thing is driven through the real two-stage path rather than hand-built messages. The comparison only holds if both sides are the same kind of thing — one assistant message against one assistant message — and that depends on the client knowing where a message ends. A tool card is one place a message ends, but not the only one: when a model round finishes and the next one starts writing prose straight away, with no tool call in between, the two passages belong to two different assistant messages even though nothing visible separates them on the wire. Reading them as one used to glue the two passages into a single transcript line, running the end of one sentence into the start of the next, and then handed the terminal comparison a concatenation to check against the engine's **last** message — so a perfectly ordinary multi-part answer was reported to the reader as *the final answer does not match the text streamed above*. The end of a model round is therefore treated as the end of an assistant message: the pending prose is committed and a new message begins. That holds even when the round reports no spend at all, because *this round is over* and *this is what it cost* are two different facts and only the first one decides a boundary. A round belonging to a **sub-flow** decides nothing for the leader, and the test pins all three identity members, empty strings included, against a control that proves the same shape really does divide when no identity is present. The rest of the section is regression: a boundary landing immediately after a tool card must not lose the fallback that lets a card-ends-the-turn answer compare against the prose before the card; repeated boundaries with no prose between them must not mint empty messages or lose track of which message the comparison should read; a reconnect that brings the finished answer still backfills only the missing tail; the leg that already carries whole messages is not divided twice; and a segment boundary that arrives **after** the round ended still replaces the segment authoritatively and hands back the row attribution, without minting a second copy of the passage. A last group covers where a message boundary meets a segment that the engine announces late or not at all, and it pins only the half that is unambiguous. A round whose prose the engine never announces, followed by one it does, used to have the first passage **silently overwritten** by the second one's authoritative text — no transcript line for it and an empty attribution list, so a host had nothing to correct; it now keeps its own line and the attribution is handed over in full, with the offset measured on that passage rather than on the authoritative text. Which of the two passages survives a compliant host's rewrite is deliberately **not** asserted: the row named belongs to an earlier message, exactly as it already did at a tool-call boundary, and the underlying cause — the client splitting messages on its own boundaries while the engine announces segments on content blocks — is recorded as a known limit rather than pinned as a desired outcome. Where the second round's prose arrives as a whole block instead, there is no segment announcement at all and both passages keep their own line **in the order they happened** — previously the whole block was written first and the earlier passage only landed at the close, so the reader saw them reversed. An announcement that arrives after its round has already ended, and whose authoritative text merely extends what was shown, is pinned on the two things that are not in question: the bytes on screen add up to the authoritative text exactly once, and the divergence signal with its offset is still emitted so a host can reword. That ordering is unreachable on the installed engine — the announcement is pushed while the message is still being assembled and the round end only after it is complete, both through one synchronous dispatcher onto one queue — so the case is defensive; the late-announcement path exists because a tool call can be admitted before assembly finishes, which a round end cannot. Finally, one guard names a layer seam rather than a behaviour: the round-end frame is folded into a neutral usage arm that carries **none** of the three identity members, so a round belonging to a forwarded background child arrives with nothing to judge and the sub-flow cutoff cannot reach it. That guard reddening is the signal that identity now survives the fold and the cutoff has become effective |
@@ -348,7 +348,7 @@ guard still cross-checks the table by name).
348
348
  | `scripts/run-background-view-test.mjs` | `createBackgroundView` lifecycle: polling/notify pairing, per-source degrade (`501 → not-configured` vs `unavailable`), the capabilities `scheduler` probe, and dispose really aborting the in-flight fleet snapshot (pure projection lives in the pure suite's W-A segment) |
349
349
  | `scripts/run-fleet-view-keys-test.mjs` | The fleet projection views, **both directions**: `FleetTaskView`/`FleetWorkflowView` ⇄ their key lists (compile-pinned) ⇄ what a maximal/minimal row really projects, plus a wire-key coverage ledger (every `FleetTaskRow` key is either projected or carries a written reason why not) and a drift ledger against the shell's render contract. A one-directional assignability check is blind to optional keys — which is how `startedAt` was silently dropped |
350
350
  | `scripts/run-usage-verbatim-channel-test.mjs` | The two complementary usage disciplines (core 3.0.0 metering semantics): the CC `ModelUsage` mirror stays pure (five pinned keys, `totalInputTokens` has no seat), while the sema-owned channel forwards the engine `turn_end.usage` object **verbatim** (six keys, incl. `totalInputTokens`) via `last_turn_usage.engineUsage` / `handle.latestEngineUsage` — honest absence on pre-3.0.0 engines, no fabricated zeros |
351
- | `scripts/run-plan-review-decide-verify-test.mjs` | `decidePlanReview`'s post-decide honesty ([2315]/[2316], engine RB-471 family): a 2xx from the decide endpoint is **not** a terminal — the wire re-pulls the task status and words the outcome by the real shape (still-locked / legal new gate / genuinely left park / unverified), never claiming success it hasn't earned. Driven against a real fake-engine HTTP server through the shipped dist |
351
+ | `scripts/run-plan-review-decide-verify-test.mjs` | `decidePlanReview`'s post-decide honesty ([2315]/[2316], engine RB-471 family): a 2xx from the decide endpoint is **not** a terminal — the wire re-pulls the task status and words the outcome by the real shape (still-locked / legal new gate / genuinely left park / unverified), never claiming success it hasn't earned; when the engine answers that the session's stored resume context cannot be read, the outcome names the unreadable row and says the decision was not applied. Driven against a real fake-engine HTTP server through the shipped dist |
352
352
  | `scripts/run-shell-gate-durable-allow-test.mjs` | #110: the durable approval leg for **shell** gates. The tool_end HOLD/REJECT predicate must cover Bash the same way park detection already does (otherwise the park poison frame `Operation aborted` hits the transcript, `endedCalls` swallows the real replayed result, and the user who pressed Yes watches a command that really ran be reported as aborted); a replayed, already-decided park must resume reading the stream instead of being reported as a failed turn; `lastEventId` must track numeric `seq` too. Mutation-proven: each of the three fixes reverted turns the gate red |
353
353
  | `scripts/run-hitl-gate-honesty-test.mjs` | [2393] the four HITL disciplines that a passing type-check cannot see. (1) The park predicate and the `tool_end` predicate must cover the **same** set — the park side admits a first-class `kind:'tool_approval'` gate for *any* tool name, and a `tool_end` frame carries no `kind`, so the frame-level judge falls back to the engine's exact abort marker; otherwise the poison frame hits the transcript and `markEnded` swallows the real replayed result (the #110 disease, reopened on kind-only gates). (2) The already-decided identity criterion is **one-shot**: its two inputs are monotonic, so without consumption one successful decide makes every later park failure — including a real `approvals.list` outage — read as "already resolved" until the 24-hop budget runs out and reports a cause that has nothing to do with what happened. (3) A `plan_review` card dismissed without an answer must be re-presentable: the idempotent re-arm short-circuit re-publishes the still-armed card, and a stale armed id (responder gone) re-arms from scratch rather than presenting a card nobody can answer. (4) `HitlSafetyError` is a safety signal — the `remember` fallback arm must re-raise it instead of auto-retrying the decide, while a plain unknown-key 400 still falls back. (5) The polling leg reschedules after an escaping throw and flips `mode()` to `idle` once it consistently fails, so the honesty surface stops reporting a dead feed as live. (6) The live-frame leg carries the fact behind "you are being asked because the auto-mode classifier could not run" all the way to the card port. Transit narrows on SHAPE only — a non-empty cause string is taken verbatim, an open set, because the word table's owner is the engine and re-checking a closed table at the package boundary would drop a legal value the day a new cause word appears, which is exactly the information worth keeping. A malformed carrier degrades to absence rather than half-minting, and absence stays absence: it covers "the classifier answered", "this ask never qualified" and "this deployment has no classifier" at once, so nothing may render it as reassurance. The guard also pins the division of labour that makes the open set safe — the same word that transits is judged again by the public display reader, which narrows to the availability axis, so a word the engine says it never stamps on this fact renders no sentence while still being visible on the card for triage A later section pins the split this release introduced on the deny close-out frame. Until now every denied tool call was stamped with the same sentence — the one that says *the user* does not want to proceed — including the calls denied automatically on a lane that has no approval surface at all, where nobody was ever asked. The guard drives all three shapes (a person pressed No, a rule settled it, nobody said which) through both close-out arms and the durable park leg, and pins that the third shape is byte-identical to the previous release: an attribution nobody supplied is not evidence for either answer. The rule-settled shape carries the shell's own reason on a second line when there is one and stands alone when there is not, because a blank line where a reason should be reads worse than no line at all. The attribution is read from own data properties only, so neither a polluted prototype nor a getter can make an automatic denial claim a person made it — and the getter case is pinned to never run at all. The transcript classification word is minted only on the two paths where the upstream transcript format really carries one; the three classifier words and the two abort words are left absent, with the abort words pinned against the strings this package actually normalises interruptions to, which are different strings |
354
354
  | `scripts/run-park-hop-progress-test.mjs` | L-80: the park re-attach loop budgets **stalled** rounds, not parks. A turn where the model keeps hitting gates and every one of them is really decided (a card was answered, the engine really moved on) must never be cut off by the hop budget — the budget counts consecutive rounds that produced no progress, and "the engine revived and immediately parked again on the same coordinates" is not progress. The three non-progress arms (already-resolved, decide-transport-exhausted, and a re-scan that was adopted but led nowhere) share one same-cause limit instead of one arm having a limit and the others having none, and every non-progress re-attach is announced once through the host callback rather than only to the debug log. When the limit is spent the resolver reads the approval queue once more and puts whatever is decidable in front of the user before it gives up; only when there is genuinely nothing to show does it fail soft, and the terminal message then carries the real cause and a real way out instead of a sentence about a budget. On the self-heal side, a reopen verdict that reports `decidedWithoutCard` — the chain settled the gate by rule, so there was no card to present — is progress, not a reopen failure, and the user is not told their message was NOT sent. Negative control: a genuinely empty queue with a run that never moves still fails soft |
@@ -378,7 +378,7 @@ guard still cross-checks the table by name).
378
378
  | `scripts/run-wiring-manifest-projection-test.mjs` | The two end-user facts carried on the engine's `wiring_manifest` frame (`modelGate`: which tools this run's model gate removed and the verbatim restore hint; `autoMode`: whether auto mode is actually armed and the engine's own reason word). Projection: both sections ride as `_sema_`-prefixed superset keys, verbatim, and no SDK-named key is minted; a frame where neither section is well-formed projects to `none/not_in_slice` (no empty arm); `modelGate` needs all three keys and treats `removed: []` as a bad value rather than a reading; `autoMode` needs a boolean plus a non-empty reason that agrees with it, and the reason word is never mapped onto the capabilities vocabulary; the frame is flat (a nested `manifest:{}` wrapper is not a supply); `eventId` rides like every other arm. Adapter: exactly one chrome event on the main lane, a sub-flow frame (any `parentToolCallId`, `null` included) yields nothing, and an absent `eventId` leaves the key absent. Added at receiving time because the shell-side gate could not see this package's behaviour: two mutations (empty `removed` accepted, sub-flow gate removed) had passed the package suite untouched 0.71.0 adds sections F–I: the fourth/fifth/sixth manifest sections (`tools` via the roster reader, `hooks[]` rows dropped one by one when malformed, `lsp` absent unless `mounted` is a boolean), the `tool_roster_delta` arm (narrowed `delta`, `malformed` when `fromDigest`/`roster` cannot be read, host applies it against its own digest), the `context_usage` arm (finite-gated scalars plus `sections[]` rows dropped one by one), and the `WiringManifestMcpEntryView` rename with `MAX_AGENT_SKILLS` gone from the surface |
379
379
  | `scripts/run-submit-wiring-manifest-test.mjs` | The non-streaming submit receipt can carry the run's opening wiring manifest (`TaskResult.wiringManifest`, additive on newer servers). `readSubmitWiringManifest` answers one of three: the key is absent on the receipt itself (older server, or a deployment whose engine never produced that frame) — not the same as unreadable; the key is present but cannot be read (not an object, or none of the nine sections survive); or a manifest view. The view is the same shape the streaming lane's chrome event carries (minus its two envelope keys) and is assembled by the same code path, so both lanes agree byte for byte on the same object. Liveness fields ride through untouched — this reader never mints a liveness verdict — and an operator-shaped receipt with extra governance sections reads to the same view as a tenant-shaped one. A zero-tool roster is a real reading, not an absence. |
380
380
  | `scripts/run-rule-offers-reader-test.mjs` | The narrowing reader behind the "don't ask again" options, now a public entry point rather than a card-port-only one. Hosts that render the frame themselves (a browser has no three-way terminal card) previously had to rebuild this reader on their side, and what it carries is a **redemption-safety** judgement, not a convenience: the batch arm is redeemed by **index**, so a reader that compacts the array after dropping a malformed entry makes the k-th option a person clicked and the k-th rule the server writes two different rules. So: a bad entry is dropped **on its own** (one bad option must not make a real one disappear) while every surviving entry keeps its **original wire index** — pinned from both ends, with the bad entries leading and trailing. A batch's *members* are the opposite: any malformed member drops the whole batch, because a conjunctive batch is one "yes" to all of them and a batch missing a member is a different grant; its honest-remainder count is a reading, not decoration, so a non-integer or negative value drops the batch rather than rendering a fabricated zero. An empty array, a non-array, an over-cap array and an all-bad array all read as **absence** rather than an empty list, because an empty list renders as "there is an option lane with nothing in it". The two wire generations are ordered by a rule, not a preference: the newer key wins outright, a newer key that is **present but unreadable** does not fall back to the retired key (borrowing the older material would pass someone else's options off as this request's), and a `null` newer key reads as absence so a relaying layer that serialises "missing" as null cannot delete the whole lane on older engines. The public entry is finally reconciled against **both** card-port legs on the same material, byte for byte, so the exported reader and the one the card sees can never become two. Two upstream vocabularies used to be **hand-copied** here, and both had fallen behind: a match word outside the copied pair dropped an otherwise valid option outright, and a batch carrying a directory-read member — a member kind the copy did not know — dropped the whole batch. Both tables now come from one place upstream and are re-exported verbatim, pinned in both directions: every word in the table must be accepted (a narrower copy reds on the words it never learned) and a word constructed to be outside it must still be refused (a reader widened to "any string" reds too), with the retired-key normalising leg sharing the same narrowing so the fix cannot land on one leg only. A member whose kind is genuinely unknown still drops **the whole batch and only that batch** — never one member, because a conjunctive batch one member short renders "yes to N" as "yes to N−1", and never the card, because the honest single beside it is intact — while a member from before the discriminant existed normalises to the historical kind rather than being refused. The additive per-segment reasons ride through verbatim, drop only the row that is malformed, and stay **absent rather than empty** when nothing survives, since an empty list would read as "confirmed nothing uncovered" while the count remains the only source of truth |
381
- | `scripts/run-resume-refusal-copy-test.mjs` | The **words** a client says when a resume is refused, minted once here instead of three times. The facts behind them already lived in this package; the sentences did not, so each client wrote its own — and those sentences answer a safety question (was my decision consumed, can this token still be redeemed), which is exactly the kind of answer that must not vary by client. Two closed sets meet here and the guard pins their relationship in both directions, because it is a premise rather than a coincidence: one set answers *can waiting help* (the codes the server mints a wait on), the other answers *what should a person be told*, they **intersect in exactly one code**, and each keeps a member the other must not have — a placement mismatch is never waitable no matter what arrives on the response, since its remedy is a changed argument rather than elapsed time, and a full governance window needs no prose because "you can wait" is the whole message. The overlapping code delegates its wait and its disposition to the existing reading rather than judging again: nine shapes of input drive both entry points and the two readings must agree byte for byte, the absent case included, because two judges always diverge somewhere. The wait is narrowed to the domain the server mints it in, which is **stricter than the shell's own copy was** — a zero now reads as no window rather than as "retry now", and the wake-up it would retry is an at-most-once action with real side effects. The third sentence is chosen by the disposition, never by the engine's prose: rewriting the message to either upstream branch's exact wording, with the window untouched, must leave all three sentences unchanged, while adding a window must change the third one and only the third one |
381
+ | `scripts/run-resume-refusal-copy-test.mjs` | The **words** a client says when a resume is refused, minted once here instead of three times. The facts behind them already lived in this package; the sentences did not, so each client wrote its own — and those sentences answer a safety question (was my decision consumed, can this token still be redeemed), which is exactly the kind of answer that must not vary by client. Two closed sets meet here and the guard pins their relationship in both directions, because it is a premise rather than a coincidence: one set answers *can waiting help* (the codes the server mints a wait on), the other answers *what should a person be told*, they **intersect in exactly one code**, and each keeps a member the other must not have — a placement mismatch is never waitable no matter what arrives on the response, since its remedy is a changed argument rather than elapsed time, and a full governance window needs no prose because "you can wait" is the whole message. The overlapping code delegates its wait and its disposition to the existing reading rather than judging again: nine shapes of input drive both entry points and the two readings must agree byte for byte, the absent case included, because two judges always diverge somewhere. The wait is narrowed to the domain the server mints it in, which is **stricter than the shell's own copy was** — a zero now reads as no window rather than as "retry now", and the wake-up it would retry is an at-most-once action with real side effects. The third sentence is chosen by the disposition, never by the engine's prose: rewriting the message to either upstream branch's exact wording, with the window untouched, must leave all three sentences unchanged, while adding a window must change the third one and only the third one The engine's folded resume refusal (`resume_blocked_by_policy`) gets its own reading — the original code as named, absent or unreadable — and a wording that neither claims nothing was consumed nor predicts whether a retry would pass. |
382
382
  | `scripts/run-resume-retry-later-test.mjs` | The two resume refusals that carry a **wait quantity** — the only members of that refusal family that do, which is the whole reason they form a closed set. Carrying a wait is not the same as being the only ones worth waiting on: a sibling refusal in the same family clears on its own and the engine says so in words, it just cannot put a number on it, so *not recognised here* must never be read as *waiting will not help*. One of the two also has a *terminal* upstream branch that arrives under the same code with the distinguishing detail only in prose, so recognition alone is not permission to say "try again": the disposition is decided by **positive evidence** and pinned from both directions — the quota-window code is evidence in itself, the preflight code counts only when the server really supplied a wait (an upstream fact, not a convention: the terminal branch throws with no detail at all, so a wait value cannot reach the client on that path), and a preflight refusal with no wait reads as *undecidable* (say what is true of both branches — nothing was consumed — and leave redeemability to the engine's own line) rather than being rendered as either a retry or an ending. Every other member means waiting will not help (change a setting, relaunch, the retained session is gone), so the recognition is a **closed set of two codes**: widening it to a family prefix would tell half the users to wait and the other half to keep waiting for something that will never arrive, and the negative controls drive exactly those codes through it, plus a same-named code on a different door (the submission-side quota refusal), the two underscore-form siblings, and a code merely quoted inside a message body. The wait value is narrowed to the same domain the server mints it in (a whole number of seconds, at least one): zero, a negative, a fraction and a non-number all read as **no window given** rather than as zero, because a zero tells the caller to retry immediately and the wake-up it would retry is an at-most-once action with real side effects. Reading is structural rather than `instanceof`, since the client is host-injected and the same class name across two bundles is two classes, and a null-prototype plain object must still be recognised. The failure classifier gains this one disposition without any existing one moving, an unknown code still falls to the honest open-set arm and its wait value is **not** believed, and an end-to-end call proves the disposition and the window reach the host while the call itself is still attempted exactly once. The recognised code set is a **frozen array**, not a type-level readonly set: the latter is a plain mutable collection at runtime and the decision reads the same instance, so one `.add` from any consumer would turn a refusal that waiting cannot fix into one that claims it can — the guard proves it by really trying to mutate the exported value and then checking the verdict did not drift |
383
383
  | `scripts/run-model-capability-probe-test.mjs` | Whether a model on the OpenAI-completions lane **thinks**, and whether that thinking can be **turned off** — a question nobody can answer by looking at a model name, and one whose wrong answer costs every later call. The probe is judgement only: the network half arrives as an injected port, so the package mints no URL, reads no credential and never calls `fetch` — pinned by a source-level assertion, because a package that reaches the network once has changed what every host must trust it with. The seven dialect words are a **copy**, reconciled element-wise against the installed engine’s own bytes in both directions, since the words belong upstream and a private table drifts the day a dialect is added; the settings package deliberately declines to restate them, so the table cannot be imported from there and this guard is what stands in for the import. The **order** the dialects are tried in is a public promise rather than an implementation detail — each extra attempt is real money and real latency against someone’s gateway — so the guard pins the exact call sequence a stub records, and reversing it reds on the wasted round trip; the template-parameter spelling leads because an observed gateway keeps thinking, and answers with an empty body, when handed the top-level switch instead. That observation is also why an empty answer is **not** accepted as *thinking is off*: a knob that deletes the reply is not a knob that disabled reasoning, and accepting it would write a spelling into the catalogue that the gateway does not honour. Two dialect words whose request bytes are identical to another’s do not each burn an attempt. The two verdicts that look alike are held apart from both directions: *tried everything, still thinking* requires at least one attempt to have **cleanly answered**, and when every attempt was refused the verdict is *could not tell* instead — and on the unanswerable path the result carries **no** thinking flag at all rather than a fabricated `false`, while the pure write-back returns the very same entry object untouched. A verdict that reasoning cannot be disabled **removes** a previously declared spelling rather than leaving it, since a refuted spelling keeps the engine sending bytes the gateway ignores while the catalogue still renders it as already off. Evidence is lengths, finish positions and status codes only — a planted secret in both the answer and the reasoning channel must appear nowhere in the result, so the record can go into a log or a ticket whole One cross-package premise is checked by really running the other package’s parser rather than quoting its documentation: everything this probe writes eventually passes through the settings schema on its way into a catalogue, and that field is declared parse-transparent precisely so the vocabulary can live on the consuming side. If it ever narrows, the spelling is stripped **silently** — indistinguishable from the probe never having run — so the guard feeds the probe’s real output through the real parser, checks the compat object comes back key for key, and checks a dialect word this client has never heard of survives too. Two shapes that must be rejected really are rejected, since otherwise the survival checks would hold on a parser that accepts anything, and a bare entry is asserted valid first, because the first run of this section reddened on a space in a fixture’s name — a fixture that cannot pass would disguise the real alarm as already having fired |
384
384
  | `scripts/run-decide-receipt-test.mjs` | What a decision verb actually **answered** — and, more importantly, what it did not. A success response on the newest lane is only an acknowledgement that the decision was accepted for delivery: the approval is still pending, and a client that clears the card on it shows either a ghost card that was already approved or a card that vanished while the decision was lost. So the package deliberately has **no** "was it resolved" predicate — nothing in that body can answer it — only the opposite one, whose `false` is likewise not evidence of resolution; resolution is only ever the next running arm on the stream. The guard pins that inversion in the product source too: the success path must no longer clear the latched gate, while the stream-observing path that really clears it must still be there. The body has four shapes with **no** key common to all of them, so every position is read as honestly absent, and the handoff handle — which run to watch from here on — requires **two** facts together, since either one alone would either point the stream at the run it already had or mint an empty handle. The record of what finally happened to an already-decided action is read through the **same** reader as every other gate record rather than a second copy, and its absence means **unknown**, never *it was allowed* — the two can even contradict each other, so the card says nothing at all when it is missing. The three refusals on that lane each get one distinct sentence and a disposition taken from **why** each was refused rather than from severity: one cannot be helped by re-sending at all, one waits on the host, one just drops an option — and none of them carries a countdown, because the server never mints a wait for them. Recognition is a **closed set**: an unrecognised code on the same prefix returns nothing rather than a guess, since that prefix also houses a safety signal whose whole rule is never to retry automatically, and the recovery handle is read as absent when unreadable rather than substituted from a different identifier that no longer appears on that lane **0.68.3 (core 7.18.0):** `gate.disposition.classifier` is read key by key into a named view (requested model, model that answered, and the ladder fallback when one happened); a half-shaped record yields no classifier at all rather than a half view, and absence stays "unknown", never "the seat answered itself" |
@@ -391,7 +391,7 @@ guard still cross-checks the table by name).
391
391
  | `scripts/run-classifier-status-test.mjs` | What state the auto-mode classifier is in **on this session** — the question a doctor line, a model settings page and a permission card’s status row all ask, and a different question from the one the approval card asks (*why am I being asked right now*), so the sentences are pinned mutually distinct from that face’s as well as from each other. The session-level half of this reading — a breaker record the engine used to keep — was **retired upstream**, and the guard now holds that retirement from **both** sides: the engine's own declarations must really no longer carry it (a fact coming back would mean the removal here was the wrong disposition, and that deserves a conversation rather than silence), and this package must carry no alias, no state word and no leftover narrowing for it — a reading kept alive for something nobody emits any more is a promise the interface cannot keep, and it left the doctor line advertising a state it can never reach. What remains is ordered by the quantity that actually decides whether the classifier is running: the fact from **this round** first, then whether this leg is armed — a decider is minted per run, so a later leg can be armed again. Not armed, and a section that never arrived, both answer **undefined** rather than *available*; that arming question has its own field and answering it twice grows a second ledger. Arming and availability are also **two words, not one**: the engine says a decider was minted *for this leg*, which is an assembly-time fact, while whether that decider answers any given round is a **per-call** one — so an armed leg reads `armed` and only a positive per-call fact (an ask whose origin is the classifier's own denial-bound fallback, which by construction stands *after* the classifier ran) reads `available`. Every other ask origin is refused as evidence and for a stated reason rather than out of caution: several are ones the classifier is structurally forbidden to answer, and for the rest a surviving ask is precisely the case where it did **not** resolve one — so reading availability off them would be a guess. The projection is a **whitelist**, so an older engine still sending the retired member loses it at the boundary while the two live facts beside it ride through untouched. Rendering never throws and never impersonates: a state word this client does not know — including the retired one, which a restored view can still carry — reaches an honest fallback that names it verbatim, carries no invented explanation of a mechanism that no longer exists, and is proven distinct from all three real sentences; prototype keys reach that same fallback rather than a function body, checked against a real out-of-table word so the comparison cannot hold vacuously |
392
392
  | `scripts/run-compaction-boundary-projection-test.mjs` | The compaction divider and the one frame that makes its anchor resolvable. The trigger word is passed through as an **open set** instead of being folded to two: the engine deliberately stopped flattening its third value (a compaction that was not optional — a prompt-too-long recovery or trim pressure) and carries what the hook layer saw, so folding it again at the package boundary re-introduces exactly what upstream had just removed, while a consumer branching on *is it manual* keeps its behaviour byte for byte. Only an unreadable word (absent, empty, non-string) falls back — that is *could not read it*, not *read it and did not recognise it*. Two superset keys ride the metadata and neither fabricates: the preserved-segment anchor is minted only when its id really reads out, because half an anchor sends the host looking up an empty string in its map, and the clamp ratio is a **disclosure** whose real zero is a fact rather than an absence. The clamp ratio also carries a registered exit condition — the service really sends it while the SDK arm has no seat for it yet, so the read is defensive and this guard reds the day that seat appears, forcing a re-check instead of leaving a cast to rot. The committed-message frame moves out of *deliberately not projected*: that classification was true about transcript rows and false about **positioning**, since the engine states that consumers build their own id-to-message map from this frame to place the divider — projecting the anchor without it hands the host something it cannot resolve. It becomes a neutral internal arm and an optional chrome ledger event, never a transcript row (the frame carries no body, so minting one would put words in the engine's mouth), with both required ids narrowed and a malformed frame recorded rather than half-minted |
393
393
  | `scripts/run-cost-absence-projection-test.mjs` | Telling **declared free** apart from **never priced**, in both directions, because the package was getting each one wrong in the opposite way. The engine separates them on the wire — an absent cost means some spend had no price table, an explicit zero means the model declared itself free — and the result projector used to require a *positive* number, so a genuinely free run could not say so; while the per-model mirror folded absence to zero, so an unpriced run told a billing consumer it cost nothing. The total is now reported as the engine stated it, with absence and non-finite values alone reading as unknown, and a negative passed through rather than corrected, since a refund is a legal figure and the package is not a second accountant. The per-model figure keeps the CC shape intact — that field is a required number and *unknown* is simply not expressible in it — so the value stays zero and a **companion superset bit** carries the distinction, which means the two are read together and a reader that only ever looked at the number is unchanged; the bit is minted only in the absent case and never as `false`, since a key present with a false value reads as a third state. The same mint point serves both the wire's per-model split and the synthesised current-model row, so neither can drift. Alongside it the cache-write figure stops being a hardcoded zero and reads the field the wire has always carried, in both the flat usage and the synthesised row, and all four flat token slots move from a null-coalesce to a finite-number guard — the stats object has an open index signature and the wire is JSON, so a string or an infinity would otherwise land in a slot the types promise is a number, compiling green and surfacing only when something sums it |
394
- | `scripts/run-permission-denial-projection-test.mjs` | The terminal result's **permission-denial list** being the wire's real one rather than a hardcoded empty array. The session vocabulary carries a list of tool calls that were denied; the projector used to mint `[]` in both the success arm and the error envelope, which folded two different statements into one — *nothing was denied on this run* and *this frame carries no such ledger at all* (an older engine, a rejection envelope, a failure event that arrives without stats) looked identical. Each denied gate on the wire's human-review ledger now becomes one record, in wire order, carrying the keys the wire can actually honour: the tool name when it reported one, and a superset field with the engine's own short, redacted one-line summary of the call's input. **Two lists, deliberately.** The reference shape requires three fields on every element — tool name, call id, and the full input object — and the wire's ledger carries only the first. Filling the other two with an empty string and an empty object would be invention; putting a half-filled element into the reference array would break the element contract, and a strict consumer validating the stream drops the *whole* result message rather than one field. So the reference array admits only fully-formed records — empty today, and filling itself the day the wire grows the two missing fields, with no code change — while every record the wire really has rides a superset carrier beside it. A contract check pins today's absence, so that day turns this guard red on purpose. The companion bit means *this reference list cannot be claimed complete*: no ledger, an unreadable row, an unrecognised decision word (a rejected plan is not a denied tool call, and a row with no decision at all is not a judgement), or a record that could not be fully formed. Only its absence lets a reader say *zero denials*; it is never minted as `false`. Rows that cannot be read drop themselves rather than the whole ledger, and both arms go through one mint point so they cannot drift. Since 0.73.4 the third CC key is sourced from the same stream's `tool_start` frame, joined by call id: a row joins only when the frame was seen on this stream, its arguments are a plain object, and no string leaf carries a transport replacement token or a cycle / depth placeholder (scan budgeted); both halves have positive controls (a fully joined list drops the discriminator, a partially joined one keeps it), the ledger's own input wins when present, the snapshot is per-stream and capped with a one-way overflow latch, and an id seen with two different argument objects never joins. Later sections add the second stream-local join and the two discriminators the headless exit-code rule needs. "Which layer denied this" is not on the denial ledger at all — it is on the gate record of the same call's close-out frame, so it is joined by call id under the same law as the arguments: the closed word table is checked on the collecting side, the ledger's own value wins if it ever arrives, a word from outside the table is not stamped, and a row that cannot be joined keeps the key absent rather than claiming nobody denied it. The classification word is carried on both lists under the same name and the same value, so a consumer needs one reader, not two. The "this run produced no tool output and was denied" flag is present only when three independent things hold at once — the denial evidence is read from the full list rather than the strict one, which can be empty for reasons that have nothing to do with denials; this stream saw no successful tool close-out; and this stream can honestly claim to have watched the run from its first frame. A stream that reconnected mid-run cannot make the last claim, so it mints nothing rather than a false negative, and the flag is never minted as false From 0.80.0 that classification has a **second source**. It used to come only from this package's own decision path, so a refusal the engine settled on its own — a deployment policy answering the card on an unattended lane, with no client involved — left the field empty even though the same stream's gate record said exactly what had happened. The engine's own settlement word now fills it when, and only when, the local one is absent: the package's own attribution always wins, because letting a replayed frame overwrite it would let the wire change what the host itself said. The word is read literally in both directions and never reverse-engineered, and the separate field naming *which layer* refused is left exactly as the wire wrote it — the two answer different questions, and rewriting one to match the other would make them say the same thing twice. |
394
+ | `scripts/run-permission-denial-projection-test.mjs` | The terminal result's **permission-denial list** being the wire's real one rather than a hardcoded empty array. The session vocabulary carries a list of tool calls that were denied; the projector used to mint `[]` in both the success arm and the error envelope, which folded two different statements into one — *nothing was denied on this run* and *this frame carries no such ledger at all* (an older engine, a rejection envelope, a failure event that arrives without stats) looked identical. Each denied gate on the wire's human-review ledger now becomes one record, in wire order, carrying the keys the wire can actually honour: the tool name when it reported one, and a superset field with the engine's own short, redacted one-line summary of the call's input. **Two lists, deliberately.** The reference shape requires three fields on every element — tool name, call id, and the full input object — and the wire's ledger carries only the first. Filling the other two with an empty string and an empty object would be invention; putting a half-filled element into the reference array would break the element contract, and a strict consumer validating the stream drops the *whole* result message rather than one field. So the reference array admits only fully-formed records — empty today, and filling itself the day the wire grows the two missing fields, with no code change — while every record the wire really has rides a superset carrier beside it. A contract check pins today's absence, so that day turns this guard red on purpose. The companion bit means *this reference list cannot be claimed complete*: no ledger, an unreadable row, an unrecognised decision word (a rejected plan is not a denied tool call, and a row with no decision at all is not a judgement), or a record that could not be fully formed. Only its absence lets a reader say *zero denials*; it is never minted as `false`. Rows that cannot be read drop themselves rather than the whole ledger, and both arms go through one mint point so they cannot drift. Since 0.73.4 the third CC key is sourced from the same stream's `tool_start` frame, joined by call id: a row joins only when the frame was seen on this stream, its arguments are a plain object, and no string leaf carries a transport replacement token or a cycle / depth placeholder (scan budgeted); both halves have positive controls (a fully joined list drops the discriminator, a partially joined one keeps it), the ledger's own input wins when present, the snapshot is per-stream and capped with a one-way overflow latch, and an id seen with two different argument objects never joins. Later sections add the second stream-local join and the two discriminators the headless exit-code rule needs. "Which layer denied this" is not on the denial ledger at all — it is on the gate record of the same call's close-out frame, so it is joined by call id under the same law as the arguments: the closed word table is checked on the collecting side, the ledger's own value wins if it ever arrives, a word from outside the table is not stamped, and a row that cannot be joined keeps the key absent rather than claiming nobody denied it. The classification word is carried on both lists under the same name and the same value, so a consumer needs one reader, not two. The "this run produced no tool output and was denied" flag is present only when three independent things hold at once — the denial evidence is read from the full list rather than the strict one, which can be empty for reasons that have nothing to do with denials; this stream saw no successful tool close-out; and this stream can honestly claim to have watched the run from its first frame. A stream that reconnected mid-run cannot make the last claim, so it mints nothing rather than a false negative, and the flag is never minted as false From 0.80.0 that classification has a **second source**. It used to come only from this package's own decision path, so a refusal the engine settled on its own — a deployment policy answering the card on an unattended lane, with no client involved — left the field empty even though the same stream's gate record said exactly what had happened. The engine's own settlement word now fills it when, and only when, the local one is absent: the package's own attribution always wins, because letting a replayed frame overwrite it would let the wire change what the host itself said. The word is read literally in both directions and never reverse-engineered, and the separate field naming *which layer* refused is left exactly as the wire wrote it — the two answer different questions, and rewriting one to match the other would make them say the same thing twice. A later section reconciles the terminal list against the denied calls seen in the same stream, so denials that never reached a human (rules, hooks, classifiers, write protection) are listed too: complete rows join the CC list, rows missing a field stay on the extended list and mark it incomplete. A further section feeds the same raw tool-result frame through the projector into both lanes and requires the denial category on the interactive transcript record, on the non-interactive frame and from the reader on the raw frame to agree, including frames whose settlement, classifier cause or classifier attribution is present but unreadable — a shape the narrowed gate view on the internal frame cannot show — and pins that the reader gives the same answer on the internal frame as on the raw one. |
395
395
  | `scripts/run-cost-reconcile-projection-test.mjs` | The **end-of-run cost reconciliation** reaching consumers at all. The engine splits a run's spend on the wire — the task's own cost, which deliberately excludes delegated sub-agents, the delegated total itself, and the within-task compaction subtotal that sits inside the own figure — and states two reconciliation identities for them. The package used to project none of it, so a cost view could only ever see one number and under-reported both delegated and compaction spend. Both structures are now projected onto the result as superset fields in the wire's integer micro-currency unit, read key by key, with unreadable keys dropped individually, an entirely unreadable structure omitted rather than emitted empty, and unknown categories passed through since the vocabulary belongs upstream. The delegated cost stays **absent when it was never priced**, never a fabricated zero. The same reader also feeds a terminal chrome arm carrying the three parts plus the reconciled total, so the two faces can never compute different answers; the reconciled total is minted only when both sides are known, and otherwise a discriminator bit says which side is unknown. **The reference field for total cost keeps its meaning** — it remains the task's own spend and the delegated total is not folded into it — because that is a shape the wider ecosystem reads; the reconciled figure is offered beside it, not in place of it. A frame that carries no stats emits no arm at all, and the existing rule that in-stream per-turn usage is not published for sub-flows is pinned unchanged, since delegated spend arrives once, at the end. The bit that says those figures are a lower bound is **per stream**, not per context: the emit context belongs to the caller and may be reused across streams, so a gap observed on one run is no evidence at all about the next one — the observation is held for the duration of one stream and handed to both projection faces by value, and the guard drives a reused context both sequentially and concurrently to prove neither direction leaks |
396
396
  | `scripts/run-task-progress-terminal-projection-test.mjs` | The one tick that says a delegated child **finished**. The engine fires exactly one final beat carrying a terminal face, and says in the same breath why it exists — so a consumer sees the row finish instead of watching it vanish after the last running beat — but the package's projection whitelist had no seat for that field and its adapter still carried the older premise in a comment, so the terminal beat arrived byte-identical to another running one: the panel row stayed up waiting for a defensive sweep (which only ever settles rows bound to a card still open this turn) or for a separate notification frame. The status now rides through as an **open set** with the vocabulary left upstream, while the question *which words are terminal* is answered by a closed pair on the adapter side — an unrecognised new word takes the running path, because guessing it terminal ends a row that is still working whereas one extra running beat merely renders late. A terminal beat settles the row directly under the lane proof its binding gives it (not the main lane a notification would use, and not by card id, since the engine is naming a child rather than closing a card), freezes the inline group-row twin in the same beat so a later sweep cannot reset the real tool count, clears the session-resident ledger, and fires the stop hook only for a child whose start really fired. It does not mark the row live or emit a second progress beat, and it shares the settled-row ledger with the other two settle legs so a replay or a double-delivery cannot produce a second end. Three things are pinned **unchanged**: a running beat, an absent status (older engines never send the field, and reading absence as terminal would make every child row disappear on its first beat), and the workflow lane gate, which still runs before any of this |
397
397
  | `scripts/run-assistant-arm-identity-test.mjs` | The identity keys on an assistant row, and an explicit account of the two that are **deliberately not** there. What the renderer received was a bare role-and-content object, so a dozen consumer sites downstream were each estimating what the message envelope should have told them. The id is taken from the engine's own event id rather than minted locally, because it has to be **the same value** on the live leg and on a durable replay — a freshly minted one would make a replayed message look new to a host's dedup and to rewind — and when the wire carries none the key is simply absent rather than filled with a random stand-in wearing an identity it does not have; it is also kept distinct from the envelope's own local render key, which is a different identity. The model name comes from what the host pinned when it opened the stream (the request was the host's to build) and is never guessed, since a wrong model name is worse than none once a billing or capability face looks it up. Usage and stop reason are **not** minted on this arm, and the reason is frame order rather than effort: content arms arrive before the turn's closing frame, so at the moment the arm is emitted the engine has not yet said what the round cost — anything put there would be an estimate, which is the very thing this work exists to remove — and synthesising a follow-up assistant update when the real figure lands is also refused, because that shape does not exist upstream and would place a message in the transcript the engine never sent. Their real values leave through the turn's own neutral arm as two superset keys, the usage one reusing the **same single mint point** the footer rollup already folds so the two faces cannot diverge, and the stop reason passed through verbatim as an open set — the machine signal for *was this turn cut short*, previously blind on both the stream and the trace. The existing behaviours beside them are pinned too: no arm at all when usage is wholly absent, and the sub-flow cut-out that keeps a child's turn from driving the leader's face |
@@ -414,7 +414,11 @@ guard still cross-checks the table by name).
414
414
  | `scripts/run-sdk-registry-transit-test.mjs` | The package's cloud control-plane surface (`sdkRegistryTransit`), which lives behind its own `./registry` subpath entry point rather than on the root barrel, and this guard holds both halves of that decision. Upstream publishes the same surface behind a subpath of its own, because the subject differs: the engine-wire surface speaks for one engine's service credential, this one for a person's rotating token, and their refresh and error semantics were deliberately never merged. Keeping it behind a second entry point means a client that never touches the control plane neither resolves nor type-checks it. Every value re-export is therefore read from the file the subpath entry actually resolves to, and must be the **same reference** as upstream's own (the control-plane client class, the three config reads, the health probe, the feedback call, the auth-path constant, the content-address helper, and the two typed error classes a host recognises with `instanceof`); every type re-export is compared with the emitted declaration file in both directions; the gate's list and the source file's export lists are likewise compared both ways; and the root entry is checked to carry none of these names, with the root barrel's own source checked to reference the entry file nowhere — a name leaking onto the root would put the cost of this surface back on clients that never asked for it. The subpath is then verified end to end: the installed SDK must really publish its own `./registry` entry and declare every transited name inside it, and this package's own `exports` must point that subpath at exactly the files the gate just judged. Portability is two checks rather than one, done with a parser rather than a text search: the upstream subpath's emitted JavaScript, walked recursively, must contain neither a `node:` specifier (static, side-effect, dynamic, `require` and re-export forms all exercised) nor a Node **global** — because the same package's third entry point is a Node-only surface that imports nothing at all and reaches for the `Buffer` global, so a specifier check alone would pass it as isomorphic, while a byte-level search of it reports two `node:` hits that live entirely inside a documentation example. A text scanner that merely strips comments first gets both directions wrong on ordinary JavaScript — a regular-expression literal containing a slash pair swallows the rest of its line, and the word in a string reads as a reference — so both scanners run off the syntax tree and are checked against fixtures for each failure direction as well as against that real material — including a dynamic import written with a template literal, which a check that accepts only quoted strings misses entirely, and a dynamic import whose target cannot be determined statically, which is refused rather than read as no edge at all. Zero-processing is likewise enforced with a syntax-tree allowlist rather than a keyword search: every top-level statement must be a named re-export carrying that one specifier, so an import followed by an in-place edit of the upstream prototype is refused with a file and line — that shape leaves the name lists untouched and even keeps the same-reference check green, since both sides are then the one object that was damaged. The last section measures, rather than merely notes, one declaration-level gap: two of the upstream subpath's declaration files reference a package the SDK lists only among its own dev dependencies. Moving such a check to a scratch directory is not isolation, because package resolution walks the ancestor directories, so the gate builds a sandbox served by a restricted compiler host and proves the isolation both ways — a decoy copy of the missing package placed one level above the sandbox must silence the errors for an unrestricted host and must not silence them for the restricted one. Then it installs this package into that same sandbox as a real consumer would, and pins the two readings that justify the entry-point split: a consumer that imports only from the root sees no unresolved-module errors at all, while a consumer that imports the subpath sees exactly the two, reported honestly rather than swallowed by this layer. The day upstream ships those declarations, that section turns red and the note comes out with it. The separation itself rests on the root closure being computed correctly, so the portability guard that computes it was extended in the same change: a template-literal dynamic import is followed like any other edge, and an edge whose target cannot be resolved statically is refused on every one of the four graphs — without that, a single line in a third file already reachable from the root would put this surface back into the root runtime while every guard stayed green. |
415
415
  | `scripts/run-rule-removal-consequence-test.mjs` | The one sentence a rule-removal confirmation surface shows for what removing the rule will actually do: one line per behaviour the rule could have been enforcing, plus a neutral line for when that behaviour cannot be read back, so a caller that hits an out-of-set or missing value never falls back to a specific claim it cannot support. The guard compares the actual output against frozen text rather than merely checking that some string came back, so a dropped word or a swapped clause is caught the day it lands, and the four sentences are pinned pairwise distinct. The line for a rule that was denying something is pinned to say the removal widens what can run rather than echoing the wording used for a rule that asks again — the two are opposite directions, and sharing a sentence between them would tell the person confirming the removal the opposite of what is about to happen. The neutral line is checked from the other side for the same reason: it must not contain a word that belongs to only one of the three behaviours, because that would answer on behalf of a state the caller was unable to determine. The lookup that turns a raw stored or transmitted value into one of the three behaviours reads by strict equality only, proven with a fully trapped proxy and a counting getter to show it never touches a property on whatever it is handed, so a value with a legitimate-looking word sitting on its prototype chain is rejected exactly like any other out-of-set value rather than being unwrapped. A closing self-check mutates one character out of each frozen sentence and asserts the exact-match comparison actually fails on it, so the guard cannot pass by checking only that a string of some kind came back |
416
416
  | `scripts/run-mcp-engine-leg-test.mjs` | The two legs behind one row on the MCP detail card. In a two-process setup the servers are hosted by the **engine**, while the Status cell on the card reports **this client's own** connection to them, and the two are independent truths: this client failing to connect does not mean the server's tools are unavailable, and the engine leg on the same screen may be saying it completed an exchange moments ago. Two mints share the work. The first reads one server's liveness off the engine leg as six readings, and its load-bearing distinctions are three. **No roster that could be read end to end** (`null`, `undefined`, anything that is not an array, a length that is not a non-negative integer, a length beyond the scan bound, or rows that could not be read at all with nothing matched) is **not** the same as an **empty** roster, which is the engine leg's positive statement that it declared no servers; the same narrower that feeds this reader answers `undefined` when a non-empty section yields no readable row, so *I could not read it* never turns into *I know it is zero*, and a roster that was not read to the end never produces *this server is not on it*. **Could not be read is not the same as absent**: absence means only that the row carries no liveness key of its own, while a record that is present but unintelligible — a cell that is not an object, a word that is not a non-empty string, or a read that fails outright — is reported as unreadable and is decided before the observation beside it, since a record that may well have said the opposite is no evidence of reachability. **An ambiguous name is answered as ambiguous**: when a roster that was read end to end matches a name more than once, the reader reports the match count instead of picking one, because the upstream contract for the sibling MCP face states that names are not guaranteed unique, and picking optimistically would contradict the worst-fact-first verdict shown on the same screen. When a row's identity could not be read at all the reader makes **no claim about that name whatsoever**, not even a count, since the row it could not read may well be a second one carrying the same name — a row that is plainly not a row, such as a hole in a sparse array, is a different matter and leaves the roster complete. The observed word is passed through **verbatim as an open set**, so a fourth word one day arrives at consumers untouched. The second mint is the one sentence that goes under Status, and it appears **only** when this client's own connection really did fail — the other Status values already tell the truth, and a sentence on top of them would only muddy the verbatim health vocabulary. **The liveness observation outranks the tool roster, and having heard from the engine includes hearing that it could not tell, and hearing something unreadable**: a word meaning *looked and could not tell*, a word this version does not recognise, and a record that arrived but could not be read each get their own sentence rather than falling back to the roster, because falling back would say this client holds nothing at all while the diagnostics page shows that very record. A value that is not a reading at all is a different case and does fall back, since nothing then establishes that the engine sent anything. Only when liveness genuinely cannot speak does the roster get a turn, and the roster itself has four answers — tools were listed, nothing was said, the engine reported zero, and the count could not be read — because **unreadable, absent and zero are three different facts**. No sentence claims that nothing at all has been seen about the server, because on two of the readings that reach the roster the engine has plainly listed it; the sentence for a silent roster states only what this client holds. The nine sentences are pairwise distinct, each one names the engine leg, and **none of them renders the liveness word itself**. Membership of the word list is decided by the one shared predicate rather than a second copy, and the branch over the known words is pinned so that a new word upstream fails the build instead of silently taking the *not recognised* sentence. Neither mint ever throws, whatever a host hands it: every value is taken through one reader that accepts **own properties only** — an inherited key is not a wire fact, and one on a shared prototype could otherwise manufacture an engine observation or suppress a real one — reads each key exactly once, and keeps a read that fails apart from a value that is absent. The roster is walked by index rather than through the array's own `find`, the array test is guarded because it can throw on its own, and a length that overstates itself would otherwise spin forever |
417
- | `scripts/run-cc-message-key-census-test.mjs` | The one rule behind CC-shaped messages this package emits: every top-level key on a `user` / `assistant` / `result` / `system` message must be a member the CC SDK mirror (`@sema-agent/agent-types`) declares for that same arm, or one of this package's `_sema_`-prefixed additive keys, or a row in a dated transition table that goes red the day its retire version arrives. Mint sites are found syntactically (identifier, quoted and computed `type` names alike) and their key sets are resolved syntactically too — inline literals, both branches of a conditional, the nullish-coalescing and logical or/and operators, `const` initializers and every `return` of a helper — so a type assertion, a `Partial<Pick<…>>` narrowing or a computed name cannot launder a key past the check, while a spread the tool cannot follow (a parameter, a `let`, a member access) fails the tool, never the product. Arms that carry a `subtype` discriminator are checked against that subtype's own member set, so a key declared only for another subtype does not pass on the strength of the arm-wide union. The runtime section drives the real adapter pipeline and pins the tool-result record's SDK-spelled `tool_use_result` (the transitional camelCase twin rides along by reference until 0.82.0). |
417
+ | `scripts/run-cc-message-key-census-test.mjs` | The one rule behind CC-shaped messages this package emits: every top-level key on a `user` / `assistant` / `result` / `system` message must be a member the CC SDK mirror (`@sema-agent/agent-types`) declares for that same arm, or one of this package's `_sema_`-prefixed additive keys, or a row in a dated transition table that goes red the day its retire version arrives. Mint sites are found syntactically (identifier, quoted and computed `type` names alike) and their key sets are resolved syntactically too — inline literals, both branches of a conditional, the nullish-coalescing and logical or/and operators, `const` initializers and every `return` of a helper — so a type assertion, a `Partial<Pick<…>>` narrowing or a computed name cannot launder a key past the check, while a spread the tool cannot follow (a parameter, a `let`, a member access) fails the tool, never the product. Arms that carry a `subtype` discriminator are checked against that subtype's own member set, so a key declared only for another subtype does not pass on the strength of the arm-wide union. The runtime section drives the real adapter pipeline and pins the tool-result record's SDK-spelled `tool_use_result` (the transitional camelCase twin rides along by reference until 0.83.0). A second runtime section feeds the same tool-result frame to both the interactive adapter and the non-interactive frame builder and requires the structured-result key and the no-output marker to be present together, absent together and equal on the two outputs, while the transitional camelCase name stays on the transcript record only; a static section checks a per-key routing table for the internal tool-result frame against the key sets resolved at the three mint sites, in both directions, and requires the non-interactive frame's structured-result value to trace back to the very same syntax node the transcript record uses — so recomputing or copying that logic on the non-interactive path fails even when the values happen to agree. |
418
+ | `scripts/run-rewind-archive-capability-test.mjs` | The read face for whether a rewind can restore the code archive, and the honest three-state answer the ends render from it. The shells used to decide this from a local backup table that only their own in-process tools ever fill, so on any session where the engine runs the tools it stayed empty and the two code-restoring rewind modes simply never appeared, while the engine had been keeping a file history the whole time. The judgement now comes from what the engine itself advertises, and each of the four bits it advertises answers a different question: whether the conversation can be forked at a message at all, whether a fork can restore the tracked set, whether a code-only restore is possible, and whether this deployment understands the current spelling of the request key. The mode that rewinds the conversation and restores the code together needs both of the first two, and the engine offers no single bit for that combination, so the combination is made here: one bit stated off is enough to rule the mode out, both stated on make it available, anything else stays unknown — reading the file-history bit alone would offer that mode on a deployment that keeps file history but has no conversation anchors, where the request can only fail. A bit that is absent means an older engine that never spoke about it, which is not the same as an engine that said no, and a bit reported in a shape this reader cannot read is a third thing again — it is recorded as unreadable rather than quietly filed under "not reported", because those two send an operator to different places. One unreadable bit does not discard its siblings; only a response that is not a capability object at all clears the cell. What the ends get is available, unavailable or unknown, and unknown stays unknown: folding it into unavailable would hide the mode again, which is the mirror image of the bug this replaces. The sentences the doctor row can print are checked to be pairwise distinct and to avoid implying a refusal the engine never made, and the spelling of the outgoing request is deliberately not made to follow the epoch bit, since the older spelling is rejected outright by current engines |
419
+ | `scripts/run-registry-quota-usage-test.mjs` | The projection of the cloud control plane's quota reading, and the three different things a missing number can mean there. This response says `null` in two places and means something different each time: no token quota is configured for this principal on this instance, and this window has no cap at all. Both are **facts the server is asserting**, not gaps in the reading — while a key that is absent or carries the wrong type is a genuine gap. All three have to survive to the screen separately, because folding them is how a user ends up staring at a confident `0`: an uncapped window rendered as if nothing were left, or a deployment that simply never configured quotas rendered as if the quota were exhausted. The complaint that started this was the opposite direction — a centrally configured quota that the command line could not see at all — so the reading also refuses to let an unreadable response masquerade as "no quota configured". Two fields deliberately do not share one signal: whether a window is exhausted and whether there is a recovery time, since the recovery time is only ever populated in the exhausted case and reading its absence as "not exhausted" would answer a question the response never answered. Counts that the server always provides are narrowed no further than the mint: a used counter has no uncapped state, so a null there is unreadable rather than zero. The wording helper carries the only human-facing phrasing, and the sentences for "no cap" and "unknown" are checked to contain no digits at all |
420
+ | `scripts/run-file-history-capture-capability-test.mjs` | The engine's file-history-capture self-description (`capabilities.fileHistoryCapture`), read the same four-state way as its sibling capability readers: an absent key is reported as not reported (never folded into `off`), words are taken as an open set so a newer mode is not mistaken for a malformed answer, `fileHistoryCaptureMode` recognises only `off` and `on-always`, and the wording for `off` speaks about capture only — whether code can be rewound is left to the rewind readings. |
421
+ | `scripts/run-model-identity-resolvability-test.mjs` | The model-identity judgement a client makes before letting anyone in: can the engine it is about to use start with a model name? Each end reports what it read from each place that can feed a model name to a local engine (complete, partial — a gateway address or a credential but no model name —, absent, or unreadable), or, for an engine that runs elsewhere, whether that engine has been seen answering; `modelIdentityResolvability` answers resolvable, unresolvable or unknown. Having part of an upstream configuration is not having enough of one, so partial lanes never add up to resolvable; a lane that was not reported or could not be read makes the answer unknown rather than unresolvable; an engine that runs elsewhere is never judged unresolvable and local lanes are never consulted for it (it does not start without a model name, so seeing it answer is enough to call it resolvable). `modelSetupDecision` combines that answer with whether this end can configure a model at all: setup is offered only for unresolvable on an end that can configure one, an end that cannot says so and points at whoever runs the engine, and unknown never opens setup. The detail and notice sentences are checked to be pairwise distinct, unknown sentences neither claim a model is configured nor that it is not, and the module is checked to import no platform I/O |
418
422
 
419
423
  Each suite carries a floor that only moves up — a refactor that stops executing a group of
420
424
  assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
@@ -1,5 +1,19 @@
1
1
  import type { AdapterContext, AdapterOutput } from '../seam.js';
2
+ import { structuredToToolUseResult, type ToolResultBody } from '../toolResult.js';
2
3
  import { type Frame, type IdOf } from './ids.js';
4
+ type RichToolUseResult = NonNullable<ReturnType<typeof structuredToToolUseResult>>;
5
+ export interface ToolResultOutcome {
6
+ readonly body: ToolResultBody;
7
+ readonly rich: RichToolUseResult | null;
8
+ }
9
+ export type ToolResultOwner = 'leader' | 'subagent';
10
+ export declare function toolResultOutcomeOf(toolName: string, end: ToolEndPayload | undefined, rawInput: PendingToolUse['rawInput'], owner: ToolResultOwner, ledgerTier?: (isError: boolean) => RichToolUseResult | null): ToolResultOutcome;
11
+ export interface ToolResultEnvelopeKeys {
12
+ readonly tool_use_result?: ToolResultBody['toolUseResult'];
13
+ readonly toolUseResult?: ToolResultBody['toolUseResult'];
14
+ readonly _sema_degraded?: ToolResultBody['degraded'];
15
+ }
16
+ export declare function toolResultEnvelopeParts(o: ToolResultOutcome): ToolResultEnvelopeKeys;
3
17
  export interface PendingToolUse {
4
18
  id: string;
5
19
  name: string;
@@ -29,3 +43,4 @@ export interface ToolCardLedger extends OpenCardView {
29
43
  resetResponseId(): void;
30
44
  }
31
45
  export declare function createToolCardLedger(ctx: AdapterContext, idOf: IdOf): ToolCardLedger;
46
+ export {};
@@ -6,36 +6,52 @@ import { isCcToolDenialKind } from '../gateVocabulary.js';
6
6
  function toolUseResultParts(v) {
7
7
  return v !== undefined ? { tool_use_result: v, toolUseResult: v } : {};
8
8
  }
9
+ export function toolResultOutcomeOf(toolName, end, rawInput, owner, ledgerTier) {
10
+ const isError = end?.isError ?? false;
11
+ const leaderLedger = owner === 'leader';
12
+ const body = end && end.output !== undefined
13
+ ? wireOutputToBody(toolName, end.output, isError, end.truncated === true, flattenWireOutput, end.structured, leaderLedger)
14
+ : degradedToolResultBody(isError);
15
+ let rich = end?.structured !== undefined
16
+ ? structuredToToolUseResult(end.structured, end.output !== undefined ? flattenWireOutput(end.output) : undefined, leaderLedger)
17
+ : null;
18
+ if (rich === null && ledgerTier !== undefined)
19
+ rich = ledgerTier(isError);
20
+ if (rich === null && !isError && toolName === 'ReportFindings') {
21
+ rich = reportFindingsToolUseResult(end?.structured, rawInput);
22
+ }
23
+ return { body, rich };
24
+ }
25
+ export function toolResultEnvelopeParts(o) {
26
+ return {
27
+ ...toolUseResultParts(o.rich !== null ? o.rich.toolUseResult : (o.body.toolUseResult ?? o.body.content)),
28
+ ...(o.body.degraded !== undefined ? { _sema_degraded: o.body.degraded } : {}),
29
+ };
30
+ }
9
31
  export function createToolCardLedger(ctx, idOf) {
10
32
  const pending = new Map();
11
33
  let assistantResponseId = null;
12
34
  let lastLiveTodos = [];
13
35
  function* buildToolResult(p, end) {
14
36
  const isError = end?.isError ?? false;
15
- const body = end && end.output !== undefined
16
- ? wireOutputToBody(p.name, end.output, isError, end.truncated === true, flattenWireOutput, end.structured)
17
- : degradedToolResultBody(isError);
18
37
  if ((p.name === 'Bash' || p.name === 'Monitor') && !isError && end?.output !== undefined) {
19
38
  const reg = detectEngineBgShellReceipt(p.name, p.rawInput, flattenWireOutput(end.output));
20
39
  if (reg) {
21
40
  yield chrome({ kind: 'bgshell_register', laneProof: mainLane(), taskId: reg.taskId, registration: reg });
22
41
  }
23
42
  }
24
- let rich = end?.structured !== undefined
25
- ? structuredToToolUseResult(end.structured, end.output !== undefined ? flattenWireOutput(end.output) : undefined)
26
- : null;
27
- if (rich === null && p.name === 'TodoWrite') {
28
- const todo = todoWriteToolUseResult(p.rawInput, lastLiveTodos, isError);
29
- if (todo !== null) {
30
- rich = { toolUseResult: todo.toolUseResult, isError: todo.isError };
31
- lastLiveTodos = todo.newTodos;
32
- }
33
- }
34
- if (rich === null && !isError && p.name === 'ReportFindings') {
35
- rich = reportFindingsToolUseResult(end?.structured, p.rawInput);
36
- }
37
- const eventId = end?.eventId ?? p.eventId;
38
43
  const parentToolCallId = end?.parentToolCallId ?? p.parentToolCallId;
44
+ const outcome = toolResultOutcomeOf(p.name, end, p.rawInput, parentToolCallId === undefined ? 'leader' : 'subagent', (err) => {
45
+ if (p.name !== 'TodoWrite')
46
+ return null;
47
+ const todo = todoWriteToolUseResult(p.rawInput, lastLiveTodos, err);
48
+ if (todo === null)
49
+ return null;
50
+ lastLiveTodos = todo.newTodos;
51
+ return { toolUseResult: todo.toolUseResult, isError: todo.isError };
52
+ });
53
+ const { body, rich } = outcome;
54
+ const eventId = end?.eventId ?? p.eventId;
39
55
  yield transcript({
40
56
  type: 'user',
41
57
  message: {
@@ -52,8 +68,7 @@ export function createToolCardLedger(ctx, idOf) {
52
68
  uuid: eventId ?? deriveTranscriptId({ toolCallId: p.id }, ctx),
53
69
  session_id: ctx.sessionId,
54
70
  ...(parentToolCallId !== undefined ? { parent_tool_use_id: parentToolCallId } : {}),
55
- ...toolUseResultParts(rich !== null ? rich.toolUseResult : (body.toolUseResult ?? body.content)),
56
- ...(body.degraded !== undefined ? { _sema_degraded: body.degraded } : {}),
71
+ ...toolResultEnvelopeParts(outcome),
57
72
  ...(isCcToolDenialKind(end?.denialKind) ? { _sema_denial_kind: end.denialKind } : {}),
58
73
  }, ctx.now());
59
74
  }
@@ -32,6 +32,7 @@ export type ReopenCardVerdict = {
32
32
  reopened: false;
33
33
  decidedWithoutCard?: true;
34
34
  pendingRowGone?: true;
35
+ _sema_decisionInFlight?: true;
35
36
  } | {
36
37
  reopened: true;
37
38
  firstSight: boolean;
@@ -78,6 +79,7 @@ export type SelfHealOutcome = {
78
79
  kind: 'plan-review-reopen-failed';
79
80
  taskId: string;
80
81
  decidePath: string | null;
82
+ decisionInFlight?: true;
81
83
  } | {
82
84
  kind: 'not-parked';
83
85
  taskId: string;
@@ -98,6 +100,7 @@ export type SelfHealOutcome = {
98
100
  kind: 'ask-reopen-failed';
99
101
  taskId: string;
100
102
  decidePath: string | null;
103
+ decisionInFlight?: true;
101
104
  } | {
102
105
  kind: 'ask-decided-without-card';
103
106
  taskId: string;