@sema-agent/client-core 0.83.3 → 0.83.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/README.md +5 -4
- package/dist/adapter/activeRunSelfHeal.d.ts +6 -1
- package/dist/adapter/activeRunSelfHeal.js +36 -4
- package/dist/hitl/armedGateRegistry.js +3 -5
- package/dist/hitl/gateIdentity.d.ts +1 -0
- package/dist/hitl/gateIdentity.js +8 -0
- package/dist/hitl/planReviewWire.d.ts +7 -0
- package/dist/hitl/planReviewWire.js +345 -76
- package/dist/hitl/toolApprovalWire.d.ts +2 -0
- package/dist/hitl/toolApprovalWire.js +23 -4
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/seatContract.d.ts +18 -1
- package/dist/seatContract.js +20 -0
- package/dist/systemReminderTag.js +25 -4
- package/docs/INTEGRATION-CLIENTS.md +139 -8
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -49,6 +49,50 @@
|
|
|
49
49
|
> 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
|
|
50
50
|
> commit 漏转在发布前就红,不再靠人记。
|
|
51
51
|
|
|
52
|
+
## 0.83.4(2026-09-26)
|
|
53
|
+
|
|
54
|
+
> 主题:patch —— 三件同发 + 两条接入指引订正。① plan 审批卡「用户亲手关掉的卡,自动触发不在同一道门上放回屏上」的判定归包:`reopenPlanReviewCard` 多一个触发者位 `trigger`,自动触发撞上「用户关卡」记号时拒开并如实回 `dismissedByUser`;用户自己的下一个动作照开;409 自愈腿按提交出身替宿主传触发者,结局带同一位、整行由本包出。② 座位审批请求补上卡请求早就有的四位 —— 强制位 `mandated`、强制理由词 `mandate`、报价缺席因由 `ruleOffersAbsence`(寄存行那条腿的强制表达)、出身词 `origin`,经一只与卡请求同形窄读的过境口 `toolPermissionRequestAskBits` 上座位。③ `<system-reminder>` 信封壳的剥离 / 解包两只纯函数上公面(`stripSystemReminderBlocks` / `unwrapSystemReminder`),并改成线性时间实现。另订正两条展示层接入指引(§104a′)。根公面运行期导出 1273 → 1277(+4:`stripSystemReminderBlocks` / `unwrapSystemReminder` / `toolPermissionRequestAskBits` / `clearPlanReviewUserDismissals`);公面类型 +2(`PlanReviewReopenTrigger` / `PlanReviewGateInstanceReader`);`ToolPermissionRequest` +4 可选位;`ReopenPlanReviewOpts` +3 可选位;`armPlanReviewApproval` 的选项 +1 可选位;`ReopenCardVerdict` 的拒开臂 +1 可选位 `dismissedByUser`;`SelfHealOutcome` 的 `plan-review-reopen-failed` 臂 +1 可选位 `dismissedByUser`;`ActiveRunSelfHealDeps.reopenPlanReview` +1 可选第二参 `{ trigger? }`;开发依赖引擎 `~7.33.1`(本包零运行期读点);peer sdk 地板 `>=11.3.0` 不动;零 wire 投影臂。
|
|
55
|
+
|
|
56
|
+
### Added
|
|
57
|
+
|
|
58
|
+
- **plan 审批卡:用户关掉的卡,自动触发不放回屏上**(CC-201):用户在 plan 审批卡上按 Esc、中止,或给了一个既不是批准也不是驳回的作答,本包在那张卡的作答口里**同步**记下「用户关卡」(按会话 × 任务;在作答口的同步段里记,排在卡后面的注入件下一拍撞 409 时记号已经在)。`reopenPlanReviewCard(taskId, { trigger })`:
|
|
59
|
+
- `trigger: 'automatic'`(注入件撞 409 的自愈、补捞、卡离屏补一拍、对账这类自动腿)撞上记号 ⇒ `{ reopened: false, dismissedByUser: true }`,不铸新卡 —— 修前这一拍会在 0.1–1.7 秒内把卡放回屏上,用户紧接着打的命令与回车落在新卡上,等于又批准一次。
|
|
60
|
+
- `trigger: 'user'`(用户自己的下一个动作,如他手打的消息撞 409)照开;卡回到屏上即撤记号。
|
|
61
|
+
- 同一条任务长出**新的** plan 门时照开:宿主经 `readGateInstanceKeys`(新具名型 `PlanReviewGateInstanceReader`,答「这条任务此刻挂着的 plan 门实例键集」)交读口,本包在用户关卡那一刻读一次、在自动触发判定时再读一次,出现一枚不在被关集合里的实例 ⇒ 视为新门、照开并撤旧记号。读回的集合在落定那一刻取快照(宿主交出自己就地改的活集合也认得出新门)。读口缺席、读不出、抛错、超时、关卡那一刻集合为空 ⇒ 证不出是新门 ⇒ 保守拒开。同步形(不带 `presentationReceiptMs`)读不了实例,撞记号即拒。
|
|
62
|
+
- 异步形的判定读回来之后、铸卡之前再核一次:这期间这道门上交了决断(包内看得见的:决断投递在飞、决断性作答;宿主报的:新可选谓词 `decisionHandedOver(taskId)` 答 `true`)⇒ 拒开并回 `_sema_decisionInFlight`(答案在路上,不是「卡是你关的」)。条目这期间换了一枚 ⇒ 这次评估不铸卡,判决按**此刻**条目如实说:用户又关了一张 ⇒ `dismissedByUser`(新记号保留);卡已被放回屏上(用户自己叫回、或同时在路上的另一次自动触发已把新门的卡放上去 —— 在本次回执窗内等那一次落定,等回来再读一次此刻条目:仍是那一次且上屏了才答卡在屏上,等的期间又换了一枚就按新条目说)⇒ `{ reopened: true, firstSight }`;条目没了(会话换代、记号账超上限被挤掉)⇒ `{ reopened: false }`(不带 `dismissedByUser`)。
|
|
63
|
+
- 每条任务至多一次在路上的重开:同一会话 × 任务上一次重开的卡已发布、回执还没到时,后到的重开(任何触发者,含 `trigger` 缺席与同步形)**并入**那一次 —— 不铸第二张卡、不退役前一张的作答口;异步形等那一次落定(上限 = 自己的回执窗),上屏 ⇒ `{ reopened: true, firstSight: false }`,没上屏 ⇒ `{ reopened: false }`;同步形当拍答 `{ reopened: true, firstSight: false }`(同步形「发布即答重开了」的既有语义)。修前:两次重开错开到达时屏上出现两张新卡,答前一张落空。
|
|
64
|
+
- 撤记号:决断性作答(包内作答口的决断臂、宿主的 `notePlanReviewAnsweredIfDecisive` 报的那一次 —— 但包内作答口已按那张卡自己展示的选项判为「非决断」的**同一张卡**,宿主记账口不再撤)、成功重开(同步形即时;回执模式回执到手时;只撤**铸卡之前**那一枚 —— 铸卡到回执之间用户又关的这张新卡是一枚新记号,不被这次成功撤掉)、`clearPlanReviewUserDismissals(sessionKey?)`(会话换代时宿主调;缺省 = 默认会话;只清这一个会话)。
|
|
65
|
+
- `trigger` 缺席 = 与 0.83.3 同答(不判记号、不读实例;唯一例外是上一条的并入:同一任务上一次重开还在路上时不再铸第二张)。`armPlanReviewApproval` 的选项多一位 `readGateInstanceKeys`(首呈卡被关时同样按门实例记)。接入文档 **§104a P-1–P-7**。
|
|
66
|
+
- **409 自愈腿接「用户关卡」**(CC-201):`ActiveRunSelfHealDeps.reopenPlanReview` 多一个可选第二参 `{ trigger }`,由本包按提交出身映射 —— `submissionOrigin: 'user'` ⇒ `'user'`,`'injected'` ⇒ `'automatic'`;出身缺席 ⇒ 只传 `taskId`(与 0.83.3 同一调用)。重开口回 `dismissedByUser: true` 时,结局 `plan-review-reopen-failed` 带同名位(严格 `true`),`activeRunSelfHealRow` 出整行:用户形、系统注入件形、steer 落 `queued` 的两形各一句 ——「计划在等你审,卡是你关的,sema 不自己放回;发条消息叫回来」—— 不给「保住对话」那条路、不给引擎决断入口(卡回得来,不是「重开不了」);端仍可经 `rowFor` 整行覆写。接入文档 **§104a P-8**。
|
|
67
|
+
- **座位审批请求补强制位 / 强制理由词 / 报价缺席因由 / 出身词**(CC-203):`ToolPermissionRequest` +4 可选位 `mandated?: true` / `mandate?: ApprovalMandateWord` / `ruleOffersAbsence?: string` / `origin?: string`(`TOOL_PERMISSION_REQUEST_KEYS` 10 → 14),语义原样来自卡请求同名位。`ruleOffersAbsence` 是寄存行那条腿的强制表达(行上没有 `mandated` 位,强制只以它 `=== "mandated"` 出现)—— 它不上座,同一张卡在卡上与座位上对 `approvalIsMandated` 就会不同答。新过境口 `toolPermissionRequestAskBits(card)`:与卡请求同一把窄读 —— `mandated` 只认自有键上的严格 `true`;`mandate` 只认闭集六词之一(精确字节,自有键);`ruleOffersAbsence` / `origin` 只认自有键上的非空串(原文字节,开集)。对经这只过境口铺座位请求的宿主,表外词、大小写变体、空串、非串、含糊真值、原型链上的同名键 ⇒ 那一位不上座位请求;四位各自过、互不推。座位校验器不校这四位(登记为刻意豁免:为一个位拒掉整条审批请求 = 把安全 ask 交给引擎超时拒绝);不经过境口自铸座位请求的宿主要自己按同一把窄读。座位上读这几位用卡请求的读口:`approvalIsMandated(req)` / `readApprovalMandate(req)` / `askSurvivesPosture(req, facts)`。接入文档 **§104a S-1–S-3**。
|
|
68
|
+
- **`<system-reminder>` 剥离 / 解包口上公面**(CC-205):`stripSystemReminderBlocks(text)` 删掉文本里**所有**成对的块;`unwrapSystemReminder(text)` 在整条文本以开标签起头、以闭标签收尾时返回两标签之间的正文(头一个开标签到末尾闭标签)。开标签文法只认引擎铸的两形:裸 `<system-reminder>` 与恰一个 `mark="<22 位 base64url>"` 属性的 `<system-reminder mark="…">`;闭标签逐字 `</system-reminder>`。块语义与 0.83.3 包内行为逐字同(只改时间复杂度,线性),见接入文档 **§104a-1**。
|
|
69
|
+
|
|
70
|
+
### Changed
|
|
71
|
+
|
|
72
|
+
- 接入指引订正两条(文档,无代码改动):① 结果帧 `errors[]` 与本包合成的终局行正文自 0.83.1 起在铸点洗过凭据,它们是**展示文**,不是语义判据 —— 判错误类型读结构位(`_sema_error_code` / 终态原因位),文案只用于上屏;② §101 的载体映射把审批入参归进凭据网射程是指引错误 —— 审批入参是授权面,用户必须看见要批的真参数:展示载体用 `displayUntrusted(x, { credentialUrls: false, credentialWords: false })`(只中和控制符 / 双向与格式字符 / 折行),截断要显式标出。接入文档 **§104a′**。
|
|
73
|
+
- 三处改按**自有键**读(与过境口同一把判法):审批卡两条腿(活卡帧 → 卡、待决行 → 卡)盖 `mandated` 只认自有键上的严格 `true`;判据口 `approvalIsMandated` 读 `mandated`(自有键上的严格 `true`)与 `ruleOffersAbsence`(自有键上的串 `"mandated"`)。JSON 形帧 / 行 / 卡请求零变化;原型链上挂着的同名键不再被当成引擎铸的强制位 / 缺席词。
|
|
74
|
+
- `<system-reminder>` 两只口改线性时间实现(修前:解包口在开标签后跟一长串空白 / 换行时回溯到秒级;剥离口在大量未闭合开标签上平方)。只改时间复杂度:块语义与 0.83.3 逐字同(见 §104a-1),包内两处读点(委派子代的提示词剥壳、自动模式分类器裁决的解包)答案不变。
|
|
75
|
+
|
|
76
|
+
### Gates
|
|
77
|
+
|
|
78
|
+
- 新门 `run-plan-review-dismissal-test.mjs`(94 格):用户关卡之后自动触发零新卡 / 记号在作答口同步段里成立 / 判决形只多 `dismissedByUser` 一位 / 反复自动触发(同步形、异步形)仍零新卡 / 用户动作照开并撤记号 / 再关再记 / 决断性作答撤记号(宿主记账口与包内作答口两路)/ 非决断作答照记 / 会话换代按会话清 / 记账点源码锚 / 重开身份卡同样记 / 按会话 × 任务分桶 / 非 plan 卡零记号 / 判定读点全文件恰一处且在在飞守卫之后、铸卡之前 / 门实例:同门拒、新门开并撤旧记号、多一枚实例即新门 / 证不出是哪道门的七形保守拒开(关卡那一刻读 null / reject / 空集 / 同步抛,判定那一刻读口抛 / 缺席 / 挂死 —— 有界等待)/ 首呈卡同样记并按门实例判 / 判定读在路上时交了决断(宿主记账口一形、决断投递在飞一形、宿主交出窗谓词一形)⇒ 拒开回「答案在路上」、被批准那张卡的作答入口不被退役 / 谓词 false 不挡、抛错保守拒开并回不带位的 `{ reopened: false }` / 陈旧评估按此刻条目分说(又关一张 ⇒ 卡是你关的;用户叫回、另一次自动触发已放回 ⇒ 卡在屏上;叫回并批准 ⇒ 答案在路上;会话换代、超上限挤掉 ⇒ 不带 `dismissedByUser`)/ 陈旧评估等放回落定之后再读一次此刻条目(等的期间用户叫回又关掉 ⇒ 卡是你关的;叫回并批准 ⇒ 答案在路上;会话换代 ⇒ 不带位)/ 新门上两次自动触发同拍恰一张新卡(含宿主入队延迟 60 / 200 ms 时陈旧评估等那一次落定、另一次自动触发的回执晚到时同样等;回执窗小于入队延迟 ⇒ 两次都答不带位的拒开)/ 新门卡没上屏不留残留条目(残留条目会占记号账一格、把别的任务最早一枚真记号挤掉)/ 每任务单飞:错开到达的两次自动触发、用户触发并入在路上的自动触发、同步形与异步形交错 ⇒ 恰一张新卡、前一张的作答入口不被退役 / 读口回宿主活集合按快照比较 / 包内作答口判为非决断的同一张卡,宿主记账口不撤记号 / 解析口单源且不上公面 / 回执窗竞态只撤铸卡之前那一枚、回执窗尽不撤 / `trigger` 缺席与 0.83.3 同答(同步形、异步形,零读口调用)/ 每会话记号账有界 / 表外触发者词按自动判。
|
|
79
|
+
- 扩门 `run-selfheal-reopen-test.mjs`(G13 段:13 格 + 1 枚判据自证):重开口回「用户关卡」⇒ 结局带位 / 用户形、注入形、steer `queued` 两形整行 / 不问宿主「保会话」出路、不给引擎决断入口 / 处置档仍是 not-delivered / 端整行覆写口仍优先 / 含糊值不打标 / 与「决断在飞」同在时按在飞说 / 注入口第二参按提交出身映射(缺席 ⇒ 单参调用)/ 注入件 steer 落 `queued` 同样带触发者 / 端到端(包内重开口当注入口:用户关卡后注入件零新卡、用户下一条消息卡回来)。
|
|
80
|
+
- 扩门 `run-seat-contract-keys-test.mjs`(K 段):四位进座位键集 / 过境口在公面 / 卡请求带位 ⇒ 原样 / 只管自己那几位 / 带位的座位请求过座位门且卡请求读口读得出 / 寄存行形(只有缺席词)的卡请求过境后座位上 `approvalIsMandated` 与卡上同答 / 缺席词按开集原文过 / 九种负样本零键 / 原型链零键 / 缺席零键 / 非对象入参零键不抛 / 只有词不长位 / 只有位不编词 / 坏形不拒整条请求;H 段覆盖表随四位豁免登记自动投毒对账。
|
|
81
|
+
- 扩门 `run-durable-card-display-keys-test.mjs`(⑰ 段 +4 格):两条腿上只在原型链上的 `mandated` 不上卡 / 自有键正控 / 判据口 `approvalIsMandated` 对只在原型链上的 `mandated` / `ruleOffersAbsence` 答 `false`、自有键照认。
|
|
82
|
+
- 扩门 `run-client-core-pure-test.mjs`(㊲b 段,B2 段地板 440 → 469):两只剥离口在根入口 / 生产方真形正样本四形 / 非引擎形负样本七形 / 截断不吞 / 无块逐字节原样 / 解包整条、非整条、非引擎形 / 与 0.83.3 实现逐条对拍(门里冻结两条旧正则,3 万条定种子随机串两只口逐字节同,另有一枚语料覆盖自证)/ 四个边角与 0.83.3 同答(两块相接、两块夹正文、字面开标签与后面的真块配对、嵌套)/ 分类器拒绝理由引用开标签(裸形、带 mark)照认且 reason 逐字 / 解包是铸造口的逆(正文含字面开标签)/ 正文含字面开标签的真块整块删净 / 正文两端空白全去、正文字面闭标签照当正文 / 截断信封解包原样 / 线性时间两格(n 与 4n 比值法)。
|
|
83
|
+
|
|
84
|
+
### Known limits(本版新增)
|
|
85
|
+
|
|
86
|
+
- 包分不出空作答是谁送的:宿主的 hook 以空作答拒掉 plan 卡时,本包同样记成「用户关卡」,自动触发随之拒开(方向保守;行句会说成「卡是你关的」)。
|
|
87
|
+
- 首呈卡的呈现口(`armPlanReviewApproval`,运行流里 plan 停泊帧到达时调)既不判也不撤「用户关卡」记号:同一停泊帧被重放时首呈卡照样再立一次;立回屏上之后记号仍在,这期间的自动触发照拒并答 `dismissedByUser`(尽管卡在屏上)。
|
|
88
|
+
- 同步形(不带回执窗)的重开读不了门实例:用户关过这条任务的卡之后,同一条任务长出的新 plan 门要等用户的下一个动作、决断性作答或会话换代才会被自动触发放回。
|
|
89
|
+
- 「用户关卡」记号账每会话至多 32 条,超出按先后挤掉最早一条;被挤掉那条任务的自动触发回到 0.83.3 的行为。
|
|
90
|
+
- 座位校验器不校四个新位;坏形值由过境口挡在座位请求之外,不经过境口自铸座位请求的宿主要自己按同一把窄读。
|
|
91
|
+
- 宿主交出窗谓词只在异步判定的读后复核里问一次:没有记号的自动触发、同步形、`trigger` 缺席、用户动作都不问它(那几条铸卡前没有等待,由宿主调重开口之前自己的守卫负责)。
|
|
92
|
+
- `<system-reminder>` 开标签文法本身不上公面:要按开标签切块(不只是删 / 解包)的端仍需自持文法并与本包两只口按正负样本对表。
|
|
93
|
+
- `stripSystemReminderBlocks` 从一个开标签配到它之后最近的闭标签:用户正文里一个没闭合的字面开标签会与后面一个真块配对,两者之间的正文一起被删掉(0.83.3 起的既有行为,本版不改)。
|
|
94
|
+
- 完整台账见接入文档 §104 末行「包侧缺口」。
|
|
95
|
+
|
|
52
96
|
## 0.83.3(2026-09-26)
|
|
53
97
|
|
|
54
98
|
> 主题:patch —— 三件入参 / 来源面放宽,零新运行期导出。① `--settings` 一类入口交来的 flag 来源设置里写的 hooks 进请求体(此前只投 managed / user / project / local 四源,flag 来源的 hooks 到不了引擎);flag 来源排在 local 之后,设置来源的拼接次序其余不变(managed 仍在最前)。② `listAllPersistedRules` 的入参只要 `list` 一口(新具名型 `RulesListFacade`;其余四口作为可选成员认得 —— 内联对象字面量、参数不写型的箭头照样可传,拼错口名照样报错),只装读 / 撤两口的宿主可以直接用。③ 模型身份判据认「目录缺省代替启动环境里的模型名」:宿主报本机引擎接受目录缺省(`engineAcceptsCatalogDefault: true`)时,只用模型目录文件的配置判 `resolvable`;报 `false` 判 `not_resolvable`;不报仍判不出(`unknown`)。根公面运行期导出 1273 不变;公面类型 +1(`RulesListFacade`);`HookSettingsSource` +1 员;`ModelIdentityLanes` 的目录一格闭集 +`complete`;`LocalEngineModelIdentityReading` +1 可选位;peer sdk 地板 `>=11.3.0` 不动;零 wire 投影臂。
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
|
|
|
35
35
|
|
|
36
36
|
## Scope
|
|
37
37
|
|
|
38
|
-
**Version:** 0.83.
|
|
38
|
+
**Version:** 0.83.4
|
|
39
39
|
|
|
40
40
|
- **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
|
|
41
41
|
B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
|
|
@@ -292,7 +292,7 @@ guard still cross-checks the table by name).
|
|
|
292
292
|
| `scripts/run-client-core-pure-test.mjs` | Consumer-view behaviour of every moved-in module; segment floors (B1–B8 + C) that only move up |
|
|
293
293
|
| `scripts/run-client-core-portability-test.mjs` | Kernel / A-layer / index import closures, the runtime-dependency equality gate, barrel reachability, and a real esbuild `--platform=browser` bundle |
|
|
294
294
|
| `scripts/run-client-core-diff-test.mjs` | Differential equivalence against the CLI reference bridge + replay-id invariant + ledger round-trip |
|
|
295
|
-
| `scripts/run-seat-contract-keys-test.mjs` | The seat IPC contract: verb list ↔ SPEC ↔ types, element-wise |
|
|
295
|
+
| `scripts/run-seat-contract-keys-test.mjs` | The seat IPC contract: verb list ↔ SPEC ↔ types, element-wise. Since 0.83.4 the seat approval request also carries the four card bits that say whether an approval must be asked and where it came from (`mandated`, `mandate`, `ruleOffersAbsence` — the parked-row form of mandated, so the seat and the card give the same `approvalIsMandated` answer — and `origin`, which the resident-posture predicate reads), and they reach the seat through `toolPermissionRequestAskBits`, which reads them exactly as the card request does — a strict own `true`, one of the closed mandate words byte for byte, a non-empty own string — so for a host that builds the seat request through that crossing an off-list word, an empty string or an inherited key never lands on the seat; the seat validator does not become a second judge that could drop a safety ask over one bit, which leaves a host that builds the request by hand to apply the same narrow read itself |
|
|
296
296
|
| `scripts/run-approval-frame-keys-test.mjs` | The tool-approval frame key mirror, element-wise against the SDK's runtime anchor (one carve-out: AHEAD_OF_ANCHOR entries — keys the server already emits but the SDK anchor has not caught up to — may lead by one generation; the gate turns red the day the SDK catches up, forcing the entry's removal — the register is occupied again — this time by the bit that says a saved allow rule cannot retire a given approval card, carrying both the release that minted it and the byte coordinates that prove it, so the lead is a dated record rather than an exemption; its predecessor left the register the other way, by being retired upstream rather than by the anchor catching up). Beside the key mirror it now guards three further faces of that bit: the closed word table the durable leg reads it through must be **the very array object the SDK exports**, not a same-looking copy — reference identity, because an equal-contents check still permits a second table that diverges the day upstream adds a member; the one predicate a client is meant to call answers over both legs — the live card's bit and the parked row's absence word, which is all the row carries, since the row has no such bit at all — and answers `false` for a malformed value exactly as its presence-only siblings do, a strictness the upstream mint shares; and the one sentence minted for it must never point the reader at writing a rule, since a rule written in answer to a mandated question can never take effect where it was written. The same bit's key also has to reach the card request itself, which the SDK's card anchor does not list — a fact that arrives at the package boundary and stops there is the shape of defect this file's guards exist to catch. Since 0.83.2 the frame also mirrors `mandate` (the word a mandated question stands on) ahead of the SDK anchor, with its own exit condition; the released server carries it from 7.102.0, so the allowance for the server fixture was removed by its own exit condition. The installed engine must declare the member with the six-word closed type, and the key must reach the card request. |
|
|
297
297
|
| `scripts/run-segment-authority-single-source-test.mjs` | The authoritative-segment replacement verdict, single-sourced. `text_end.content` and the `text_delta` stream stopped being byte-identical the day the engine started redacting the former through the same filter as the result, so every consumer now has to decide six ways what to do with the segment it has half-emitted — and until this release that decision existed **twice**: once here for the transcript lane, once in the shell for the print lane, hot-fixed a version apart. The verdict is now one pure function both lanes call, and the guard pins it on the quantity that actually decides the outcome: whether the authoritative text still *starts with* the bytes that already left, not whether a flush has happened — the latter is a precondition, and anchoring on it withholds a perfectly ordinary answer. Each of the six forms is checked with its counter-case, the prefix length is pinned to UTF-16 code units against a non-ASCII sample whose UTF-8 byte count differs (slicing by bytes leaves the very thing being redacted on screen), and the withheld-segment ledger is compared by normalised equality rather than substring, because a short redaction marker quoted in an unrelated later answer would otherwise suppress that answer entirely. The same file pins the session-level memory-capture declaration to one mint point — the wire value is a single-member closed set, and a consumer that spells it wrong gets a loud refusal rather than a silently dropped privacy request — and pins the SDK URL/health transit to be the **same function reference**, since wrapping it would discard the one guarantee the transit exists for. A last section strips comments with the TypeScript parser and asserts the second expression has not grown back |
|
|
298
298
|
| `scripts/run-print-bash-iserror-test.mjs` | The print lane's Bash `is_error` authority (structured over regex). A second section pins where the denial classification word lands on this lane: on the message envelope, never inside the tool-result block, because that block is forwarded verbatim to the provider on compaction and a self-minted key there is the shape of an old, real defect. A word outside the upstream table — or an empty string, a non-string, or nothing at all — mints no key rather than a guess, and the word never moves the error flag, because attribution does not decide anything |
|
|
@@ -362,7 +362,7 @@ guard still cross-checks the table by name).
|
|
|
362
362
|
| `scripts/run-client-core-singleton-test.mjs` | Module-level singletons ⇄ `docs/refactor/p1-scan/singleton-manifest.json`, **both directions**: an unregistered singleton is red (registering it forces someone to answer "what if this got duplicated"), a stale entry is red, and the `dupRisk: high` count only goes down |
|
|
363
363
|
| `scripts/run-catalog-loader-gates-test.mjs` | The model-catalog candidate chain (`loadCatalogWithSources`) and the provider device-code seam: offline ⇒ `bundled` with an honest `online.reason`, a good source ⇒ `online` plus a cache write, a second offline run ⇒ `cacheHit`; the three hostile source shapes (malformed JSON, `schemaVersion: 99`, off-domain `http`) each fall through to the bundled table, and an off-allowlist target is **never dialled** — including a `302` to another host, proven by a real loopback server's hit counter staying at zero; a one-byte edit to `catalog.sha256` drops that source while an unavailable sidecar only warns; and the device-code poller's `pending → ok` / `expired` arms run against a real loopback HTTP server with an injected clock |
|
|
364
364
|
| `scripts/run-abortable-sleep-test.mjs` | The shared `abortableSleep(ms, signal)` leaf (consumed by `workflowClient.ts` and `agentSession/backgroundView.ts`'s poll backoff): normal timeout resolution, immediate wake-up on `abort` mid-wait, `clearTimeout` really firing on that path, and a post-resolve late abort staying a no-op |
|
|
365
|
-
| `scripts/run-durable-card-display-keys-test.mjs` | The durable approval row's two display keys survive the row→card recast in `surfaceFsApprovalAndDecide`: `governanceForced` stamps on strict `true` only (absence is "no evidence", never `false`), `ruleSuggestions` passes through the same shape-narrowing reader as the live-frame leg and lands on the **read-only** card key — plus a standing pin that the durable leg never stamps the redeemable `ruleSuggestions` card position (the `/decide` body has no rule slot; offering a "don't ask again" option there would be an affordance nothing can honour), and a section for the parked twin of the classifier-unavailable fact: the upstream declares that key on the parked action itself, verbatim and under the same name as the synchronous ask, so this leg reads it rather than guessing a carrier name the way the deliberately unprojected keys must. The guard drives both legs with the same cause and asserts the card ends up byte-identical either way — the observable consequence of one reader serving two key paths, and the thing that silently diverges the day someone writes a second copy. Its own reach is printed rather than implied: what is proven is the package-boundary promise "on the row ⇒ on the card", not that today's engine flattens that key onto the pending row. A further section covers the two display facts the recast had been dropping for far longer. One of them the row has carried all along under a DIFFERENT NAME than the live frame uses — the frame puts it at the top level, the row nests it under the risk descriptor — and that difference in name is exactly why it went unnoticed; unlike the keys this leg deliberately refuses to project, its carrier is witnessed in the engine's own artefact rather than guessed. Neither is decoration: the shell's stand-aside arm reads them, so a call that matched a remembered allow rule which could NOT silence it looked like an ordinary ask on the durable path and was auto-approved with no card at all. Both land on the SAME card slot the live leg uses (one shape for the ends), verbatim bytes, present only when non-blank, never folded into an empty string — and the guard pins the discipline in both directions, including that a top-level key the upstream row does not actually have must still not grow this position. The security-class approval bit (`requiresRealApproval`) rides a parked row's card when the row carries it at top level, on strict `true` only, while look-alike nested carriers are ignored; this is pinned with a constructed row, because today's pending list does not carry the bit yet. A second, separate bit (`irreversibleParkGate`) marks a parked card whose row sits on the irreversible-ask gate kind — a gate-kind fact that covers asks the engine flagged for real approval at the first decision plus safety-tightened gates, with a known engine gap for approval demands raised only on a storage recheck — on the park path only, on the exact gate word only, and never in place of the real bit. Since 0.83.2 the recast also carries the row's ask origin (`origin`) exactly as the live-frame leg does — a non-empty string, verbatim, open vocabulary, never invented or defaulted — and the word a mandated question stands on (`mandate`), accepted only when it is one of the six known words and read back through `readApprovalMandate`; a word outside that set, or a malformed value, leaves the card without it. The word never adds the separate mandated key to the card, but a known word on its own makes `approvalIsMandated` answer true, because the engine only sends the word as the whole reason for that bit. An `origin` inherited through the row's prototype chain is not read. |
|
|
365
|
+
| `scripts/run-durable-card-display-keys-test.mjs` | The durable approval row's two display keys survive the row→card recast in `surfaceFsApprovalAndDecide`: `governanceForced` stamps on strict `true` only (absence is "no evidence", never `false`), `ruleSuggestions` passes through the same shape-narrowing reader as the live-frame leg and lands on the **read-only** card key — plus a standing pin that the durable leg never stamps the redeemable `ruleSuggestions` card position (the `/decide` body has no rule slot; offering a "don't ask again" option there would be an affordance nothing can honour), and a section for the parked twin of the classifier-unavailable fact: the upstream declares that key on the parked action itself, verbatim and under the same name as the synchronous ask, so this leg reads it rather than guessing a carrier name the way the deliberately unprojected keys must. The guard drives both legs with the same cause and asserts the card ends up byte-identical either way — the observable consequence of one reader serving two key paths, and the thing that silently diverges the day someone writes a second copy. Its own reach is printed rather than implied: what is proven is the package-boundary promise "on the row ⇒ on the card", not that today's engine flattens that key onto the pending row. A further section covers the two display facts the recast had been dropping for far longer. One of them the row has carried all along under a DIFFERENT NAME than the live frame uses — the frame puts it at the top level, the row nests it under the risk descriptor — and that difference in name is exactly why it went unnoticed; unlike the keys this leg deliberately refuses to project, its carrier is witnessed in the engine's own artefact rather than guessed. Neither is decoration: the shell's stand-aside arm reads them, so a call that matched a remembered allow rule which could NOT silence it looked like an ordinary ask on the durable path and was auto-approved with no card at all. Both land on the SAME card slot the live leg uses (one shape for the ends), verbatim bytes, present only when non-blank, never folded into an empty string — and the guard pins the discipline in both directions, including that a top-level key the upstream row does not actually have must still not grow this position. The security-class approval bit (`requiresRealApproval`) rides a parked row's card when the row carries it at top level, on strict `true` only, while look-alike nested carriers are ignored; this is pinned with a constructed row, because today's pending list does not carry the bit yet. A second, separate bit (`irreversibleParkGate`) marks a parked card whose row sits on the irreversible-ask gate kind — a gate-kind fact that covers asks the engine flagged for real approval at the first decision plus safety-tightened gates, with a known engine gap for approval demands raised only on a storage recheck — on the park path only, on the exact gate word only, and never in place of the real bit. Since 0.83.2 the recast also carries the row's ask origin (`origin`) exactly as the live-frame leg does — a non-empty string, verbatim, open vocabulary, never invented or defaulted — and the word a mandated question stands on (`mandate`), accepted only when it is one of the six known words and read back through `readApprovalMandate`; a word outside that set, or a malformed value, leaves the card without it. The word never adds the separate mandated key to the card, but a known word on its own makes `approvalIsMandated` answer true, because the engine only sends the word as the whole reason for that bit. An `origin` inherited through the row's prototype chain is not read, and since 0.83.4 neither is an inherited `mandated` on either leg — both legs stamp it from an own strict `true`, the same reading the seat crossing uses; the single judge `approvalIsMandated` reads `mandated` and `ruleOffersAbsence` the same way, so an inherited key no longer makes it answer true. |
|
|
366
366
|
| `scripts/run-session-memory-status-test.mjs` | The session **memory-status** read face (S-53): the two judgements three clients would otherwise each get wrong. First, *same status, different code* — this route's 404 carries two unrelated meanings (`not_found.session` = unknown or non-owned session; `not_found.route` = a pre-7.53 server that has no such route at all), so dispatching on the **status** would report "your deployment lacks this surface" as "your session does not exist". The verdict is anchored on `errorCode`, the two 404s are pinned to **different** verdicts, and — the load-bearing negative control — a 404 carrying **no** code falls to `failed` rather than guessing either way, since a wrong guess in either direction is a false statement a user would act on. 501 is allowed a codeless fallback because both of its arms mean the same thing here, and `capability.*` stays split from `feature.*` because those two share a status while their dispositions are opposite. Second, *absence means something different per key*: `optOutSource` and `lastCaptureAt` are legitimately absent on a **healthy** session (a zero-history session really is `{captureOptedOut:false, committedCount:0, foldedCount:0}` with no degradation at all), so reading absence as "off/none/0" asserts something unprovable. Two combined readers are pinned: capture opt-out is read from **both** its keys (a record-store fault yields `indeterminate`, never `active` — the difference between "your conversation is being remembered" and "nobody knows"), and last-capture is a **three-state** read whose discriminator is the *other* key, because `lastCaptureAt`'s absence alone covers both "ledger unreadable" and "genuinely no contributions" and therefore decides nothing; the two shapes are pinned to different verdicts so a single-key read turns red. The thin wrapper is the only IO: it never throws, drops malformed keys to absence rather than trusting them (an unreadable value must answer "don't know", never render as truth), refuses to spend a request on an empty `sessionId`, and passes `signal` through untouched |
|
|
367
367
|
| `scripts/run-crash-converged-projection-test.mjs` | The `crashConverged` read face on `GET /v1/approvals` (L-38): what the *previous life* of a crashed local engine left behind, projected for every client. Three judgements are pinned. First, **absence is not an empty list** — a missing key (an older server, deps not present, or a carrier that is not an array at all) returns `undefined`, and the client renders nothing; an empty array returns a present zero-count object, which is the server actually saying "none". Folding the first into `{total:0}` would have the client assert "nothing was left behind" on a surface a person uses to decide whether it is safe to re-run something — the worst possible direction for a false statement — so the two cases are pinned to different **return shapes** and a test asserts the two verdicts are unequal. Second, bucketing is a **four-term conjunction**: `orphanState === 'pending'` *and* `resumeSafe === true` *and* both approval-evidence keys (`originalDecision`, `decidedAtMs`) absent. A fifth term rejects any row carrying an **accessor**, and accessors are never invoked at all — reading one means synchronously running someone else's code, and `catch` catches throwing, not *never returning*, so a looping getter would pin the startup thread forever (the row cap does nothing against that shape). The same rule covers the three untrusted reads outside the row as well — the envelope's `crashConverged` key, the carrier's `length`, and every numeric index are read as own property *descriptors* and only data descriptors are used, so accessors and prototype entries read as absent and are never invoked. Such a key is treated as absent: if it was a required field the row is counted as dropped, if it was optional or additive the row survives without it. That also closes the ordering attack, since spreading runs getters in property order and an earlier one could `delete` the approval evidence before it is ever copied (measured before the fix: such a row reached the resume-safe bucket), and the check therefore moves ahead of the read, onto the property descriptors — from which the snapshot is then built directly, because checking descriptors and *then* spreading is two independent observations of the same row, and a non-throwing proxy can make the two `ownKeys` calls disagree (first showing `originalDecision: 'approve'` so the row reads as plain data, then omitting that configurable key so the snapshot loses the evidence; measured before the fix: the dangerous row reached the resume-safe bucket after exactly two enumerations, and after it, one). Keys are written with `Object.defineProperty` rather than plain assignment, because `'__proto__'` is a legal own enumerable key and `o['__proto__'] = x` does not store a value — it calls the prototype setter, letting a row whose own properties are all plain data (so the accessor gate never fires) inject a prototype whose `sessionId` getter deletes the approval evidence from the snapshot during validation; `defineProperty` fires no setter, so the key survives as ordinary additive data and the snapshot keeps `Object.prototype`. A row that simply arrives with a custom prototype is treated the same way, since the snapshot only enumerates own properties: approval evidence sitting on the prototype would never reach it, and a perfectly ordinary object with no proxy and no accessors could otherwise be called safe to re-run — real bodies come from `JSON.parse` and always carry `Object.prototype`, so nothing genuine trips it). Validation itself runs on a **null-prototype** dictionary and the bucketing verdict is carried out of that same pass rather than re-read from the delivered row, because every property lookup on an ordinary `{}` reaches `Object.prototype`: a polluted `sessionId` getter there would delete the approval evidence from the snapshot mid-validation and send the row to the safe bucket (measured before the fix). The row handed to the client is still an ordinary object — the null prototype is an implementation detail of the check, not of the value) — real JSON bodies are all data properties, so only a middle-layer-synthesised payload ever trips it, and it too lands in the human bucket rather than being dropped. The `decided` arm means the human had already approved and side effects may be half-landed, so it always goes to the human bucket, as does `resumeSafe === false` and — the last two terms — any row whose own fields contradict each other, since `pending` claims nothing ran while that evidence says somebody pressed approve. Deciding "not safe" costs one extra question (recoverable); deciding "safe" wrongly has somebody re-run work that already partly happened (not). A 2x2 truth table pins that exactly one cell is resume-safe, so reading either key alone turns red, and the contradictory rows are routed to the human bucket rather than dropped — they are real orphans, and the ones most worth showing. Third, unreadable rows are **dropped and counted**, never thrown and never passed through: the product is declared as `CrashConvergedRow`, so letting a row missing a required field — or carrying one of the wrong type — past would be a lie at the type level, and the closed literal discriminators (`decision` / `cause` / `orphanState`) decide family membership rather than being an open vocabulary. The measuring stick stops at the **type** floor, though: degenerate-but-well-typed values (`ts: NaN`, an empty `toolName`) are kept, because swallowing a real orphan over a decorative field is the worse direction, and the one deliberate exception is `approvalId`, which must be non-empty to be a row identity at all. `dropped` is kept separate from `total` so unreadable rows never inflate "N approvals were affected"; each row is a **one-shot snapshot** — every own enumerable key is read exactly once, and validation, bucketing and the handed-back value all read that same snapshot, so additive upstream keys survive while a **non-idempotent** getter (one that never throws, just answers differently on a second read) can no longer erase the approval evidence between the check and the bucketing (measured before the fix: such a row landed in the resume-safe bucket while its checked value was `"approve"`). Hostile carriers are counted rather than allowed to reject: **every** touch of the carrier is guarded — envelope property reads, `Array.isArray` itself (it throws on a revoked proxy), the `length` read, each indexed read and each row's property reads — and a traversal that dies halfway returns absence rather than a half-counted total. A row that cannot be read never takes the batch with it: its own shape check is inside its own guard, so one revoked-proxy row costs a `dropped` tick rather than collapsing the whole projection to absence — which a client would have read as "this deployment does not offer the surface". Traversal goes by **numeric index, never the carrier's own iterator protocol**, because `for...of` hands the carrier the question of which rows exist: an array carrying an overridden `Symbol.iterator` can yield nothing (measured before the fix: a real orphan became `{total:0}`, which a client reads as "the server said there are none") or swap a dangerous `decided` row for a safe-looking one (measured: `fake-safe` was returned in place of `real-danger`). Row count is capped at 100000 and the cap is checked **before** the walk: requiring only a non-negative integer `length` does not stop a proxy trap reporting a billion, and this surface runs on the startup / `--resume` path, where a synchronous spin freezes the thread (measured before the cap: twenty million rows took 18.3 seconds and twenty million index reads; a billion does not come back). The honest boundary is stated rather than overclaimed — a proxy can still lie in its `length` or index traps, which is the same thing as a host injecting a lying transport — and the widening of `ApprovalsResourceLike.list()` is proven **additive** by really running tsc over a legacy `{ pending }` mock *and* over the real `AgentClient` path — the projector takes `unknown` precisely because a parameter shaped as "an object with an optional `crashConverged`" is a TypeScript weak type that the installed SDK's own `list()` return shape shares no property with, which only a real-client compile would have caught — with a known-red control so a clean run means the checker spoke |
|
|
368
368
|
| `scripts/run-self-orchestration-denial-test.mjs` | The three judgements behind a **denied self-orchestration request** (server 7.57.0), each of which all three clients would otherwise get wrong on their own. First, whether to retry at all is a **conjunction that may not be loosened**: HTTP 501 *and* an `errorCode` that is **exactly** `capability.self_orchestration_required`. That code shares its shape with every other `capability.*` 501, so dispatching on the prefix would drag "some other capability is not wired up" into the retry arm — those requests do not become acceptable once the two keys are gone, so the client would spend a request and then tell the user the wrong reason. Negative controls cover all four directions: a sibling `capability.*` code, a truncated or suffixed variant of the right one, a codeless 501 (it decides nothing, so it decides nothing — no guessing), and the right code under 500 / 400 / 503 or a string `"501"`. The classifier reads structurally rather than by `instanceof` (a host may inject its own transport; across realms or duplicate SDK instances an understandable error would read as unreadable), so a class instance, a bare `{status, errorCode}` literal and an error carrying those fields on its **prototype** all reach the same verdict — and a hostile proxy or a throwing getter yields `null` instead of throwing, because this classifier runs inside a `catch` block where anything it throws escapes the caller's own guard. Second, removing the intent is a **structural** operation, not wording: `selfOrchestration` sits at the top level while `ultracode` sits under `settings` — two different stamping legs — and a client hand-writing `delete` will miss the second one, which costs the user the same failure twice. The single stripper is pinned to touch exactly those two: other `settings` sub-keys and their values survive byte for byte, `deferTools` is left alone (pulling `Workflow` out would be a behaviour change, not a removal of intent), additive unknown keys survive at both levels, the input object is never mutated, `settings` is only dropped entirely when `ultracode` was really there and nothing else remains (an already-empty one is left as is), a non-object `settings` is not touched at all, an `ultracode` that only exists on the prototype does not count, and the whole thing is idempotent. The end-to-end leg runs a real `buildTaskRequest` product through it and asserts the stripped body still passes the registration gate key by key. Third, on the capabilities body, **absence is not "switched off"**: a pre-7.57 server has no `workflowsGate` key at all, so reading absence as "the engine says no" asserts something the server never said, and the mirror-image disease is folding an **unrecognised** `denial` into `null`, which would have the client render "nothing was denied" when the truth is "denied, for a reason I do not recognise". Five shapes are pinned — caps unreadable, gate absent, closed-set member, unknown value, accessor — with the unknown arm carrying the raw token (or an empty one when the value is not even a string) and never collapsing to `null`. All four untrusted reads go through own **data descriptors** only, and the guard pins the getter invocation count at zero, since `catch` catches throwing but not *never returning*; a descriptor trap that throws and a revoked proxy both yield honest absence rather than an exception — though *what* absence means differs by field, and the guard pins that split rather than a blanket rule: an accessor on `workflows`, `workflowsGate` or `engineCan` reads as absent, while an accessor on `denial` reads as `{unknown:''}`, because a key that is **not there** is the gate saying "nothing was denied" whereas a key that is there but cannot be read is "denied, and I could not read why" — folding the second into the first is exactly the false statement this face exists to prevent. Two further pins came out of an adversarial review. The exported retry list is **frozen at runtime**, not merely `as const`: the verdict hands out that same reference, so any consumer splicing it once would poison every later verdict in the process — the guard asserts `Object.isFrozen`, that four different mutation attempts leave it byte-identical, and that a verdict issued *after* those attempts still carries the original two entries. And the classifier reads `denial` only **after** both criteria have passed, since it is not a criterion but an extra field on the verdict: the guard pins the getter invocation count at zero for any error that does not match and at most one for an error that does. The scope line is drawn explicitly rather than overclaimed — "no getter ever runs" holds for `projectWorkflowsGate`, which reads **wire JSON** where every field is an own data property by definition, but not for the classifier, which reads a **thrown value** that may well be an SDK `APIError` class instance carrying `status` and `errorCode` on its prototype; insisting on own data descriptors there would report a perfectly readable error as unreadable, so that side promises only that it never throws. A final pin covers the **integration document's own worked example** rather than the library: the shipped SDK's `tasks.stream()` is an `async` generator, so calling it issues no request at all — the POST happens inside `streamRaw` on the first iteration, and a `try` wrapped around the `stream(...)` call itself can never catch the 501. A client following a submit-shaped recipe on the streaming leg would never run the classifier, and the whole strip-and-retry path would silently do nothing. The guard drives the **real** `TasksResource` against a fake transport, offline, and pins both halves: the synchronous leg is in flight the moment it is called, the streaming leg has issued zero requests after the call and raises on the first `next()` — and it does so through the **real** error path, with `openStream` returning an actual 501 `Response` that the SDK's own `errorFromResponse` turns into the typed error, pinning the `openStream`→`errorFrom` call order so a transport that stops minting `errorCode` cannot pass. The documented recipe is then **executed** rather than keyword-counted: exactly one retry, a second body that really lost both keys while every other setting survives byte for byte, the caller's own request object left untouched, one disclosure and only one, a second 501 propagating with the request count still at two, and — after the first 501 — an abort leaving the count at one with nothing disclosed. A last leg is type-level: `stripSelfOrchestrationIntent` carries an SDK `TaskRequest` overload, because the wide `Record<string, unknown>` form erases the caller's type and the document's "strip and resubmit" line would not compile without an unsafe cast; a real tsc run over a virtual file proves both the narrow and the wide path, with a known-red control — and it compiles the document's two recipes **verbatim**, extracted from the section itself, because a recipe that does not compile is a recipe that was never given: `{ transientOk: true, signal }` is a TS2379 under `exactOptionalPropertyTypes`, which no amount of prose review had caught. The last thing pinned is the one that would have been quietest of all: the SDK's `stream()` returns only on a `done` or `failed` frame, so a stream truncated mid-run — or yielding nothing at all — ends the `for await` just as normally as a completed one. The documented `runOnce` therefore tracks whether it ever saw a terminal frame and raises when it did not, the guard's success fixture emits a real terminal and asserts the handler received it, and a truncated-stream control asserts that shape is reported as a failure with no retry and nothing disclosed. That terminal-frame rule then needed one more turn of its own: the underlying reader returns *normally* when the signal is aborted, so the check as first written rewrote a user's cancellation into a generic stream fault — a client keying off `AbortError` to suppress the error would instead have shown a failure, or resubmitted. Cancellation is therefore checked first, a real-SDK case aborts from inside the handler and asserts the original `AbortError` survives with no retry and nothing disclosed, and the document is checked for that ordering. The harness runs the documented `handle` and `transcript.note` as real spies rather than pushing frames itself, the drive loop rethrows exactly as the document does, and the disclosure ledger is proven to be the caller's own array by a positive identity assertion — without which the cancellation leg's "nothing disclosed" would have been vacuously true. Each recipe is compiled **on its own**, with a preamble that declares only what a host supplies and injects no library symbol, since compiling them together let the second one borrow the first one's imports, and the preamble's own types are decoupled from what the recipes import so the "remove the imports and it must fail" control fails for the right reason — which is checked by attribution, not merely by redness. Ordering is the last thing to get right: the cancellation check must come before the truncation error but **both** must sit behind the terminal-frame test, because a cancellation that lands after the run already reported `done` would otherwise overwrite a real outcome — one that may have already had effects — with "cancelled", and a person reading that will run it again. Aborting from inside `handle(done)` and `handle(failed)` are both pinned to still report success, and the ordering assertion is anchored inside the streaming `runOnce` body rather than the section, since the section's first `throwIfAborted` belongs to the synchronous recipe and would have made a reversed streaming recipe pass — and that ordering check is now anchored on the TypeScript AST rather than on text, since a comment reproducing the two statements in the right order let a genuinely reversed body pass. One more timing fact had to be written into the recipe: a single SSE read buffers several frames and the SDK yields them back to back, so checking the signal only after the loop lets a cancelled run keep consuming the rest of the chunk — measured, an abort inside `handle(turn_start)` still swallowed the `done` that followed and reported success. The recipe therefore re-checks after every non-terminal frame. Finally, the behavioural matrix is no longer run against a copy of the recipe: both recipes are extracted from the document, transpiled, and **executed** with injected host objects, so the disclosure assertion really exercises the document's own `transcript.note(disclose(...))` line, and the synchronous leg gets the same full matrix the streaming one does |
|
|
@@ -371,7 +371,7 @@ guard still cross-checks the table by name).
|
|
|
371
371
|
| `scripts/run-type-superset-ledger-test.mjs` | The type/wire **superset ledger** (`docs/type-superset.json`): positions this package adds on top of a CC-shaped contract, each carrying the evidence for what CC's own type surface does or does not have there. Completeness is deliberately uneven and the ledger says so. The `_sema_*` private-key class is checked in **both** directions (a key in the source that never entered the ledger is red, naming key and file; a ledger row whose key left the source is red) — but only for keys written as literals, which is the convention the ledger mandates. A key assembled by string arithmetic is beyond what any static rule can enumerate, so the guard fails closed on every shape it *can* decide (a bare `_sema_` prefix is red wherever it appears, save one pinned guard site) and leaves the rest as a convention violation for review to catch, rather than claiming a completeness it does not have. The two hand-surveyed classes are only checked for coordinate and evidence integrity, never discovered. Both directions read the source through the **TypeScript AST**, not a text scan, and they read two different sets out of it. A *key site* is an identifier, or a string whose whole value is the key — so `'_sema_decision-v2'` is carried whole rather than truncated at the first non-identifier character into some *other* key that happens to be registered. A *mention* is the key appearing inside a longer string, which is prose, not usage. The staleness direction counts key sites only: a comment or a doc sentence left behind after the last real mint site is deleted must not keep the row alive (mutation-proven — with both the comment and the prose string untouched, removing the one real site turns the guard red). And because a prefix can be concatenated or interpolated into a key no static set will ever see, the bare `_sema_` literal is refused outright rather than traced: every occurrence is red except the single inline `startsWith` guard the sanitizer needs, because the set of expressions a bare prefix can travel through on its way to a concatenation is open-ended and enumerating it is always one form behind. Every row's `host` must still resolve, with the key being a real **member of that declaration** rather than a string occurring somewhere in the same file — `governanceForced`/`delegation` each live on two different shapes in one file, and a member commented out is a member deleted, which a text-shaped check happily reads as still present. And the direction worth the most: each machine-form `ccAbsenceEvidence` is re-derived from the row's own `key` — the ledger's recorded string must match that derivation verbatim, since a row quietly witnessing `\bnever_present\b` is green forever while watching nothing (mutation-proven: the same edit passes the unbound form and is caught by the bound one) — and the check runs against the names the installed `@sema-agent/agent-types` `.d.ts` set actually declares, parsed with the TypeScript AST rather than grepped, so a name CC merely mentions in a comment cannot force the row into the manual escape hatch and thereby retire the very witness that was supposed to fire the day CC declares that name for real. That escape hatch is gated by an allowlist living **in the guard**, not the ledger, so claiming it costs a reviewed diff. Missing material never reads as a pass, and the verdict splits by *why* it is missing: no TypeScript parser skips the suite before it starts; a missing `agent-types` still runs and prints the first three directions, then exits **1** when `package.json` declares the mirror but it is not installed — a broken install must not retire the repository's only "the day CC declares this name" alarm, and reporting it as a skip would leave "never evaluated" and "evaluated, no drift" indistinguishable to the runner — and exits 3 only when nothing declares the mirror at all, which is the one case where the direction genuinely does not apply. Either way a run that evaluated no witness is never counted as one that did. When the mirror *is* present its **installed version** is witnessed too (the two declared floors must agree with each other and the installed copy must meet them), since four preflight probes are satisfied by an arbitrarily stale mirror — they prove the extractor speaks, not that it is current. Every direction carries a positive control — known-present CC symbols, a comment-only sample proving the extractor distinguishes declaration from mention, and synthetic corpora fed through the **same** discriminator function the real verdict uses, so a verdict quietly rewritten to return nothing takes its own control down with it |
|
|
372
372
|
| `scripts/run-rules-side-test.mjs` | The persisted-permission-rules lane's shared decision half. The two capability bits are checked as **two independent gates** — a worker can honestly advertise the rules lane while predating the revoke routes, and that shape must *hide* the governance surface rather than render a dead entry. Failure classification is by **disposition, not cause**: the two 404s (route missing vs. dead ticket) never share a bucket, a 503 `rule_import_retry` means *the ticket is still alive* (the opposite handling of a dead one), and a stale-cursor 400 drops the cursor and re-lists from the top exactly once — never resuming a stale keyset, never surfacing a partial governance list, and never paging past the hard cap. The persist-ack reader is **merged into** `readToolApprovalRespondAck`: the three-state verdict (`persisted` / `refused` / `unknown`) is derived only from an ack that passed the package's structural narrowing, and a half-shaped object such as `{rulePersisted: true}` with no `delivery` reads as `unknown` — the pre-merge shell read would have said `persisted`, which is precisely the double-ledger drift this file closes, so that case is pinned in reverse. The local-allow-rule skeleton pins all five narrowings (whole-tool, tool-name match, literal anchor with the escaped-star counter-example, bare interpreter prefix consulted only for Bash, and the canonical dangerous-pattern overlay) **with their refusal strings byte-for-byte** — the cli's 128-assertion suite anchors the same strings, so a one-character edit here changes observable behaviour on three clients — and asserts the parse is a pure function of its input, because the same call backs both "render the option" and "resolve the selected value" `listAllPersistedRules` needs only `list` (`RulesListFacade`; the other four methods are known to the parameter type as optional members): a synthetic consumer compiled against the built declarations passes a two-method object, a list-only literal, a `Pick` slice, the named type, the full facade, and the full or two-method facade written as an inline object literal — including arrow functions with untyped parameters — all with zero diagnostics, while a misspelled method name in such a literal is still reported. |
|
|
373
373
|
| `scripts/run-park-decision-layer-test.mjs` | The decision layer behind the "stuck behind a card" family, shared by every client. A pending row that is **not in the queue** is three states, not one: a bounded, interruptible re-probe loop distinguishes *a decidable row*, *not born yet* (no positive evidence that anything settled — an empty queue proves nothing) and *settled elsewhere*, always probes at least once so a zero budget keeps the pre-fix semantics verbatim, cuts a hung read face off at the window rather than only noticing afterwards, and reports the honest failure when the window is spent instead of inventing a decision. The decision-note reader is likewise three-state: an explicit `noteRecorded: false` outranks an echoed note body, absence renders **no line at all**, and untrusted note text is flattened and bounded before it ever reaches a renderer. Row routing anchors on the deciding quantity — a row carrying `gateKind: "human"` with `toolName: "Write"` is a tool gate, because `human` is the engine's *generic* "someone must decide", not a synonym for a question — and the queue scan refuses to surface a row it cannot positively prove belongs to this session. A chain that fails after the row vanished is split by whether a card was ever presented: decided-elsewhere, or not-its-turn-yet. A row-level single-flight makes "at most one card per pending item" structural rather than incidental. The resume three-way card pins the option **order** (the zero-effect choice sits at index 0, because the frame carries no default-focus field and a stray Enter must not attach or cancel), renders only options the wired verbs can honour, collapses every ambiguous answer to zero action, omits the liveness line entirely when the engine gave no evidence, and — when there is no card lane at all — prints three real routes and exits on a dedicated code rather than reporting success |
|
|
374
|
-
| `scripts/run-selfheal-reopen-test.mjs` | The 409 active-run self-heal decision chain: `governanceForced` narrows on strict `true` only; triage prefers the wire's `pendingGate.kind` and falls back to the status table (an off-table kind is never guessed into a card arm — hands-off plus the honest wording); a first-sight card makes zero closed/reopened claims and a host presentation receipt of `presented: false` demotes the outcome to reopen-failed; park-row ownership is a fail-closed positive proof (own-run ledger or session id — unprovable is not owned); the three gate-identity key literals live in exactly one mint (`hitl/gateIdentity.ts`, AST string-token scan); the armed-gate presentation ledger is per-session; and the `plan_review` reopen arm shares the arm arm's card body, three-state verdict and delivery pipe, consuming the presentation history once a decision is delivered. The same chain also carries the `running` three-way card: both plan-family gate kinds route to the plan arm and all four ask-family kinds to the ask arm (an off-table kind still never gets guessed into either); the card is offered only for verbs that can actually be honoured and a missing presenter means zero action rather than a silent cancel; a steer is sent **exactly once** with its three delivery outcomes worded apart (a `queued` receipt is the wire correcting the triage input, so the named park word decides which card gets reopened, and an unrecognised park word drives neither arm), and a steer failure is split into *provably not delivered* (4xx) and *delivery unknown*, because telling a user to resend a non-idempotent instruction that may already have landed is how duplicates get made. After a user-chosen cancel, "the session is free" is asserted only from a whitelist of terminal states — park states hold the claim, an unrecognised state word is not a release, a failed read is *unknown* rather than a release, and only a 404 counts as one — and the honest timeout line quotes how long it really waited. The two "card could not be reopened" rows can carry a host-declared way to keep the conversation, which says the card comes back on resume only if it is still waiting: it is placed before the route that abandons it, never offered for an injected submission, while a decision is still on its way, or once the pending approval has been proven gone (the outcome then carries a flag saying so; the proof only counts before the cleanup card is shown, so a fallback after the card carries no flag unless a fresh read finds the run finished, and a recheck that finds the approval back clears it), the host function is not even called in those cases, and it is treated as unavailable when it throws or returns an empty value; a host can also switch off the engine decide route on the interactive rows while the cancel route stays, and with neither given all four rows are pinned byte-for-byte to the text the previous release produced. |
|
|
374
|
+
| `scripts/run-selfheal-reopen-test.mjs` | The 409 active-run self-heal decision chain: `governanceForced` narrows on strict `true` only; triage prefers the wire's `pendingGate.kind` and falls back to the status table (an off-table kind is never guessed into a card arm — hands-off plus the honest wording); a first-sight card makes zero closed/reopened claims and a host presentation receipt of `presented: false` demotes the outcome to reopen-failed; park-row ownership is a fail-closed positive proof (own-run ledger or session id — unprovable is not owned); the three gate-identity key literals live in exactly one mint (`hitl/gateIdentity.ts`, AST string-token scan); the armed-gate presentation ledger is per-session; and the `plan_review` reopen arm shares the arm arm's card body, three-state verdict and delivery pipe, consuming the presentation history once a decision is delivered. The same chain also carries the `running` three-way card: both plan-family gate kinds route to the plan arm and all four ask-family kinds to the ask arm (an off-table kind still never gets guessed into either); the card is offered only for verbs that can actually be honoured and a missing presenter means zero action rather than a silent cancel; a steer is sent **exactly once** with its three delivery outcomes worded apart (a `queued` receipt is the wire correcting the triage input, so the named park word decides which card gets reopened, and an unrecognised park word drives neither arm), and a steer failure is split into *provably not delivered* (4xx) and *delivery unknown*, because telling a user to resend a non-idempotent instruction that may already have landed is how duplicates get made. After a user-chosen cancel, "the session is free" is asserted only from a whitelist of terminal states — park states hold the claim, an unrecognised state word is not a release, a failed read is *unknown* rather than a release, and only a 404 counts as one — and the honest timeout line quotes how long it really waited. The two "card could not be reopened" rows can carry a host-declared way to keep the conversation, which says the card comes back on resume only if it is still waiting: it is placed before the route that abandons it, never offered for an injected submission, while a decision is still on its way, or once the pending approval has been proven gone (the outcome then carries a flag saying so; the proof only counts before the cleanup card is shown, so a fallback after the card carries no flag unless a fresh read finds the run finished, and a recheck that finds the approval back clears it), the host function is not even called in those cases, and it is treated as unavailable when it throws or returns an empty value; a host can also switch off the engine decide route on the interactive rows while the cancel route stays, and with neither given all four rows are pinned byte-for-byte to the text the previous release produced. When the plan reopen refuses because the user closed that card (`dismissedByUser`), the outcome carries the flag and one package-owned line says so — you closed it, a message brings it back — in the typed, the injected and the queued-steer forms, with no way-to-keep and no engine decide route; the reopen port receives `{ trigger }` mapped from the submission origin (typed ⇒ `user`, injected ⇒ `automatic`, absent ⇒ the old single-argument call). |
|
|
375
375
|
| `scripts/run-terminal-identity-copy-test.mjs` | Terminal-state **identity**, in both lanes where a stop gets a name. A run stopped by this deployment's own governance knobs — the open-set `limits.*` family, `output.invalid`, and the `blocked` contract terminal a ReportBlocked agent produces — is not a provider failure, and labelling it `API Error:` sends the reader to check the network, the key and the quota when the handle is the `--max-turns` they passed themselves. Those terminals now render a neutral row; the reverse direction is guarded just as hard, because asserting "this is *not* an API error" on a code the package does not recognise is the same misfiling pointed the other way — a real `gateway HTTP 502`, a `conflict.session_active_run` and any unknown code all keep the `API Error:` prefix, and the row keeps its `isApiErrorMessage` class flag so brief-mode visibility filtering does not silently drop it. The second half is who the rejected submission belonged to: the self-heal copy told every caller "Your message was NOT sent … send it again", which is three separate untruths for a system injection (a plan-review outcome, a cron wake-up, a task notification) — not the user's message, and not re-sendable, since a host queue marks those non-editable and non-recallable. The injected form says so instead, and the one sentence that promises re-delivery is pinned to the single disposition that earns it: `selfHealSubmissionDisposition` is the same function the host consults before putting the item back on its queue, so the promise and the behaviour cannot drift apart, and the arms where no card could be surfaced state plainly that nothing was delivered and nothing will retry. Since 0.72.6 the same gate pins the **follow intent** after a steer (): a message handed to a live run only pays off if someone tails that run's own event stream, so `steerFollowIntent` decides from the delivery word whether to tail now, after the pending decision, or only after a wake — and the "watch that run" sentence ("watch that reply" on the rows about follow-up messages sema sent on its own) turns into a factual "sema is following that run" ("… that reply") **only** when the host declares it attached that tail, so a shell that did not wire it can never claim it did. Since 0.83.0 the rows about follow-up messages sema sent on its own carry no engine-internal words and say what happened per delivery shape, promising a resend or "nothing for you to do" only where the code guarantees it |
|
|
376
376
|
| `scripts/run-additive-key-passthrough-test.mjs` | The one disease shape behind two legs: a **closed whitelist / flattening arm** dropping a fact that is already on the wire, while both sides of the seam look correct. (1) The `task_progress` projection carries a registered **key ledger** — a frame populated with every key the service really projects is pushed through the shipped `eventToSdkMessage`, and the set of wire keys that survive must equal the registered pass-through list **name for name in both directions**, so quietly forwarding one more key is as red as quietly dropping one. `model` (the child run's model id, minted by core as `prepared.model.id` and projected by the server since 7.52.1) is the key this batch adds, with the same conditional the server itself applies: a non-empty string or no key at all — an empty string is neither a model id nor "unknown". The ledger is also checked against the fenced list in `docs/INTEGRATION-CLIENTS.md` §3d, so a doc that still says seven keys while the code forwards eight is red rather than merely stale. (2) The decide-failure arms carry the server's S-02 `currentPending` pointer key from a 409 `approval_stale` refusal onto the outcome the host reads. The reader is structural rather than `instanceof`, because the client is host-injected and the class identity is not this package's to assume; a half triple never mints (half a pointer cannot relocate anything), an empty string is not presence, and `checkpointToken` never transits. Both the allow and the deny leg are driven end to end through the real durable approval path — as is the accept-session leg, where a refusal carrying the pointer key must now re-raise instead of silently re-sending the human's answer for the **old** card as a plain approve (one decide call, pointer preserved), while a legacy 400 still falls back exactly as before — and all three flattening points must call the one shared reader — the same-shape residue check that makes "fixed one arm and left the twin" red instead of invisible. (3) The same disease growing on the REQUEST side: the `.mcp.json` → server-spec projection rebuilds each server key by key, and the settings schema deliberately leaves some keys parse-transparent — whatever JSON the file carries reaches the engine untouched, because validating them where the whole domain parses all-or-nothing would let one bad declaration take every server down silently. The whitelist had no row for the newest of them, so an operator's per-tool declarations — the ones the write fence reads — were stripped at the package boundary while both sides looked correct. The criterion is not "is that key handled" but the transparent-key table read out of the INSTALLED schema at runtime, reconciled name-for-name against this leg's ledger, so the day upstream adds a third one this turns red and forces an explicit decision. Behaviour is pinned on both transports, by object identity rather than deep equality (a rebuild would be a second judge), and malformed values must transit UNCHANGED rather than be refused here — the engine refuses them loudly and names the server, whereas a package-side judge can only swallow a declared protection quietly. Absence still mints no key, unknown keys still never reach the wire (the fix is the dropped key, not the gate), and the one transparent key this leg deliberately does not forward is a ledger entry with its own exit condition: it belongs to the deployment plane, and the day the request-plane type declares it the entry's premise is gone and the gate says so |
|
|
377
377
|
| `scripts/run-esc-halt-plan-test.mjs` | The Esc stop decision every client shares: fire the **turn-level** halt first, and escalate to a **run-level** cancel in exactly two cases — the engine itself answered with a 409 from the closed code set (it is saying "there is no in-flight turn here; use cancel for a run-level stop"), or that shot came back with no verdict at all *and* the shell can independently prove a permission card was on screen. Everything else does not escalate. The asymmetry is the whole point and every negative control guards the same direction — deciding *not* to escalate costs the user one more choice on a busy-session card (recoverable), deciding to escalate wrongly tears down a run that was alive and takes every in-flight tool with it (not). So: the closed code set is a **frozen** value, not a `ReadonlySet` — type-level immutability does not stop a consumer's `.add()`, and the guard proves it by really trying to mutate the exported value and then checking the verdict did not drift; the escalation gate is the **conjunction** of that closed set and the 409 status, since honouring the code alone lets a 500 that merely quotes it drive a destructive call; `interrupt.not_held` and `steering.not_running` are deliberately outside the set (the first means *this replica* has no live face — the run may be perfectly alive on another); an unreadable code falls to the no-escalation side; a `parked` flag never overrides a verdict the engine did give, and only strict `true` counts when it did not. The first shot is unconditional by construction — it does not consult `parked`, because the 409 it earns is exactly the verdict the gate wants — and the verdict itself is a closed machine-readable reason word, not display copy. A third escalating case was added once tearing the stream stopped reaping the run: with detach armed, a shot that never lands leaves the run going all the way to the end of the turn, so the Esc the user pressed has no effect at all and nothing on screen says so — the old behaviour had a silent backstop (tearing the stream ended the run) and that backstop is gone. The new fact is held to the same three disciplines as `parked`: it is read only where the engine gave no verdict, it is judged **after** `parked` so an existing host's reason word does not change under it, and only strict `true` counts. Absence is proven to be a no-op rather than asserted — the guard carries its own reference implementation of the previous version's table, runs the full grid through both, requires zero divergence when the new field is omitted, and first shows the comparison really does report a difference on the one cell where the two versions are meant to differ |
|
|
@@ -431,6 +431,7 @@ guard still cross-checks the table by name).
|
|
|
431
431
|
| `scripts/run-display-untrusted-projection-test.mjs` | The single display-safety outlet (`displayUntrusted`) and the credential wash on the end-of-run rows this package mints. The outlet composes two credential nets (URL structure: userinfo, every query value, the fragment, path parameters and path segments that start with a known secret prefix; key/value words such as `Authorization: Bearer ...`, `Authorization: token ...` or `api_key=...`, plus well-known secret literals that appear without a label, such as `sk-...`, `ghp_...`, `AKIA...`, JWTs and the body of a PEM private key) with three character nets (control characters, bidirectional and format characters, whitespace folding). The credential nets match on a view of the text with ANSI sequences, format characters, control characters and the outlet's own escape tokens stripped, and map the result back onto the original, so colouring or an invisible character wedged between a label, its separator and its value cannot hide the value, and no stray marker is left behind. Whitespace of any length around the separator is accepted. Hosts, ports, paths, query key names and surrounding prose stay byte-for-byte, clean text comes back unchanged, the result is idempotent (also with a length cap), a length cap never splits an escape token or a surrogate pair, an invalid cap means no cap, and every net can be switched off on its own. A few narrow shapes are left alone because they name something rather than carry a value (a plain English word after `bearer` or `basic`, a back-quoted credential variable name, a plain integer after `tokens:`, a list of key names after `keys:`), each with a counter-example that is still washed. Regional flag emoji built from tag characters are kept whole. The existing single-line helpers (`escapeDisplayControlChars`, `collapseLabel`, `capForDisplay`, peer sender names and the hook failure banner) now run on the same engine and are held byte-identical to their previous output over every BMP code unit plus random strings. The approval decision-note echo, the subagent resume receipt (and its failure debug line) and the startup list of plugin hooks that will not run now also drop bidirectional and format characters (and, for the receipt, C1 controls); a note that is empty after cleaning is treated as absent. The synthetic end-of-run rows (`API Error:`, `Run stopped:`, `Model output error:`, `Outcome unknown:`) and the result frame's `errors[]` pass both credential nets before they leave the package, on the print lane and on the interactive lane (which also keeps the row-class flag); this covers a blocked reason whoever wrote it, while assistant text rows, a successful `result` and salvaged output are never touched, and a non-string `errors[]` entry is passed through unchanged. The known-secret-prefix check is a local copy of the configuration package's detector and is compared with the installed one entry by entry. |
|
|
432
432
|
| `scripts/run-ask-survives-posture-test.mjs` | The single posture predicate `askSurvivesPosture(card, facts)` for sessions whose standing mode would otherwise answer approval cards on the user's behalf (bypass-style modes). It reads two facts and returns one of three verdicts. The first is the ask origin stamped on the card: the question tool (`content_question`), an organization rule (`org_rule`), a hook (`hook`), an explicit ask rule (`ask_rule`), an organization policy or rule store that could not be read (`org_unavailable`, `rule_store_unavailable`) and the classifier's hand-off after its denial limit (`denial_limit_fallback`) must still be asked (the engine requires a real person to answer all three) and every other origin this build knows is left to the posture only once the host has also reported that its own ask rules did not match. The second is the host's own reading of its settings ask rules for this call: a positive match must be asked, and a command the host could not fully parse counts as no match. When the host reported no reading, every card outside those seven origins gets `unknown`, because an origin says who asked and not that the user's own ask rules did not match; an origin this build does not know gets `unknown` even after a reported non-match. `unknown` is never an approval: the host falls back to its own settings rules. The guard checks the verdict for every origin word, both with no host reading and with a reported non-match, against an independent table whose word set must equal the package's origin list, so a new upstream word fails the guard until it is classified; it covers the combinations of both facts, malformed inputs (non-boolean readings, empty or non-string origins, prototype keys, a different letter case), the fact that the predicate does not read the stronger bits on the card (those stay with the host's earlier checks), real card requests produced by the live-frame, parked-row and suspended-ask paths, and a closed, frozen verdict shape. |
|
|
433
433
|
| `scripts/run-engine-agent-absence-projection-test.mjs` | Absent background agents: when the engine stops reporting a background agent and no final state has arrived, the row is marked absent and this package owns every decision about it, so all clients agree. One predicate says whether a row is absent (the mark, not the status, decides). An absent row keeps its last known status, never counts as running, and is never counted as completed, failed or stopped; its elapsed time stops at the last moment it was seen, and its sentence says it may still be running. The end-of-turn sweep never settles an absent row (or a resident one). A row that comes back, or a real final state for the current cycle, clears the mark; a late final state from an earlier cycle does not. Absent rows are never removed at the short grace window. After the hard limit (30 minutes from the last time they were seen) the host is asked for the background-agent registry reading of each row: only a reading that the agent has ended or is not listed lets the row go, and each removal is returned as a fact the host must act on and announce; a reading of running, unknown, missing or unrecognised keeps the row and schedules nothing, so no standing poll is created. Until a registry reading is available every absent row stays. A row someone is viewing is held and reported separately only once the registry confirms it is gone. The row sentence, the removal sentence and the late-result sentence come from one place, never state an outcome or that the agent finished, and escape control characters in names, in the engine's removal word and in the late-result status. An end-to-end cell drives the real fleet projection and the real absence channel through every decision. |
|
|
434
|
+
| `scripts/run-plan-review-dismissal-test.mjs` | An automatic reopen does not put back a plan-review card the user closed (the first-presentation path neither checks nor clears that record, so a replayed park frame still presents its card). A plan-review card the user dismissed (Esc, abort, or any answer that is not approve or reject) is recorded per session and run at the moment of dismissal, synchronously, before anything queued behind the card can be released; a reopen marked `trigger: 'automatic'` then refuses with `{ reopened: false, dismissedByUser: true }` instead of minting a new card the user's next keystroke would land on, while the user's own next action (`trigger: 'user'`) reopens it and clears the record. The record is keyed by gate instance when the host supplies an instance reader: a new plan gate on the same run is still surfaced, and anything that cannot prove the gate is new (no reader, a failed, empty, thrown or timed-out read) refuses on the conservative side. The asynchronous form re-checks after its reads and before minting — a decision handed over meanwhile (seen by the package, or reported by the host's optional hand-over predicate) answers as "your answer is on its way"; a record that changed meanwhile makes the stale evaluation mint nothing and answer from the current record: another close refuses as the user's close and keeps the newer record, a card already back on screen (the user's own action or a concurrent automatic reopen, waited for within the receipt window and re-read once the wait is over) answers `reopened: true`, and a record that is gone (session change, ledger overflow) answers a plain refusal without `dismissedByUser`; a hand-over predicate that throws refuses with a plain `{ reopened: false }`. Each run has at most one reopen on its way: a reopen that arrives while an earlier one's card is published but not yet settled joins it instead of minting a second card and retiring the first card's answer path. Instance readers are snapshotted when they resolve, so a host that hands over its own live set still gets a new gate recognised; and a decisive-looking host answer note does not clear the record for the very card the package's own responder already judged non-decisive (the label was not on that card). A successful reopen replaces only the record taken before the card was minted (with an on-screen marker, not a deletion), a decisive answer clears it, the per-session ledger is bounded, and a session change clears its bucket. Without `trigger` the reopen answers exactly as before, apart from joining a reopen already on its way. |
|
|
434
435
|
|
|
435
436
|
Each suite carries a floor that only moves up — a refactor that stops executing a group of
|
|
436
437
|
assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { ActiveRunBusySignal } from './runStream.js';
|
|
2
|
+
import type { PlanReviewReopenTrigger } from '../hitl/planReviewWire.js';
|
|
2
3
|
export declare const PLAN_REVIEW_GATE_KIND = "plan_review";
|
|
3
4
|
export declare const PLAN_REVIEW_GATE_KINDS: readonly string[];
|
|
4
5
|
export declare const ASK_PARK_GATE_KINDS: readonly string[];
|
|
@@ -33,6 +34,7 @@ export type ReopenCardVerdict = {
|
|
|
33
34
|
decidedWithoutCard?: true;
|
|
34
35
|
pendingRowGone?: true;
|
|
35
36
|
_sema_decisionInFlight?: true;
|
|
37
|
+
dismissedByUser?: true;
|
|
36
38
|
} | {
|
|
37
39
|
reopened: true;
|
|
38
40
|
firstSight: boolean;
|
|
@@ -54,7 +56,9 @@ export interface ActiveRunSelfHealDeps {
|
|
|
54
56
|
listOwnedPendingApprovals?: (opts?: {
|
|
55
57
|
signal?: AbortSignal;
|
|
56
58
|
}) => Promise<number>;
|
|
57
|
-
reopenPlanReview?: (taskId: string
|
|
59
|
+
reopenPlanReview?: (taskId: string, opts?: {
|
|
60
|
+
trigger?: PlanReviewReopenTrigger;
|
|
61
|
+
}) => ReopenCardVerdict | Promise<ReopenCardVerdict>;
|
|
58
62
|
reopenAskPark?: (taskId: string) => Promise<ReopenCardVerdict>;
|
|
59
63
|
offerRunningChoice?: (req: RunningChoiceRequest) => Promise<'steer' | 'cancel' | 'wait'>;
|
|
60
64
|
offerStaleParkChoice?: (req: StaleParkChoiceRequest) => Promise<'cancel' | 'wait'>;
|
|
@@ -81,6 +85,7 @@ export type SelfHealOutcome = {
|
|
|
81
85
|
decidePath: string | null;
|
|
82
86
|
decisionInFlight?: true;
|
|
83
87
|
pendingRowGone?: true;
|
|
88
|
+
dismissedByUser?: true;
|
|
84
89
|
} | {
|
|
85
90
|
kind: 'not-parked';
|
|
86
91
|
taskId: string;
|
|
@@ -82,6 +82,16 @@ export function readCancelRequested(body) {
|
|
|
82
82
|
function reopenRefusedForDecisionInFlight(verdict) {
|
|
83
83
|
return verdict !== null && verdict.reopened === false && verdict._sema_decisionInFlight === true;
|
|
84
84
|
}
|
|
85
|
+
function reopenRefusedForUserDismissal(verdict) {
|
|
86
|
+
return verdict !== null && verdict.reopened === false && verdict.dismissedByUser === true;
|
|
87
|
+
}
|
|
88
|
+
function planReviewReopenTriggerFor(origin) {
|
|
89
|
+
if (origin === 'user')
|
|
90
|
+
return 'user';
|
|
91
|
+
if (origin === 'injected')
|
|
92
|
+
return 'automatic';
|
|
93
|
+
return undefined;
|
|
94
|
+
}
|
|
85
95
|
function reopenDelivered(verdict) {
|
|
86
96
|
return verdict.reopened === true && verdict.presented !== false;
|
|
87
97
|
}
|
|
@@ -287,7 +297,9 @@ export async function attemptActiveRunSelfHeal(signal, runs, deps) {
|
|
|
287
297
|
}
|
|
288
298
|
async function planVerdict(taskId, deps) {
|
|
289
299
|
try {
|
|
290
|
-
|
|
300
|
+
const trigger = planReviewReopenTriggerFor(deps?.submissionOrigin);
|
|
301
|
+
const verdict = trigger === undefined ? deps?.reopenPlanReview?.(taskId) : deps?.reopenPlanReview?.(taskId, { trigger });
|
|
302
|
+
return (await verdict) ?? { reopened: false };
|
|
291
303
|
}
|
|
292
304
|
catch {
|
|
293
305
|
return { reopened: false };
|
|
@@ -311,6 +323,7 @@ async function planReviewArm(taskId, signal, deps) {
|
|
|
311
323
|
decidePath: signal.pendingGate?.decidePath ?? null,
|
|
312
324
|
...(reopenRefusedForDecisionInFlight(verdict) ? { decisionInFlight: true } : {}),
|
|
313
325
|
...(verdict.pendingRowGone === true ? { pendingRowGone: true } : {}),
|
|
326
|
+
...(reopenRefusedForUserDismissal(verdict) ? { dismissedByUser: true } : {}),
|
|
314
327
|
};
|
|
315
328
|
}
|
|
316
329
|
async function askParkArm(taskId, signal, runs, deps) {
|
|
@@ -617,7 +630,8 @@ export function activeRunSelfHealRow(outcome, signal, copy, origin, follow) {
|
|
|
617
630
|
const engineDecide = copy?.engineDecidePath !== false;
|
|
618
631
|
const keep = (outcome.kind === 'plan-review-reopen-failed' || outcome.kind === 'ask-reopen-failed') &&
|
|
619
632
|
outcome.decisionInFlight !== true &&
|
|
620
|
-
outcome.pendingRowGone !== true
|
|
633
|
+
outcome.pendingRowGone !== true &&
|
|
634
|
+
(outcome.kind !== 'plan-review-reopen-failed' || outcome.dismissedByUser !== true)
|
|
621
635
|
? keepSessionWayOutOf(copy)
|
|
622
636
|
: null;
|
|
623
637
|
const base = activeRunSelfHealBaseRow(outcome, signal, copy?.wayOut ?? DEFAULT_WAY_OUT, following, keep, engineDecide);
|
|
@@ -632,6 +646,7 @@ export function activeRunSelfHealRow(outcome, signal, copy, origin, follow) {
|
|
|
632
646
|
}
|
|
633
647
|
}
|
|
634
648
|
const INJECTED_LEAD = 'A follow-up message sema sent on its own (not one you typed)';
|
|
649
|
+
const USER_CLOSED_CARD_NOT_PUT_BACK = 'sema does not put a card you closed back on screen by itself';
|
|
635
650
|
function injectedSubmissionRow(outcome, following = false) {
|
|
636
651
|
const lead = INJECTED_LEAD;
|
|
637
652
|
const id = 'taskId' in outcome && outcome.taskId ? ` (id ${outcome.taskId})` : '';
|
|
@@ -647,6 +662,11 @@ function injectedSubmissionRow(outcome, following = false) {
|
|
|
647
662
|
return (`${lead} was not delivered: this session is still busy with an earlier reply${id}, and your answer to the decision ` +
|
|
648
663
|
`it is waiting on is still on its way. sema did not retry, so the model has not seen it.`);
|
|
649
664
|
}
|
|
665
|
+
if (outcome.kind === 'plan-review-reopen-failed' && outcome.dismissedByUser === true) {
|
|
666
|
+
return (`${lead} was not delivered: this session is held by an earlier reply${id} whose plan is waiting for your review, and ` +
|
|
667
|
+
`you closed that approval card — ${USER_CLOSED_CARD_NOT_PUT_BACK}. Send a message when you want the card back; ` +
|
|
668
|
+
`sema did not retry, so the model has not seen it.`);
|
|
669
|
+
}
|
|
650
670
|
if (outcome.kind === 'running-steered' && outcome.delivery === 'queued') {
|
|
651
671
|
const queuedOn = `${lead} is queued on an earlier reply${id} that is paused waiting for a decision`;
|
|
652
672
|
if (outcome.reopened !== null && reopenDelivered(outcome.reopened)) {
|
|
@@ -655,6 +675,10 @@ function injectedSubmissionRow(outcome, following = false) {
|
|
|
655
675
|
if (reopenRefusedForDecisionInFlight(outcome.reopened)) {
|
|
656
676
|
return `${queuedOn}; your answer to it is still on its way, and the message is picked up once that answer is applied.`;
|
|
657
677
|
}
|
|
678
|
+
if (reopenRefusedForUserDismissal(outcome.reopened)) {
|
|
679
|
+
return (`${queuedOn}; you closed that card yourself, so sema did not put it back on screen — send a message to bring it ` +
|
|
680
|
+
`back, and the message is picked up once that decision is made.`);
|
|
681
|
+
}
|
|
658
682
|
return `${queuedOn} sema could not show here; the message is picked up only once that decision is made.`;
|
|
659
683
|
}
|
|
660
684
|
if (outcome.kind === 'running-steered' && outcome.delivery === 'parked_for_wake') {
|
|
@@ -747,6 +771,11 @@ function activeRunSelfHealBaseRow(outcome, signal, wayOut, following = false, ke
|
|
|
747
771
|
case 'plan-review-reopen-failed': {
|
|
748
772
|
if (outcome.decisionInFlight === true)
|
|
749
773
|
return decisionInFlightRow(`on a plan review (run ${outcome.taskId})`);
|
|
774
|
+
if (outcome.dismissedByUser === true) {
|
|
775
|
+
return (`The previous turn is parked waiting for a plan review (run ${outcome.taskId}), and you closed its approval card — ` +
|
|
776
|
+
`${USER_CLOSED_CARD_NOT_PUT_BACK}. It did NOT cancel the run — that would have discarded the plan for you. ` +
|
|
777
|
+
`Your message was NOT sent; send it again to bring the card back.`);
|
|
778
|
+
}
|
|
750
779
|
const viaEngine = engineDecide && outcome.decidePath
|
|
751
780
|
? ` You can also decide it on the engine directly: POST ${outcome.decidePath}.`
|
|
752
781
|
: '';
|
|
@@ -774,8 +803,11 @@ function activeRunSelfHealBaseRow(outcome, signal, wayOut, following = false, ke
|
|
|
774
803
|
: reopenRefusedForDecisionInFlight(outcome.reopened)
|
|
775
804
|
? `Your answer to that decision is still on its way to the engine, so sema did not show the card again; ` +
|
|
776
805
|
`that run moves on once the engine applies it.`
|
|
777
|
-
:
|
|
778
|
-
|
|
806
|
+
: reopenRefusedForUserDismissal(outcome.reopened)
|
|
807
|
+
? `You closed that decision card yourself, so sema did not put it back on screen; nothing will resume that ` +
|
|
808
|
+
`run until that decision is made — send a message to bring the card back.`
|
|
809
|
+
: `sema could not surface that decision card here, so nothing will resume that run until that decision is made; ` +
|
|
810
|
+
`${wayOut} if you no longer want it.`;
|
|
779
811
|
return (`${handle} is not actually working right now: the engine reports it ${parked} on a decision it needs ` +
|
|
780
812
|
`from you, so it queued your message on that park instead of running it. ${next} ${textOnly} ` +
|
|
781
813
|
`Nothing was cancelled, and your message did NOT start a new turn.`);
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { createSessionSlot, DEFAULT_SESSION_KEY } from '../sessionSlot.js';
|
|
2
|
-
import { armedKeyFromQuestionId, PLAN_REVIEW_QUESTION_ID_PREFIX, planReviewQuestionId, REOPEN_ID_TAIL } from './gateIdentity.js';
|
|
2
|
+
import { armedKeyFromQuestionId, PLAN_REVIEW_QUESTION_ID_PREFIX, planReviewQuestionId, REOPEN_ID_TAIL, taskIdFromPlanReviewQuestionId, } from './gateIdentity.js';
|
|
3
3
|
const armedGatesByKey = createSessionSlot();
|
|
4
4
|
const armedListenersByKey = createSessionSlot();
|
|
5
5
|
const planReviewGenByKey = createSessionSlot();
|
|
@@ -149,10 +149,8 @@ export function notePlanReviewAnsweredFor(sessionKey, questionId) {
|
|
|
149
149
|
if (noted.has(questionId))
|
|
150
150
|
return;
|
|
151
151
|
noted.add(questionId);
|
|
152
|
-
const
|
|
153
|
-
|
|
154
|
-
const taskId = cut >= 0 ? base.slice(0, cut) : base;
|
|
155
|
-
if (taskId.length === 0)
|
|
152
|
+
const taskId = taskIdFromPlanReviewQuestionId(questionId);
|
|
153
|
+
if (taskId === undefined)
|
|
156
154
|
return;
|
|
157
155
|
clearArmedGateFor(sessionKey, planReviewArmedKeyFor(sessionKey, taskId));
|
|
158
156
|
const gens = genMapFor(sessionKey);
|
|
@@ -4,4 +4,5 @@ export declare function approvalCallKey(gatedCallId: string | undefined, taskId:
|
|
|
4
4
|
export declare function askGateQuestionId(callKey: string): string;
|
|
5
5
|
export declare function planReviewQuestionId(taskId: string): string;
|
|
6
6
|
export declare function liveFrameCallKey(approvalId: string): string;
|
|
7
|
+
export declare function taskIdFromPlanReviewQuestionId(questionId: string): string | undefined;
|
|
7
8
|
export declare function armedKeyFromQuestionId(questionId: string): string;
|
|
@@ -14,6 +14,14 @@ export function planReviewQuestionId(taskId) {
|
|
|
14
14
|
export function liveFrameCallKey(approvalId) {
|
|
15
15
|
return `${HITL_FRAME_CALL_KEY_PREFIX}${approvalId}`;
|
|
16
16
|
}
|
|
17
|
+
export function taskIdFromPlanReviewQuestionId(questionId) {
|
|
18
|
+
if (typeof questionId !== 'string' || !questionId.startsWith(PLAN_REVIEW_QUESTION_ID_PREFIX))
|
|
19
|
+
return undefined;
|
|
20
|
+
const base = questionId.slice(PLAN_REVIEW_QUESTION_ID_PREFIX.length);
|
|
21
|
+
const cut = base.indexOf(REOPEN_ID_TAIL);
|
|
22
|
+
const taskId = cut >= 0 ? base.slice(0, cut) : base;
|
|
23
|
+
return taskId.length > 0 ? taskId : undefined;
|
|
24
|
+
}
|
|
17
25
|
export function armedKeyFromQuestionId(questionId) {
|
|
18
26
|
const base = questionId.startsWith(HITL_ASK_QUESTION_ID_PREFIX)
|
|
19
27
|
? questionId.slice(HITL_ASK_QUESTION_ID_PREFIX.length)
|
|
@@ -25,8 +25,12 @@ export declare function isPlanReviewPark(result: unknown): result is {
|
|
|
25
25
|
sessionId?: string;
|
|
26
26
|
};
|
|
27
27
|
export declare function _resetArmedPlanReviewsForTest(): void;
|
|
28
|
+
export type PlanReviewReopenTrigger = 'user' | 'automatic';
|
|
29
|
+
export type PlanReviewGateInstanceReader = (taskId: string) => Promise<ReadonlySet<string> | null>;
|
|
30
|
+
export declare function clearPlanReviewUserDismissals(sessionKey?: string): void;
|
|
28
31
|
export declare function armPlanReviewApproval(result: unknown, sessionKey?: string, opts?: {
|
|
29
32
|
submittedInPlanMode?: boolean;
|
|
33
|
+
readGateInstanceKeys?: PlanReviewGateInstanceReader;
|
|
30
34
|
}): boolean;
|
|
31
35
|
export type PlanReviewDecisionEffect = 'took_effect' | 'advanced' | 'still_parked' | 'not_sent' | 'not_applied' | 'unconfirmed';
|
|
32
36
|
export interface PlanReviewOutcomeMeta {
|
|
@@ -43,6 +47,9 @@ export interface ReopenPlanReviewOpts {
|
|
|
43
47
|
deliverDecision?: (taskId: string, decision: 'approve' | 'reject') => void | Promise<void>;
|
|
44
48
|
sessionKey?: string;
|
|
45
49
|
presentationReceiptMs?: number;
|
|
50
|
+
trigger?: PlanReviewReopenTrigger;
|
|
51
|
+
readGateInstanceKeys?: PlanReviewGateInstanceReader;
|
|
52
|
+
decisionHandedOver?: (taskId: string) => boolean;
|
|
46
53
|
}
|
|
47
54
|
export declare function _resetActiveReopenRespondersForTest(): void;
|
|
48
55
|
export declare function reopenPlanReviewCard(taskId: string, opts: ReopenPlanReviewOpts & {
|