@sema-agent/client-core 0.67.1 → 0.68.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +211 -0
  2. package/README.md +67 -3
  3. package/dist/adapt/arms.js +27 -2
  4. package/dist/adapt/turnFlags.d.ts +14 -0
  5. package/dist/adapt/turnFlags.js +4 -1
  6. package/dist/adapter/activeRunSelfHeal.d.ts +53 -6
  7. package/dist/adapter/activeRunSelfHeal.js +79 -8
  8. package/dist/adapter/downstream/eventToSdkMessage.d.ts +18 -1
  9. package/dist/adapter/downstream/eventToSdkMessage.js +33 -2
  10. package/dist/adapter/downstream/terminalToSdkResult.d.ts +75 -6
  11. package/dist/adapter/downstream/terminalToSdkResult.js +144 -44
  12. package/dist/adapter/runStream.d.ts +22 -2
  13. package/dist/adapter/runStream.js +190 -52
  14. package/dist/adapter/types.d.ts +4 -28
  15. package/dist/autoModeUnavailable.d.ts +17 -9
  16. package/dist/autoModeUnavailable.js +26 -8
  17. package/dist/classifierStatus.d.ts +32 -4
  18. package/dist/classifierStatus.js +5 -3
  19. package/dist/engineErrorCodes.d.ts +52 -0
  20. package/dist/engineErrorCodes.js +117 -0
  21. package/dist/engineNoticeCodes.d.ts +95 -1
  22. package/dist/engineNoticeCodes.js +124 -1
  23. package/dist/gateVocabulary.d.ts +18 -7
  24. package/dist/gateVocabulary.js +21 -8
  25. package/dist/hitl/parkResolver.d.ts +0 -14
  26. package/dist/hitl/parkResolver.js +22 -9
  27. package/dist/hitl/toolApprovalWire.d.ts +2 -1
  28. package/dist/hitl/toolApprovalWire.js +1 -0
  29. package/dist/ownKey.d.ts +34 -0
  30. package/dist/ownKey.js +36 -0
  31. package/dist/retryStatus.d.ts +13 -2
  32. package/dist/retryStatus.js +4 -1
  33. package/dist/runTerminal.d.ts +87 -14
  34. package/dist/runTerminal.js +89 -15
  35. package/dist/toolResult.js +8 -0
  36. package/dist/workflowClient.d.ts +22 -0
  37. package/dist/workflowClient.js +37 -0
  38. package/docs/INTEGRATION-CLIENTS.md +469 -10
  39. package/package.json +2 -2
package/CHANGELOG.md CHANGED
@@ -49,6 +49,217 @@
49
49
  > 挡住 ⇒ 本批把它机械化——④a0 对 `pending` 行**要求段头已是日期形**(`(未发布)` 直接红),阶段一
50
50
  > commit 漏转在发布前就红,不再靠人记。
51
51
 
52
+ ## 0.68.0(2026-09-13)
53
+
54
+ 🔴 **BREAKING 提货批**(core **7.16.0 → 7.17.1**;三件破坏性 + 一次**引擎地板抬升**到 core ≥7.17.0)。
55
+ 逐件的铸点读法 / 缺席语义 / 三端换装清单 / 黑盒判据 G33-01…21(含 05b–05f 六格)/ 逐键处置表见
56
+ `docs/INTEGRATION-CLIENTS.md` **§33**(母本 §33z)。
57
+
58
+ 🔴 **本批最该记住的一句话**:`#711` 把「这一轮没量出账」的 wire 形从**不发 usage** 换成了
59
+ **六个 0 + 判别位**。同一个真实情形换了字节之后,本包里每一处「读 usage 的数字」都从「读到缺席」
60
+ 变成了「读到一个精确的零」—— 异源对抗复审**连跑六轮**,一共在**五处**读点 + 一条**跨轮边界**上
61
+ 抓到同一个病形(逐处逐条记在 §33b 的表里)。**占位零的判据不是「有没有 usageMissing」,是
62
+ 「这一格是不是零」**:`usageMissing` 可以与真数字同帧,而占位恒为全零 ⇒ 非零必是真读数。
63
+ 另一句同样承重:**「这一轮结束了」与「这一轮花了多少」是两件事** —— 合成一个动作,
64
+ 过滤掉数字就会顺手把模型轮边界也过滤掉。
65
+
66
+ ### 🔴 BREAKING(三件,逐件说清「为什么不留兼容尾门」)
67
+
68
+ - **B1 `turn_end.usage` 恒在场 ⇒ 「usage 缺席」臂整条退役**(core 7.17.0 #711)。上游的铸点自 7.17.0
69
+ 起是无条件的(`turnUsage ?? {六个 0}` + `usageMissing:true`),**「没有 usage 的 `turn_end`」不再是
70
+ 一条合法形** ⇒ 0.65.1 / B-088 那条「三者任一在场即发」的臂、以及它逼出来的三处
71
+ `usage !== undefined ? … : …` 与两处 `usage?.x ?? 0` **全部删掉**。缺席改按**契约违约**收口:
72
+ ①走宿主丢帧留痕口(判词 `turn_end_usage_absent`,开集 ⇒ 宿主不必改型),且**那一行说的话**与别的
73
+ 丢帧不同(缺省句「this build's projector has no arm for it」对本形是假话);②**整帧不投影**
74
+ (没有 usage 就没有数字可交,折 0 就是把「不知道」写成已知账);③**同时立下界位**(不立的话,
75
+ 一条真丢了账的 run 会在终帧上被渲成一笔精确的账)。
76
+ 🔴 **不留按引擎版本分岔的兼容读** —— 那要求包边界持有一个它读不出来的量(引擎版本不在帧上),
77
+ 猜版本比响亮地说「这帧不合契约」更坏。⇒ 引擎地板抬到 **core ≥7.17.0**(如实登记:更老的引擎上
78
+ 真实的「该轮零 usage」帧会被判成违约,账仍诚实但少那一拍的实时增量)。
79
+ 同批**族扫产物**(异源对抗复审轮一订正过一次,订正本身值得记):`usageMissing` 那一轮的
80
+ **占位零**不许当读数。⚠️ 关键一条 —— `usageMissing:true` **不保证**六格是零(core 在同一轮里
81
+ 可能已攒到真数字而另一次调用报了缺账,判别位与真数字同帧并存)⇒ 判据只能锚在**能证明是真读数
82
+ 的那一半**:占位恒为 `0`,所以**非零有限数必是真读数**(照铸/照累加),而 `0` 分不出占位与真零
83
+ (不铸)。落在 `turn_usage.outputTokens` 与子代分表行的 `cacheReadTokens` 两处。
84
+ (轮一的写法是「缺账轮整条跳过」——那会把真读数丢掉:两帧 cache 10 / 100 且第二帧带判别位时
85
+ 合计应为 110,整条跳过给出 10,而实时增量腿仍交 10 与 100,实时面与终局分表当场对不上。)
86
+ 同族第三处:**`RunStreamHandle` 这个公开出口**此前无条件覆盖两份读数且**没有任何判别位** ⇒
87
+ 只吃它的 footer 会把「不知道」渲成一笔精确的零账。修形 = **未测量的轮不覆盖**(保留上一次真读数,
88
+ 与 #711 之前逐字相同 ⇒ 没跟车的消费者零回归)+ 新增 `latestUsageMissing`(never false,测到账的
89
+ 轮删键)。
90
+ 同族**第四处**(轮二实抓,也是这条链的**末端**):A 层 `result` 臂的**终帧补发腿**
91
+ (`response_metrics{phase:'end'}`)拿的是终帧 usage,而缺账 run 的终帧 usage 是占位全零 ⇒ 修前
92
+ 会补发一条没有任何判别位的精确零度量,而本臂的宿主义务逐字是「驱动 responseLength reducer」。
93
+ 前三处收口了而末端没跟上 = 过滤没贯穿到底。修形同一条判据(下界位在场且读数为 0 ⇒ 不补发,
94
+ 非零照发),并补一条 **`runStream` → `adapt` 整链**回归(缺账零 / 真零 / 缺账非零三形)。
95
+ 同族**第五处(🔴 公面 BREAKING,轮三实抓)**:公开读器 `turnEndUsage` 的 `undefined` 含义从
96
+ 「`usage` 这一格缺席」收窄成**「这一轮的账不知道」**(整格缺席 / `usageMissing` 且六格全零两种
97
+ 入形都答它;**非零照交**)。理由是**跨版本一致性**:#711 之前「这一轮没量出账」的 wire 形是
98
+ 不发 usage ⇒ 它答 `undefined`;#711 之后同一件事的形变成六个 0,不改的话同一个真实情形在引擎
99
+ 升级前后由同一个公开读器给出两个相反的答案,而调用方一个字都没改。⚠️ 要**原样**镜像的调用方改调
100
+ `turnUsageToModelUsage`(纯映射,不带缺席语义)—— `runStream` 的违约闸走的正是后者,否则一条
101
+ **合法**的占位帧会被误判成违约并整帧丢掉(反向钉在门里)。
102
+ 🔴 **轮四再订正两处**(两条都对照 main 复现过):① `RunStreamHandle` 的规矩从「缺账轮不覆盖」
103
+ 收窄成「**占位**轮不覆盖」—— `usageMissing` 可以与真数字同帧(core 在同一轮里攒到过数字而另一次
104
+ 调用报了缺账),blanket 跳过会把真读数丢掉(实测 5 → 42+缺账位,main 的 handle 到 42、blanket
105
+ 写法停在 5);读数更新与判别位从此**分开决定**。② **坏形 `usage`**(`null` / 标量 / 数组):wire 是
106
+ JSON、SSE 解析原样透传 ⇒ 这些形真到得了包边界,而 `usage: null` 修前会让映射当场抛 `TypeError`
107
+ ——**整条流断掉**(后面的 `done` 一并丢);标量则被映成一份「看起来已测量」的全零账。成形判据
108
+ 一律改成「非 null 的非数组对象」,坏形与缺席走同一条违约路,**流不断**。
109
+ 🔴 **轮五再订正一处(过滤本身引入的跨轮回归)**:A 层 `turn_usage` 臂的 `emitTurnUsageEnd` 一直在
110
+ 做**两件事** —— 发 end 度量 **与** 复位下一 model round 的 TTFT 基线。缺账轮没有数字可发,整条
111
+ 跳过那个动作就**连模型轮边界也一起跳过**:下一轮不再发 `response_metrics{start}`,它的用量按
112
+ **上一轮**的基线对账(整链实测:首轮 400 字符缺账、次轮 4 字符 + 50 tokens,计数从 150 掉到 101,
113
+ 下一轮 TTFT 一并丢失)。⇒ 边界拆成独立动作 `noteModelRoundBoundary()` 无条件走,**只压基线、
114
+ 不置「已发过 end」**(终帧若真带了一笔非零账,补发腿仍该补)。判据分家的那句话值得记下来:
115
+ **「这一轮结束了」与「这一轮花了多少」是两件事** —— 合成一个动作,过滤数字就会顺手把边界也过滤掉。
116
+ - **B2 终态词两表分源**(L-247;core [7067] @cli 行顺答)。`TERMINAL_STATUSES` /
117
+ `TERMINAL_NOT_SUCCESS_STATUSES` **删除(不留别名)**,换成三张**字面元组** + 派生型:
118
+ `TERMINAL_CAUSE_KINDS`(core 终局**因由**闭集 `completed|failed|blocked|paused`)与
119
+ `RUN_TERMINAL_STATUSES` / `RUN_TERMINAL_NOT_SUCCESS_STATUSES`(**server run 行状态面**)。
120
+ 病根:修前一张表把**两个属主**的词合在一起 —— `killed` 根本不是 core 的词(core 因由集里没有它),
121
+ 而 `paused` 是 core 的词却**不是**「run 结束了」。合成一张之后,「core 加第五个因由」与「server 加
122
+ 一个行状态词」在这一端长得一模一样,消费方只能把新词折进已知词(不安全侧)或两边都漏。
123
+ 🔴 **不留别名**的理由:留一个别名 = 端可以继续按合起来的那张表读,分源就白做了。
124
+ 谓词 `isTerminalStatus` / `isTerminalNotSuccess` 名字与语义**未动**(现在是类型守卫),新增
125
+ `isTerminalCauseKind`。core 那张闭集是**抄件**不是意见:新门对**实装 devDep core** 的
126
+ `TERMINAL_CAUSE_IS_REPLAYABLE` 键集(声明面与运行期面双读且要求同源)逐词双向对账 + 顺序同源,
127
+ 并把「为什么是镜像而不是 re-export」做成**存活断言**(core barrel 今天只 re-export 类型
128
+ `TerminalCause`;它哪天导出闭集值,门当场红逼人复核)。
129
+ - **B3 `delegation.ask_unresolvable` 的 `detail.settlementKind` → `detail.cause`**(core 7.17.0 #709 ②)。
130
+ 本包新给 `ASK_UNRESOLVABLE_CAUSES` 闭集与事实窄读器 `readAskUnresolvable`;`settlementKind`
131
+ **零残留**并登记进退役普查(`src/` 代码位置零命中 + 上游存活断言)。
132
+ 🔴 **不做双读**:两者不是改名 —— in-fold 拒绝那一臂(`mandate_unreconstructible`)**根本没有结算**
133
+ 骑在帧上,旧位在那一形上恒缺席;一条「读不到 cause 就读 settlementKind」的兼容读会在**最需要它的
134
+ 那一形**上给出缺席,并且让端以为自己兼容了老引擎。
135
+ - **B4(编译期 BREAKING,L-245)两只措辞铸点的入参从 `unknown` 收窄成闭集成员型** ——
136
+ `classifierDenyCauseDetail(cause: ClassifierDenyCause)` / `ruleStoreUnreadableDetail(kind:
137
+ RuleStoreUnreadableKind)`;`CLASSIFIER_DENY_CAUSES` / `RULE_STORE_UNREADABLE_KINDS` 同批改字面元组
138
+ 并各出一个成员型,措辞表的型改 `Record<成员型, string>`。那句「a word newer than this client」的兜底
139
+ **结构不可达**(唯一到达铸点的路已按闭集判过)、而且**说的是假话**(真有一个比这一端新的词时它压根
140
+ 不会到这里),更坏的是它假装这一面是开集 ⇒ 没人给这张表配编译期围栏,core 加词那天一声不响。
141
+ 收窄之后**加词 = 表少一个键 = 编译期当场红**。
142
+ ⚠️ **隔壁那一句刻意相反且两边都对**:`askOriginDetail` 的兜底**留着** —— `AskOrigin` 在 wire 上是
143
+ 开集,那一句真的会被走到。一套面一条规矩:闭集则收窄 + 围栏,开集则保留兜底。
144
+
145
+ ### Added(additive;端可以不接)
146
+
147
+ - **`readWorkflowParks`**(core 7.17.0 #652 / server [7084] D 段)—— `GET /v1/workflows/:id` 的
148
+ `parks[]` 三键读器。🔴 **凭据在本包是结构性不可达的**:读器**逐键挑** `{callKey, sessionId,
149
+ originRunId}` 铸行、从不 spread,所以上游哪天在这张行上多放一个 `token`(或任何别的凭据形),
150
+ 它**在结构上**到不了本包的产物;门用一条带凭据的投毒行直接证它(产物**整棵树**序列化后零命中,
151
+ 且同一把尺子对**输入**判得出),并对源码本身钉「读器里零对象展开」。
152
+ 🔴 **缺席两义**:`parks` 整键缺席 = 旧引擎写的记录、**证不出**有没有 park;`parks: []` = 一句正面
153
+ 事实。两者的下一步相反(拒 resume / 放行 resume),折成一个字节就是本仓反复在修的那条病。
154
+ - **`WORKFLOW_PARK_REFUSAL_CODES` / `isWorkflowParkRefusalCode`** —— workflow **park 真相**四拒码
155
+ (`park_truth_unreadable` / `park_not_pending` / `park_binding_broken` / `park_requires_run_store`)。
156
+ 判据**绝不靠 `workflow.park_` 前缀放宽**(前缀是命名巧合不是契约)。
157
+ 🔴 `workflow.journal_incompatible` **仍是一个活码**(退役的是它的「journal entry missing」那一条臂),
158
+ 门里有反向钉守着 —— 把整个码当退役会让一族真拒绝在这一端无声消失。
159
+ - **`DURABLE_MANDATE_SOURCES` / `readDurableGateUnavailable`**(core 7.17.0 #709 ①)——
160
+ `source` **闭集**(恢复动作的分支键)、`cause` **开集透传**(core 今天不导出那两个词的闭集,铸点是
161
+ 一个三元表达式 ⇒ 抄一份就是本包自铸词表)、两个布尔位**读不出则键不铸**(不折 `false`)。
162
+ - **`heldBy` / `cancelRequested` 读点**(L-230;server 7.73.0 S-122 P-45)—— `waitForClaimRelease` 多一条
163
+ **更早、更硬**的释放证据:`heldBy === null` 是引擎**直说**「会话交出来了」,比拿终态词推断强
164
+ (park 态保留 claim 正是那条推断会踩的坑)。三态刻意不是布尔:非空串 = 还占着 / `null` = 释放了 /
165
+ **整键缺席 = 老引擎,读不出**(回落既有白名单,**老引擎零行为变化**)。`ClaimReleaseVerdict` 新增
166
+ `lastHolder`(与 `lastStatus` 同律:只记最近一次**有回答**的探测);`confirmedHeld` 改两条取并。
167
+ 新增 `readCancelRequested` 三态读器,且它读的是**「请求已受理」不是「已取消」**。
168
+ - **`SEMA_SUBAGENT_USAGE_PARTIAL_KEY` / `subagentUsageIsPartial`**(L-244 包侧半场)—— 子代用量面的
169
+ 下界判别位**自有名**,与终帧的 `_sema_usage_lower_bound` **分名**(修前壳把两件事铸成同一个键名,
170
+ 于是一个按键名聚合的面会把「一条只是还没收口的子代行」算成整条 run 的账不可信)。
171
+ - **`ENGINE_STOP_REASONS` / `engineStopReasonToCc`**(L-246 A13)—— 引擎 turn 停止原词 → CC
172
+ `stop_reason` 的**唯一映射口**,从壳下沉包内(CC 皮肤词汇表本来就该住这里;住在壳里的后果是三端
173
+ 各抄一份五个词)。**三态**:`string` / `null`(映到**诚实缺席**)/ `undefined`(**本端不认识**)——
174
+ 后两者合成一个值会让引擎加第六个词那天被渲成一次「这一轮没有 stop_reason」的肯定事实。
175
+ - **`ClassifierRoundObservation`**(L-245 B6)—— `classifierStatusOf` 第二参的**窄观测型**:端手里只有
176
+ 两格裸串时**直接交 `{disposition}`**,不必再铸一个假门记录喂进来(索引签名是故意的,整只 wire 帧
177
+ 照样喂得进来)。
178
+ - **码册 / 词表跟车四件**:`ENGINE_NOTICE_CODES` +1 `config.artifact_host_invalid`(audience `operator`;
179
+ 五十一码 ⇒ 五十二码)/ `STRUCTURED_DETAIL_TYPES` +1 `artifact`(**同一个恒绿病根第五次**,抬对账物后
180
+ engine-vocab ⑤ 段当天红抓出)/ `BrainRetryErrClass` 补第七桶 **`stall`**(它与 `transport` 分家,
181
+ 下一步不同)/ `RetryStatus` 的 `stalled` 臂**透传引擎那句 `detail`**(修前这一位只在两条 error 臂上
182
+ 被读,最长的那一段等待反而看不见引擎给的解释)。
183
+
184
+ ### Changed(内部结构,端零感知)
185
+
186
+ - **落键姿势单源化**(L-246 B2)—— `__proto__` 陷阱的 `defineProperty` 落键从两份同形实现
187
+ (`terminalToSdkResult` / `parkResolver`)收成单源 `src/ownKey.ts` 的 `putOwnKey`。
188
+ index 闭包文件数棘轮 158 → **159**(逐条账在 `scripts/run-client-core-portability-test.mjs` 头注:
189
+ 零 import 纯叶、零 Node 内建、零新外部包;抬这个数换来的是**少一份同形实现**)。
190
+ - **`DurableRunRecordView` 具名**(修前是 `DurableRunVerbs.get` 返回位上的内联匿名形)。
191
+ - **typeshape 棘轮**:`unknownExport` 324 → **336**(−4 收窄/具名、+16 全是 wire 边界读器入参或 wire 值位,
192
+ 逐条实测差集写在门的 `RATCHET` 头注里);`b4` 恒 **20** 不动、`bareUnknownReturn` 恒 **0** 不动。
193
+
194
+ ### Removed(BREAKING)
195
+
196
+ - `TERMINAL_STATUSES` / `TERMINAL_NOT_SUCCESS_STATUSES`(随 B2 更名,**不留别名**)。
197
+
198
+ ### 门(新增两道 + 扩六道)
199
+
200
+ - 🆕 `scripts/run-terminal-word-source-test.mjs` —— 两表分源的对账门(见 B2)。
201
+ - 🆕 `scripts/run-workflow-park-truth-projection-test.mjs` —— parks 三键读器的**凭据结构性不可达**、
202
+ 缺席两义、四拒码对 core 真字节直证、`journal_incompatible` 仍活的反向钉。
203
+ - 扩:`run-assistant-arm-identity-test`(S-1 四组,**红先绿后**的证据就在这一段)/
204
+ `run-subagent-usage-projection-test`(占位六零不当读数)/ `run-engine-notice-catalog-test`(两只窄读器)/
205
+ `run-selfheal-reopen-test`(`heldBy` 三态与老引擎既有路)/ `run-engine-vocab-floor-test`
206
+ (**G2-d**:`BrainRetryErrClass` 对实装 core 双向等值 —— 此前这一族**连门都没有**)/
207
+ `run-retired-vocabulary-census-test`(登记 +1 `settlementKind`)。
208
+ - **枚举器补层**(`run-client-core-singleton-test`):`Object.freeze([… ] as const)` 且**无类型注解**的
209
+ 写法此前**整类**不进模块级单例普查(初值是属性访问调用 ⇒ 工厂判据不认;没有注解 ⇒ 注解路走不到)。
210
+ 补层当天在 `src/` 里实捞出 **5 条存量**(`peerFrames` ×4 / `subagentContentStore` ×1)——
211
+ 它们不是新风险,是本来就该在册而一直隐身的。自检语料同批加两条已知判决。
212
+
213
+ ### 已知局限(本版新增;完整台账见 `docs/INTEGRATION-CLIENTS.md` §6e/§7)
214
+
215
+ - **引擎地板**:B1 起本包的 `turn_usage` 臂按 core ≥7.17.0 的铸形写。在更老的引擎上,真实的
216
+ 「该轮零 usage」帧走违约路(留痕 + 不投影 + 立下界位)—— 账仍诚实,丢的是那一拍的实时增量。
217
+ - **`parks[]` 供给面未端到端实证**:server 7.74.0 尚未发布(本批写作时 registry latest = 7.73.1),
218
+ 读器按 [7084] D 段的契约写并由构造素材覆盖;真 wire 上的首次实证归 test 线。
219
+ - **sdk / agent-types 本批不升**(devDep 停 `^8.8.0` / `^0.2.0`):①抬 peer 地板本身是一次**非 additive**
220
+ 面,按本包既例要独立记账与影响面账,不搭 BREAKING 车;②实测直证 sdk 9.1.0 **不带**本批任何一个键
221
+ (`RunStatus` 逐字未变、无 `WorkflowRun.parks`、无 `heldBy`/`cancelRequested`)⇒ 本批零受迫。
222
+ `heldBy` / `cancelRequested` / `parks` 三处因此走**结构视图读**(与本包 `riskDescriptor.probeCause` /
223
+ `ruleOffers` 同款姿势),类型半场候 sdk 班车。
224
+
225
+ ## 0.67.2(2026-09-12)
226
+
227
+ **异源对抗复审轮一三条 [medium] 的修复批**(patch;零公面导出
228
+ 增删、零 wire 键增删、零 BREAKING;逐项的形 / 判据 / 三端待办见 `docs/INTEGRATION-CLIENTS.md` §32f、
229
+ §32g 两段「0.67.2 订正」)。
230
+
231
+ - **I-1 流内 usage 缺口观测位是 per-stream,不再住在共享 `EmitContext` 上** —— 那一位此前写在
232
+ `ctx.usageMissingObserved`、且**只置 `true` 永不清**,而 ctx 是**调用方的对象**、可以复用给多条流
233
+ (`startedAtMs` 本来就是这么用的)⇒ 顺序复用时上一条流的缺口把**下一条账数得全**的流的终帧与
234
+ `run_cost_reconciled` 一起标成下界,并发复用时两条互串。修形:观测位落在 `runStreamInner` 的局部量上,
235
+ 终局把**本流快照**按值同时交给两个投影口(`terminalToSdkResult` 第三参 / `readRunCostFacts` 第二参)——
236
+ 两面仍读同一次计算,但谁都读不到别人那份。`EmitContext.usageMissingObserved` **退役**(该位原本就逐字
237
+ 写着「宿主不要自己填」)。反钉:顺序复用第二条流两面都不铸 + 同 ctx 上真有缺口的第三条流照铸 +
238
+ 并发交错互不串 + ctx 上不再留位。
239
+ - **I-2 两种身份的键空间碰撞不再绕过混合出身检测** —— 行键的两个命名空间(真身份 `sourceTaskId` /
240
+ 回落 `parentToolCallId`)**字面可相等**而上游不保证互斥;修前裸 id 当累加键,撞字面的两行在累加层
241
+ 就并掉、行上的出身位只剩一种 ⇒ 0.67.1 那道「出身混合 ⇒ partial 恒立」的闸读不出混合被绕过
242
+ (三帧反例:行数与轮数两条对账同时成立、`partial` 不铸,而 A 实际 40 / B 20 被错并成 30 / 30)。
243
+ 修形:累加键按出身前缀隔离(`s:` / `p:`)、逐事件记来源;交付面的键**仍是裸 id**(端零改),跨空间
244
+ 同字面合并成一行并立新的行级判别位 **`keyCollision: true`**(never false)、`partial` 恒立。
245
+ 合并臂照走 `putOwn`(`__proto__` 纪律不因第二条路被绕开)。
246
+ - **I-2b 子代分表本身也不再住在共享 `EmitContext` 上**(件 I-1 的同形存量,轮二复审实抓)——
247
+ `ctx.nestedUsageByTask` 流结束后留在调用方对象上,而三只终帧投影器是**公面导出**:端「A 走
248
+ `runStream`、B 直调终帧投影」共用一个 ctx 时,B 的终帧带出 **A 的**分表,且 B 的 `nested` 计数恰好
249
+ 对得上时连 `partial` 都不铸(一张属于别人的表被标成「可证完整」)。修形与 I-1 同一条:分表随终帧
250
+ 第三参按值传,`EmitContext.nestedUsageByTask` **退役**;没传快照 ⇒ 两个分表键都不铸(诚实缺席)。
251
+ - **I-3 门负控的备份改独占创建** —— `scripts/run-gate-negative-controls-test.mjs` 的
252
+ `existsSync` + `copyFileSync` 之间没有互斥且复制默认允许覆盖,两实例交错即可把「唯一复原依据」
253
+ 换成**已篡改**的内容(源复原不回来、备份也被删),而本套的全部立论就是「演练不留痕」。改
254
+ `copyFileSync(..., COPYFILE_EXCL)`(检查与创建同一次系统调用),已存在 ⇒ 响亮拒绝不覆盖;
255
+ `EEXIST` 之外的 errno 原样抛出。新增自证④ 在 `mkdtemp` 隔离目录里调度那次交错。
256
+
257
+ **消费方待办**:无。wire 键零增删;`_sema_nested_usage_by_task` 的行上多一个 additive 判别位
258
+ `keyCollision`(它在场时 `_sema_nested_usage_by_task_partial` 必然同时在场 ⇒ 只读 `partial` 的端行为
259
+ 逐位不变);`EmitContext` 上退役两格(`usageMissingObserved` / `nestedUsageByTask`),三端零读写点、经
260
+ `runStream` 的路径逐位不变。cli 侧只需把 lock 抬到 0.67.2。
261
+
262
+
52
263
  ## 0.67.1(2026-09-12)
53
264
 
54
265
  **test [7055] 对 0.67.0 的 G32-01~30 验证批回帖三处真实发现的修复批**(patch;零公面导出增删、零
package/README.md CHANGED
@@ -35,7 +35,7 @@ Renamed from **`@sema-agent/wire-cc-adapter`** (0.1.x, deprecated — see *Migra
35
35
 
36
36
  ## Scope
37
37
 
38
- **Version:** 0.67.1
38
+ **Version:** 0.68.0
39
39
 
40
40
  - **Today** — the adapter seam, the whole `adapt()` pipeline (all 14 A-layer arms plus the
41
41
  B/D/E tool-card layers), the notification/caps/model families, the adapter kernel (stream driver
@@ -206,6 +206,67 @@ Known intentional deltas from the CLI are enumerated in `ADAPTER_DIVERGENCES`. T
206
206
  ledger is a different table (`MIGRATED_COMPENSATIONS`): the entries there behave identically on
207
207
  both sides — what it records is why a compensation exists and when it can be deleted.
208
208
 
209
+ ## Upgrading to 0.68.0 — read this first (breaking)
210
+
211
+ This release follows an engine release in which three facts that used to be *sometimes absent* became
212
+ *always present*. Wherever this package had an arm for "the engine did not send it", that arm was either
213
+ dead code or was quietly reading a positive fact as nothing. The arms are gone; absence now means one of
214
+ three **distinguishable** things, and the distinction is the point.
215
+
216
+ **1. Per-turn usage is now always on the wire.** The engine emits a zeroed usage object together with an
217
+ explicit "this turn was not measured" flag instead of omitting usage. So a turn-end frame with **no** usage
218
+ at all is no longer a legal shape — it is a contract violation, and this package now says so out loud
219
+ (through the host's dropped-frame sink, with its own verdict word and its own sentence, because the generic
220
+ one — *this build has no arm for it* — would point the reader at the wrong side) rather than silently
221
+ projecting something. Such a frame is **not** projected at all, and the run's totals are marked as a lower
222
+ bound, because dropping the frame must not let a run that genuinely lost an accounting turn report an exact
223
+ number. **The engine floor moves with it**: on older engines a real "no usage this turn" frame takes that
224
+ path. There is deliberately no version-sniffing compatibility read — that would require this package to hold
225
+ a number it cannot see, and guessing is worse than saying plainly that a frame does not match the contract.
226
+ One consequence worth knowing: on a turn flagged as unmeasured, the zeroes are a *placeholder*, and that
227
+ rule is carried all the way down the chain — the per-turn output count, the per-sub-agent cache-read total,
228
+ the handle a footer reads, the end-of-turn metric the response-length reducer consumes, and the public
229
+ per-turn usage reader all withhold the placeholder rather than hand out an exact zero. **A non-zero value is
230
+ never withheld**: a placeholder is by construction all zeroes, so anything non-zero was genuinely measured
231
+ and is passed through as a lower bound. One of those five is a **public** reader whose meaning therefore
232
+ changed: the value it returns for *nothing known about this turn* is now the same before and after the
233
+ engine upgrade — which is the point, since otherwise one unchanged caller would have silently started
234
+ reading "this turn cost exactly zero". If you want the raw, judgement-free mapping instead, call the pure
235
+ mapper beside it.
236
+
237
+ **2. The ending words are now two tables, kept apart by who owns them.** The single combined list is
238
+ **removed with no alias** and replaced by the engine's closed set of *reasons a run ended* and the server's
239
+ set of *row states a run can finish in*. They share three words but not all of them: one word for *something
240
+ outside stopped it* exists only on the server side, and one for *it paused and can be resumed* exists only on
241
+ the engine side and means very nearly the opposite of an ending. Merged, a new word on either side looked
242
+ identical, and the tempting move — folding the unknown word into a known one — is exactly the mistake this
243
+ package exists to prevent. Both tables are literal tuples with derived types, so a client should **derive**
244
+ rather than hand-copy the words. The two predicates keep their names and meanings (and are now type guards);
245
+ an alias was deliberately not left behind, since one would let a reader keep consuming the merged list and the
246
+ split would have bought nothing.
247
+
248
+ **3. The "an approval could not be resolved" notice now carries a cause, not a settlement word.** These are
249
+ not two names for one thing: one of the three causes is an in-fold refusal that carries no settlement at all,
250
+ so the old field was always absent exactly where it mattered most. A "read the new field, else the old one"
251
+ compatibility read would therefore answer *nothing* on the one shape that needs it, while letting a client
252
+ believe it had covered older engines. This package does not do that read, and a standing check keeps anyone
253
+ from adding one later.
254
+
255
+ **Also breaking, at compile time only:** two sentence-minting functions narrowed their parameter from an
256
+ unvalidated value to the closed set they actually serve. The fallback sentence they carried was structurally
257
+ unreachable *and* untrue — it announced "a word newer than this client" for a value that could never arrive —
258
+ and worse, it made the surface look open, so nobody had put a compile-time fence on the table. With the
259
+ parameter narrowed, adding a word upstream now fails to compile until the sentence is written. The neighbouring
260
+ function whose vocabulary really **is** open keeps its fallback: one rule per surface, not one rule for all.
261
+
262
+ Everything else in this release is additive and safe to ignore until you want it: a reader for the parked
263
+ approvals on a workflow record (built so that a credential **cannot** structurally reach the output, and with
264
+ *no field at all* kept distinguishable from *an empty list*), a closed set for that family's refusal codes,
265
+ readers for two notice payloads, a direct answer for "has that run released the session yet" that supersedes
266
+ inferring it from a status word (absent on older engines, where behaviour is byte-identical), a distinctly
267
+ **named** key for the sub-agent lower-bound bit that used to collide with the run-level one, and the
268
+ engine-stop-word to session-vocabulary mapping moved in here so the clients stop each keeping a copy.
269
+
209
270
  ## Guards
210
271
 
211
272
  ```bash
@@ -250,7 +311,7 @@ public-surface guard checks that last one).
250
311
  | `scripts/run-wire-auth-source-test.mjs` | **When** the outbound credential is read. A literal string is consumed at construction — the transport captures it in a closure and every later request reuses that one copy — so once the engine is replaced by another session and the credential rotates, a long-lived client keeps presenting the old one and the only way out is to rebuild the client along with everything hanging off it. The credential position now also accepts a getter that is called **once per outbound request**. The guard anchors on the deciding quantity, which is not "was the getter called" — reading once at construction and reusing the result would satisfy that too, and is exactly the shape being removed — but *which read produced the value on the wire*: it changes the getter's answer between two requests through the same client and requires the second request to carry the new one, and it requires construction to read the getter **zero** times. The three-state credential semantics are replayed per request rather than assumed: on loopback an unavailable credential sends **no** authorization header at all rather than a fabricated one, off loopback it sends the fail-closed anonymous identity so the deployment answers with an honest 401, and the guard shows a single client moving between those states across successive requests. A getter that throws is fail-soft — the request still goes out under the no-credential branch, because a broken credential port should not take the whole wire down, and the exception may itself carry credential material. The same-origin relay form is checked to stay out of the getter path entirely, and every request is checked to keep the credential in the authorization header only — never in the URL, never in another header |
251
312
  | `scripts/run-subagent-durable-divert-test.mjs` | The side-channel that keeps a **sub-agent's** content out of the leader's transcript, on the replay leg. A content frame stamped with a parent tool-call id belongs to a child, and rendering a child's tokens as the leader's own text is the pollution this divert exists to prevent — but the predicate only listed the four **live** frame shapes, while the durable leg replays the same segment in its **aggregated** form. Those frames fell straight through onto the main projection path, which is how a reconnect or a resumed session ended up with the child's answer printed as the leader's. The anchor is unchanged and shared: the parent tool-call id is what says whose frame this is, and whether the frame is an increment or a whole segment has nothing to do with whose it is — judging the two shapes separately is exactly how one of them got missed. Folding the aggregate into a synthetic increment would have been the smaller diff and the wrong one: an increment means *append*, so a segment that already streamed live and then replays whole would be counted **twice**. The two are kept distinct and the aggregate absorbs instead — a whole segment whose prefix is what the buffer already holds replaces it, which also makes a redelivery of the same frame idempotent, and a prefix that does not match falls back to appending both rather than deciding on the engine's behalf which version counts. Segment boundaries stay with the tool frames rather than moving into the aggregate arm, since closing there would turn a second replay of one segment into a second entry, and the increment arm is pinned to keep appending so a token run that happens to be a prefix of the next does not silently lose characters |
252
313
  | `scripts/run-subagent-content-budget-test.mjs` | The **byte** budget on the sub-agent transcript ledger. It used to be bounded only by *counts* — so many entries per child, so many children — and a count is not a budget when a single entry has no ceiling of its own: one tool result carrying an inlined attachment, or one long model answer, and a single slot sits on tens of megabytes. The guard anchors on how many bytes are **still held** after over-filling, not on whether truncation fired, because an implementation that flags the overflow without actually dropping anything satisfies the second and not the first. Dropping is required to leave a record — how much went and where the retained content now starts — and that record has to reach the render plan, because content that vanishes with no marker gives the reader a transcript shorter than what happened with nothing to say so; the record is one per child, updated in place, pinned to the front, and excluded from the budget it describes. Order matters and is checked: oldest entries go first and the live tail is trimmed only as a last resort, since taking the text the user is watching stream while older history survives is the wrong end. The total budget evicts a whole least-recently-used child rather than shaving every child, and the configuration surface is fail-loud on zero, negatives, non-finite and non-integer values — a silently ignored budget is the exact failure this exists to remove — with the rejection proven atomic so a bad second field cannot leave half a configuration behind. The defaults are checked to be a magnitude that can really be reached, since a number too large to hit is a field rather than a budget |
253
- | `scripts/run-subagent-usage-projection-test.mjs` | Per-subagent usage, split by task. The engine's final accounting carries the delegated spend as **one total** — tokens, turns, task count — and no per-task breakdown, while every sub-flow turn on the stream carries its own usage. This package used to fold that away at the leader/sub-flow divide (a child's output tokens must never reconcile the leader's response length), so a client showing a subagent's detail pane had nothing to print. The split table can therefore only be accumulated from the stream, and this guard pins what that costs. The two existing leader-only arms stay **byte-for-byte unchanged** — the new arm is additive and always carries the sub-flow's own lane proof, so a host cannot mistake a child's numbers for the session window. Attribution is by the engine's own originating-task id — deliberately not a second `taskId`, which the event identity does not carry and whose absence would silently collapse every child under one parent call — falling back to the parent call id; a turn that answers neither is dropped rather than filed under an invented row, because merging two children's ledgers is worse than missing one. Cache-read tokens are read from the **engine's own shape** rather than the mirrored one, since the mirror fills that member with zero when the wire omits it and reading it there would erase the difference between *not reported* and *no cache hit*. A turn that reported no usage at all still counts as a turn and still adds its zeros — the numbers are a lower bound, and dropping the round would make the bound less true, so the honesty bit rides on the row instead and is never spelled `false`; such a round still emits its live arm, because the frame that says "this round has no account" is the one a real-time consumer most needs and the easiest one to drop. The same honesty bit also survives a terminal that carries no statistics at all: what the stream observed is unioned with what the final record says, so a run that already reported an unmeasured round cannot come out the other end looking like an exact zero. Finally the table says whether it is **partial**, and that verdict is anchored on the quantity that actually decides it: the engine's own totals. Turn count and row count must both reconcile before the table claims to cover the whole run; anything else — including totals that cannot be read — marks it partial, so the failure direction is always the safe one (a complete table called partial, never the reverse). The two accounts are kept separate and are never added together or used to correct each other |
314
+ | `scripts/run-subagent-usage-projection-test.mjs` | Per-subagent usage, split by task. The engine's final accounting carries the delegated spend as **one total** — tokens, turns, task count — and no per-task breakdown, while every sub-flow turn on the stream carries its own usage. This package used to fold that away at the leader/sub-flow divide (a child's output tokens must never reconcile the leader's response length), so a client showing a subagent's detail pane had nothing to print. The split table can therefore only be accumulated from the stream, and this guard pins what that costs. The two existing leader-only arms stay **byte-for-byte unchanged** — the new arm is additive and always carries the sub-flow's own lane proof, so a host cannot mistake a child's numbers for the session window. Attribution is by the engine's own originating-task id — deliberately not a second `taskId`, which the event identity does not carry and whose absence would silently collapse every child under one parent call — falling back to the parent call id; a turn that answers neither is dropped rather than filed under an invented row, because merging two children's ledgers is worse than missing one. Cache-read tokens are read from the **engine's own shape** rather than the mirrored one, since the mirror fills that member with zero when the wire omits it and reading it there would erase the difference between *not reported* and *no cache hit*. A turn that reported no usage at all still counts as a turn and still adds its zeros — the numbers are a lower bound, and dropping the round would make the bound less true, so the honesty bit rides on the row instead and is never spelled `false`; such a round still emits its live arm, because the frame that says "this round has no account" is the one a real-time consumer most needs and the easiest one to drop. The same honesty bit also survives a terminal that carries no statistics at all: what the stream observed is unioned with what the final record says, so a run that already reported an unmeasured round cannot come out the other end looking like an exact zero. Finally the table says whether it is **partial**, and that verdict is anchored on the quantity that actually decides it: the engine's own totals. Turn count and row count must both reconcile before the table claims to cover the whole run; anything else — including totals that cannot be read — marks it partial, so the failure direction is always the safe one (a complete table called partial, never the reverse). The two accounts are kept separate and are never added together or used to correct each other. One more thing the totals cannot settle: the row key has **two namespaces** — the originating-task id and the parent call id it falls back to — and nothing upstream promises they are disjoint, so the same literal can name one child's identity and another child's parent call. Accumulation therefore keys on the origin as well as the id; the delivered table still keys on the bare id, and a cross-namespace clash is merged into one row that says so, with the partial verdict forced, because a row count and a turn count can both reconcile while the attribution behind them is wrong. The table itself is likewise a **per-stream snapshot** handed to the terminal projector by value rather than left on the caller's context: the three terminal projectors are public, so a host may drive one run through the stream and project another's terminal directly on the same context, and a table left behind would be attributed to whoever projects next — silently called complete whenever that run's own totals happen to match. Without a snapshot, both table keys are simply absent |
254
315
  | `scripts/run-result-text-backfill-test.mjs` | What happens when the terminal frame's answer text and the text already on screen do not match. A turn's answer normally streams in and the terminal frame carries the same words again, so the two agree — but when the connection drops mid-answer and the reconnect brings the finished version, "this turn already produced assistant text" is true, the terminal fallback is skipped entirely, and the screen stays permanently short of whatever arrived while the stream was down, with nothing to say so. Four cases are pinned. Nothing on screen yet: render the terminal text whole, byte for byte the previous behaviour. On-screen text is a **prefix** of the terminal text: emit only the missing tail, and the guard measures the deciding quantity — the total bytes that reached the screen must equal the terminal text, which fails both for a missing tail and for a re-render that would print the first half twice; when the two are already equal, nothing is emitted at all. Terminal text is a prefix of what is on screen (an engine-side trim): touch nothing, since there is nothing missing and overwriting with the shorter version would erase what the reader already saw. Neither is a prefix of the other: emit **nothing** and raise a fact instead — which version counts is the engine's to say, and appending the terminal version after the streamed one composes a passage nobody ever wrote. That fact carries lengths rather than text, so a renderer is not handed a third version to choose from, and its declared duty is to *reword* the transcript line, never to render more. A cross-segment case proves the comparison reads the whole committed answer rather than the last segment, and the whole thing is driven through the real two-stage path rather than hand-built messages |
255
316
  | `scripts/run-engine-vocab-floor-test.mjs` | Engine-mirrored vocabularies (structured card whitelist, self-reported tool face, control verbs, recogniser sets) against the *installed* `@sema-agent/core` |
256
317
  | `scripts/run-limits-env-failloud-test.mjs` | `SEMA_HEADLESS_*` env-lane limits reject invalid values as loudly as the flag lane (no silent "no budget" runs) |
@@ -292,14 +353,17 @@ public-surface guard checks that last one).
292
353
  | `scripts/run-decide-receipt-test.mjs` | What a decision verb actually **answered** — and, more importantly, what it did not. A success response on the newest lane is only an acknowledgement that the decision was accepted for delivery: the approval is still pending, and a client that clears the card on it shows either a ghost card that was already approved or a card that vanished while the decision was lost. So the package deliberately has **no** "was it resolved" predicate — nothing in that body can answer it — only the opposite one, whose `false` is likewise not evidence of resolution; resolution is only ever the next running arm on the stream. The guard pins that inversion in the product source too: the success path must no longer clear the latched gate, while the stream-observing path that really clears it must still be there. The body has four shapes with **no** key common to all of them, so every position is read as honestly absent, and the handoff handle — which run to watch from here on — requires **two** facts together, since either one alone would either point the stream at the run it already had or mint an empty handle. The record of what finally happened to an already-decided action is read through the **same** reader as every other gate record rather than a second copy, and its absence means **unknown**, never *it was allowed* — the two can even contradict each other, so the card says nothing at all when it is missing. The three refusals on that lane each get one distinct sentence and a disposition taken from **why** each was refused rather than from severity: one cannot be helped by re-sending at all, one waits on the host, one just drops an option — and none of them carries a countdown, because the server never mints a wait for them. Recognition is a **closed set**: an unrecognised code on the same prefix returns nothing rather than a guess, since that prefix also houses a safety signal whose whole rule is never to retry automatically, and the recovery handle is read as absent when unreadable rather than substituted from a different identifier that no longer appears on that lane |
293
354
  | `scripts/run-approval-frame-chrome-arms-test.mjs` | The two in-stream approval frames finally reaching every host through the shared pipeline instead of one shell's private branch — the shape of a layering defect: hosts that only consume the package could not rebuild their pending cards after a reconnect, and did not clear a card the engine had withdrawn. The payload is deliberately carried as the **envelope** the upstream types declare rather than the first-version card: the stream parser applies no predicate, so narrowing here would let a legitimately newer frame pass as the older shape and invite consumers to read keys a newer card never promised. The guard therefore pins that every open key survives untouched, that an unknown version still passes through, and that narrowing is left to the host's own predicates — with the fallback being a generic card and a person, **never** an automatic denial. A frame whose version cannot be read at all is reported as malformed rather than dropped in silence, because both frames carry user-visible decisions and state changes. Both arms are registered as **required** host duties, and their duty text names the load-bearing rules a host would otherwise have to rediscover: which predicate to narrow with, that the reconnect preamble — not a replayed historical frame — is the authority on which cards exist, and that a withdrawal frame can be lost entirely. Unlike the sibling arms, these carry **no** sub-stream cutoff: an approval raised under a delegated call still has to reach a person, and filtering it by ownership is the host's job, not a reason to discard it. Finally the upstream bytes that justify the envelope discipline are checked to still be there, since the whole design rests on them |
294
355
  | `scripts/run-terminal-status-vocabulary-test.mjs` | One place that decides whether a run has **ended** and whether it ended badly — written because that judgement had already been hand-copied three times, so the day the engine added a word for *the agent itself reported it cannot continue*, every copy missed it and a panel settled a self-reported failure as a success. The distinction the table exists for is pinned from both sides: that word belongs in it, while the two words meaning *waiting for a person to decide* deliberately do **not** — reading those as endings would bury a run that is actively waiting on the reader. A word this client does not know answers *no*, and the guard states plainly that *no* is not evidence of success: proving success means reading the positive side, so negating this predicate is the very mistake that caused two earlier incidents. The fleet lane gets the same treatment from the other direction: a workflow parked on a durable approval used to fall through to *running*, leaving the person with no hint that a card was waiting, and it now lands on the same rendered word the task lane already used — same fact, same word, checked end to end on a real row. Why the word was added directly rather than carried as a private superset key is checked mechanically against the upstream declaration being open, so the day it closes this reds and the decision gets revisited. The residue sweep is the point: the source tree must contain **no** further inlined copy of the judgement, each of the three former sites is checked to really read the single predicate, and the one reviewed exemption carries its reason **and** a liveness assertion, so an exemption whose justification expires cannot quietly keep standing |
356
+ | `scripts/run-terminal-word-source-test.mjs` | Two tables of ending words, kept apart by **who owns them** — because they used to be one. The engine's own closed set of reasons a run ended, and the server's set of row states a run can finish in, overlap in three words but not in all of them: one word for *something outside stopped it* exists only on the server side, and one for *it paused and can be resumed* exists only on the engine side and means very nearly the opposite of an ending. Merged into a single list, those two sources became indistinguishable, so a new word on either side looked the same as a new word on the other, and the safest-looking move — folding the unknown word into a known one — is the exact mistake that has caused incidents here before. The engine-owned table is checked as a **copy, not an opinion**: it is reconciled word-for-word and in order against the installed engine package, read from both its declaration and its runtime bytes with the two required to agree, so the day upstream adds a fifth reason this reds before anything ships. The two dividing words are each pinned from both sides, including against the upstream declaration directly rather than only against this package's own list. Why the table is copied rather than re-exported is itself an assertion with an expiry: the day upstream publishes the set as a value, this guard reds and the decision gets revisited. The renamed tables leave **no alias** behind, since an alias would let a reader keep consuming the merged list and the split would have bought nothing |
357
+ | `scripts/run-workflow-park-truth-projection-test.mjs` | The read face for *which approvals a workflow run left parked* — and the credential that must never ride along with it. Upstream strips the redemption token from that response, and this package's reader is built so the token **cannot** come back: each row is assembled field by field from the three identity keys, never copied wholesale, so an extra key appearing upstream is structurally unable to reach anything this package hands a UI. The guard proves that rather than asserting it — a poisoned row carrying a secret is read, and the secret is searched for across the **entire** serialized result, with the same search proven to find it in the input so a blind search cannot pass; renaming the credential key does not help it through, because the rule is *only these three*, not a blocklist; and the reader's own source is checked to contain no object spread, since one such line would quietly void all of it. The other half is an absence distinction with opposite consequences: a record with **no** parks field at all was written by an older engine and proves nothing about whether approvals are waiting, while an empty list is a positive statement that none are — collapsing those two would let a run whose parked approvals cannot be proven be resumed anyway, so they are kept literally distinguishable, and a payload whose rows are all unreadable answers *unknown* rather than *none*. The four refusal codes for this family are checked code by code against the engine's real bytes, never matched by name prefix, and the older umbrella code they were split out of is asserted to still be **alive** — treating the whole code as retired would make a family of real refusals vanish silently |
295
358
  | `scripts/run-retired-vocabulary-census-test.mjs` | Whether a retirement really happened. When upstream removes a family, a downstream package can cut it out or keep a courteous alias — and the alias is the worse outcome: three clients keep writing branches for something nobody emits, and a status line advertises a state it can never reach. Choosing the clean cut only means something if a guard holds it, since a comment saying *retired* is not an exit code. Each registered entry is held two ways: the name must be gone from **code positions** in this package (comments stripped first, because the explanation is supposed to stay) and off the published surface, and — the half that keeps this from being self-congratulation — it must really be gone **upstream**, since that is the entire reason it was removed here; if it comes back, the disposition deserves reconsideration rather than silence. The scanner proves it can speak by finding a symbol that is genuinely present before any absence is believed, and distinguishes a mention inside a comment from one in a string literal, which is exactly the form being cleared. A closing check runs the other way: the retirement **story** must remain in the comments, including a promise this package made earlier and has now had to withdraw — deleting the history alongside the code is a bad way to satisfy *zero hits*, and leaves the next reader with code that has no reason |
296
359
  | `scripts/run-classifier-status-test.mjs` | What state the auto-mode classifier is in **on this session** — the question a doctor line, a model settings page and a permission card’s status row all ask, and a different question from the one the approval card asks (*why am I being asked right now*), so the sentences are pinned mutually distinct from that face’s as well as from each other. The session-level half of this reading — a breaker record the engine used to keep — was **retired upstream**, and the guard now holds that retirement from **both** sides: the engine's own declarations must really no longer carry it (a fact coming back would mean the removal here was the wrong disposition, and that deserves a conversation rather than silence), and this package must carry no alias, no state word and no leftover narrowing for it — a reading kept alive for something nobody emits any more is a promise the interface cannot keep, and it left the doctor line advertising a state it can never reach. What remains is ordered by the quantity that actually decides whether the classifier is running: the fact from **this round** first, then whether this leg is armed — a decider is minted per run, so a later leg can be armed again. Not armed, and a section that never arrived, both answer **undefined** rather than *available*; that arming question has its own field and answering it twice grows a second ledger. Arming and availability are also **two words, not one**: the engine says a decider was minted *for this leg*, which is an assembly-time fact, while whether that decider answers any given round is a **per-call** one — so an armed leg reads `armed` and only a positive per-call fact (an ask whose origin is the classifier's own denial-bound fallback, which by construction stands *after* the classifier ran) reads `available`. Every other ask origin is refused as evidence and for a stated reason rather than out of caution: several are ones the classifier is structurally forbidden to answer, and for the rest a surviving ask is precisely the case where it did **not** resolve one — so reading availability off them would be a guess. The projection is a **whitelist**, so an older engine still sending the retired member loses it at the boundary while the two live facts beside it ride through untouched. Rendering never throws and never impersonates: a state word this client does not know — including the retired one, which a restored view can still carry — reaches an honest fallback that names it verbatim, carries no invented explanation of a mechanism that no longer exists, and is proven distinct from all three real sentences; prototype keys reach that same fallback rather than a function body, checked against a real out-of-table word so the comparison cannot hold vacuously |
297
360
  | `scripts/run-compaction-boundary-projection-test.mjs` | The compaction divider and the one frame that makes its anchor resolvable. The trigger word is passed through as an **open set** instead of being folded to two: the engine deliberately stopped flattening its third value (a compaction that was not optional — a prompt-too-long recovery or trim pressure) and carries what the hook layer saw, so folding it again at the package boundary re-introduces exactly what upstream had just removed, while a consumer branching on *is it manual* keeps its behaviour byte for byte. Only an unreadable word (absent, empty, non-string) falls back — that is *could not read it*, not *read it and did not recognise it*. Two superset keys ride the metadata and neither fabricates: the preserved-segment anchor is minted only when its id really reads out, because half an anchor sends the host looking up an empty string in its map, and the clamp ratio is a **disclosure** whose real zero is a fact rather than an absence. The clamp ratio also carries a registered exit condition — the service really sends it while the SDK arm has no seat for it yet, so the read is defensive and this guard reds the day that seat appears, forcing a re-check instead of leaving a cast to rot. The committed-message frame moves out of *deliberately not projected*: that classification was true about transcript rows and false about **positioning**, since the engine states that consumers build their own id-to-message map from this frame to place the divider — projecting the anchor without it hands the host something it cannot resolve. It becomes a neutral internal arm and an optional chrome ledger event, never a transcript row (the frame carries no body, so minting one would put words in the engine's mouth), with both required ids narrowed and a malformed frame recorded rather than half-minted |
298
361
  | `scripts/run-cost-absence-projection-test.mjs` | Telling **declared free** apart from **never priced**, in both directions, because the package was getting each one wrong in the opposite way. The engine separates them on the wire — an absent cost means some spend had no price table, an explicit zero means the model declared itself free — and the result projector used to require a *positive* number, so a genuinely free run could not say so; while the per-model mirror folded absence to zero, so an unpriced run told a billing consumer it cost nothing. The total is now reported as the engine stated it, with absence and non-finite values alone reading as unknown, and a negative passed through rather than corrected, since a refund is a legal figure and the package is not a second accountant. The per-model figure keeps the CC shape intact — that field is a required number and *unknown* is simply not expressible in it — so the value stays zero and a **companion superset bit** carries the distinction, which means the two are read together and a reader that only ever looked at the number is unchanged; the bit is minted only in the absent case and never as `false`, since a key present with a false value reads as a third state. The same mint point serves both the wire's per-model split and the synthesised current-model row, so neither can drift. Alongside it the cache-write figure stops being a hardcoded zero and reads the field the wire has always carried, in both the flat usage and the synthesised row, and all four flat token slots move from a null-coalesce to a finite-number guard — the stats object has an open index signature and the wire is JSON, so a string or an infinity would otherwise land in a slot the types promise is a number, compiling green and surfacing only when something sums it |
299
362
  | `scripts/run-permission-denial-projection-test.mjs` | The terminal result's **permission-denial list** being the wire's real one rather than a hardcoded empty array. The session vocabulary carries a list of tool calls that were denied; the projector used to mint `[]` in both the success arm and the error envelope, which folded two different statements into one — *nothing was denied on this run* and *this frame carries no such ledger at all* (an older engine, a rejection envelope, a failure event that arrives without stats) looked identical. Each denied gate on the wire's human-review ledger now becomes one record, in wire order, carrying the keys the wire can actually honour: the tool name when it reported one, and a superset field with the engine's own short, redacted one-line summary of the call's input. **Two lists, deliberately.** The reference shape requires three fields on every element — tool name, call id, and the full input object — and the wire's ledger carries only the first. Filling the other two with an empty string and an empty object would be invention; putting a half-filled element into the reference array would break the element contract, and a strict consumer validating the stream drops the *whole* result message rather than one field. So the reference array admits only fully-formed records — empty today, and filling itself the day the wire grows the two missing fields, with no code change — while every record the wire really has rides a superset carrier beside it. A contract check pins today's absence, so that day turns this guard red on purpose. The companion bit means *this reference list cannot be claimed complete*: no ledger, an unreadable row, an unrecognised decision word (a rejected plan is not a denied tool call, and a row with no decision at all is not a judgement), or a record that could not be fully formed. Only its absence lets a reader say *zero denials*; it is never minted as `false`. Rows that cannot be read drop themselves rather than the whole ledger, and both arms go through one mint point so they cannot drift |
300
- | `scripts/run-cost-reconcile-projection-test.mjs` | The **end-of-run cost reconciliation** reaching consumers at all. The engine splits a run's spend on the wire — the task's own cost, which deliberately excludes delegated sub-agents, the delegated total itself, and the within-task compaction subtotal that sits inside the own figure — and states two reconciliation identities for them. The package used to project none of it, so a cost view could only ever see one number and under-reported both delegated and compaction spend. Both structures are now projected onto the result as superset fields in the wire's integer micro-currency unit, read key by key, with unreadable keys dropped individually, an entirely unreadable structure omitted rather than emitted empty, and unknown categories passed through since the vocabulary belongs upstream. The delegated cost stays **absent when it was never priced**, never a fabricated zero. The same reader also feeds a terminal chrome arm carrying the three parts plus the reconciled total, so the two faces can never compute different answers; the reconciled total is minted only when both sides are known, and otherwise a discriminator bit says which side is unknown. **The reference field for total cost keeps its meaning** — it remains the task's own spend and the delegated total is not folded into it — because that is a shape the wider ecosystem reads; the reconciled figure is offered beside it, not in place of it. A frame that carries no stats emits no arm at all, and the existing rule that in-stream per-turn usage is not published for sub-flows is pinned unchanged, since delegated spend arrives once, at the end |
363
+ | `scripts/run-cost-reconcile-projection-test.mjs` | The **end-of-run cost reconciliation** reaching consumers at all. The engine splits a run's spend on the wire — the task's own cost, which deliberately excludes delegated sub-agents, the delegated total itself, and the within-task compaction subtotal that sits inside the own figure — and states two reconciliation identities for them. The package used to project none of it, so a cost view could only ever see one number and under-reported both delegated and compaction spend. Both structures are now projected onto the result as superset fields in the wire's integer micro-currency unit, read key by key, with unreadable keys dropped individually, an entirely unreadable structure omitted rather than emitted empty, and unknown categories passed through since the vocabulary belongs upstream. The delegated cost stays **absent when it was never priced**, never a fabricated zero. The same reader also feeds a terminal chrome arm carrying the three parts plus the reconciled total, so the two faces can never compute different answers; the reconciled total is minted only when both sides are known, and otherwise a discriminator bit says which side is unknown. **The reference field for total cost keeps its meaning** — it remains the task's own spend and the delegated total is not folded into it — because that is a shape the wider ecosystem reads; the reconciled figure is offered beside it, not in place of it. A frame that carries no stats emits no arm at all, and the existing rule that in-stream per-turn usage is not published for sub-flows is pinned unchanged, since delegated spend arrives once, at the end. The bit that says those figures are a lower bound is **per stream**, not per context: the emit context belongs to the caller and may be reused across streams, so a gap observed on one run is no evidence at all about the next one — the observation is held for the duration of one stream and handed to both projection faces by value, and the guard drives a reused context both sequentially and concurrently to prove neither direction leaks |
301
364
  | `scripts/run-task-progress-terminal-projection-test.mjs` | The one tick that says a delegated child **finished**. The engine fires exactly one final beat carrying a terminal face, and says in the same breath why it exists — so a consumer sees the row finish instead of watching it vanish after the last running beat — but the package's projection whitelist had no seat for that field and its adapter still carried the older premise in a comment, so the terminal beat arrived byte-identical to another running one: the panel row stayed up waiting for a defensive sweep (which only ever settles rows bound to a card still open this turn) or for a separate notification frame. The status now rides through as an **open set** with the vocabulary left upstream, while the question *which words are terminal* is answered by a closed pair on the adapter side — an unrecognised new word takes the running path, because guessing it terminal ends a row that is still working whereas one extra running beat merely renders late. A terminal beat settles the row directly under the lane proof its binding gives it (not the main lane a notification would use, and not by card id, since the engine is naming a child rather than closing a card), freezes the inline group-row twin in the same beat so a later sweep cannot reset the real tool count, clears the session-resident ledger, and fires the stop hook only for a child whose start really fired. It does not mark the row live or emit a second progress beat, and it shares the settled-row ledger with the other two settle legs so a replay or a double-delivery cannot produce a second end. Three things are pinned **unchanged**: a running beat, an absent status (older engines never send the field, and reading absence as terminal would make every child row disappear on its first beat), and the workflow lane gate, which still runs before any of this |
302
365
  | `scripts/run-assistant-arm-identity-test.mjs` | The identity keys on an assistant row, and an explicit account of the two that are **deliberately not** there. What the renderer received was a bare role-and-content object, so a dozen consumer sites downstream were each estimating what the message envelope should have told them. The id is taken from the engine's own event id rather than minted locally, because it has to be **the same value** on the live leg and on a durable replay — a freshly minted one would make a replayed message look new to a host's dedup and to rewind — and when the wire carries none the key is simply absent rather than filled with a random stand-in wearing an identity it does not have; it is also kept distinct from the envelope's own local render key, which is a different identity. The model name comes from what the host pinned when it opened the stream (the request was the host's to build) and is never guessed, since a wrong model name is worse than none once a billing or capability face looks it up. Usage and stop reason are **not** minted on this arm, and the reason is frame order rather than effort: content arms arrive before the turn's closing frame, so at the moment the arm is emitted the engine has not yet said what the round cost — anything put there would be an estimate, which is the very thing this work exists to remove — and synthesising a follow-up assistant update when the real figure lands is also refused, because that shape does not exist upstream and would place a message in the transcript the engine never sent. Their real values leave through the turn's own neutral arm as two superset keys, the usage one reusing the **same single mint point** the footer rollup already folds so the two faces cannot diverge, and the stop reason passed through verbatim as an open set — the machine signal for *was this turn cut short*, previously blind on both the stream and the trace. The existing behaviours beside them are pinned too: no arm at all when usage is wholly absent, and the sub-flow cut-out that keeps a child's turn from driving the leader's face |
366
+ | `scripts/run-gate-negative-controls-test.mjs` | Whether the registry-shaped guards among the 74 suites above actually turn red when the material they check really breaks — a census had found 16 of them clean enough to rehearse safely (closed sets, mirrors, baselines, floors, a type-shape ratchet) without touching any judgement code. Each is exercised by tampering a disk copy of the real material, spawning the guard's own unmodified script, asserting it exits non-zero and names the disease, then restoring the file byte-for-byte. Seven guards of the same shape and 51 behaviour/projection suites are catalogued rather than rehearsed this round — see `docs/GATE-NEGATIVE-CONTROLS.md` for the full table, the reasons, and a one-minute manual replay recipe for each blind one. The suite cross-checks its own case count against that document's row counts in both directions, so a case quietly dropped from the array without the document following is itself an undeclared blind guard. The backup that makes the restore possible is taken by **exclusive create**: checking for it and then copying are otherwise two steps, and two instances can pass the check together — the later one overwrites the only clean copy with material the earlier one has already tampered, and the rehearsal that promises to leave no trace leaves a permanently corrupted file instead. That interleaving is rehearsed too, in a throwaway directory of its own |
303
367
 
304
368
  Each suite carries a floor that only moves up — a refactor that stops executing a group of
305
369
  assertions is a failure, not a quieter pass. Guards anchor on the **installed artefact's content**
@@ -717,11 +717,22 @@ const streamEventArm = function* (m, { text }) {
717
717
  };
718
718
  const turnUsageArm = function* (m, { text, flags }) {
719
719
  const ot = typeof m.outputTokens === 'number' ? m.outputTokens : undefined;
720
+ // 🔴 0.68.0(异源对抗复审轮五实抓)——**两件事分家**:
721
+ // · 「这一轮结束了」⇒ 无条件 drain + 压下一 model round 的基线(TTFT 那一拍);
722
+ // · 「这一轮花了多少」⇒ 只有**真有数字**时才发 end 度量。
723
+ // #711 之后,缺账轮这一拍**没有数字可发**(占位零不许当读数)。修前两件事合在
724
+ // `emitTurnUsageEnd` 一个动作里 ⇒ 过滤掉数字就连**模型轮边界**也一起过滤掉了:下一轮不再发
725
+ // `response_metrics{start}`,它的用量按**上一轮**的基线对账(整链实测:首轮 400 字符缺账、
726
+ // 次轮 4 字符 + 50 tokens,计数从 150 掉到 101,下一轮的 TTFT 记录一并丢失)。
727
+ yield* text.drainLive();
720
728
  if (ot !== undefined) {
721
- yield* text.drainLive();
722
729
  // ⟨帧序耦合 7/7⟩ 这一拍置 endEmitted,后到的 result 臂据它决定要不要补发 end 度量。
723
730
  yield* flags.emitTurnUsageEnd(ot);
724
731
  }
732
+ else {
733
+ // 🔴 不发假零度量,但边界照压(`endEmitted` **不置** —— 终帧若真有一笔账,补发腿仍该补)。
734
+ flags.noteModelRoundBoundary();
735
+ }
725
736
  };
726
737
  // ══ M0 编排半场 —— result(收口三形之一;与 D7/D8 的次序差异见矩阵 §2.4,不可归一)═══════════════
727
738
  /**
@@ -749,7 +760,21 @@ const resultArm = function* (m, { ctx, idOf, text, cards, panel, flags }) {
749
760
  yield* text.takeThinking();
750
761
  yield* flags.emitResultEndIfNeeded(() => {
751
762
  const u = m.usage;
752
- return typeof u?.outputTokens === 'number' ? u.outputTokens : undefined;
763
+ const ot = typeof u?.outputTokens === 'number' && Number.isFinite(u.outputTokens) ? u.outputTokens : undefined;
764
+ // ── 🔴 0.68.0(#711 同形的**第四处**,异源对抗复审轮二实抓)────────────────────────────────
765
+ // 这是终帧的**补发腿**:`turn_usage` 那一拍没发过 end 度量时,由它拿终帧的 usage 补一发。
766
+ // 而 core 7.17.0 之后,一轮没量出账的 run,它的终帧 usage 是**占位全零**(同帧的
767
+ // `_sema_usage_lower_bound` 才是那句「这笔账不知道」)⇒ 修前这里会补发一条**没有任何判别位**
768
+ // 的 `response_metrics{phase:'end', outputTokens: 0}`,而本臂的宿主义务逐字是「驱动
769
+ // responseLength reducer(ttft/对账)」—— 一条精确零会被当成一次真的「这一轮吐了 0 个 token」。
770
+ // 同一条判据在本包一共五处:公面读器 `turnEndUsage`(`eventToSdkMessage.ts`)/
771
+ // `turn_usage.outputTokens` / 子代分表 `cacheReadTokens` / `RunStreamHandle`(三处在
772
+ // `adapter/runStream.ts`)/ **本处**。前四处收口了而这条**消费链的末端**没跟上 = 过滤没贯穿到底。
773
+ // ⇒ 同一条判据:**下界位在场且读数是 `0` ⇒ 那是占位,不补发**(`outputTokens` 读不出时本来
774
+ // 就不发,这一形沿用既有的那条路);**非零照发** —— 占位恒为 0,非零必是真的量到过,
775
+ // 连带丢掉它会让一条有产出的缺账轮在对账面上凭空少一笔。
776
+ const lowerBound = m._sema_usage_lower_bound === true;
777
+ return lowerBound && ot === 0 ? undefined : ot;
753
778
  });
754
779
  yield* text.takeAnswerSegment();
755
780
  // 具名成 terminalText:M1 收编后 `text` 是流合并器的名字,原来的同名局部会遮蔽它。
@@ -27,6 +27,20 @@ export interface TurnFlags {
27
27
  onFrameBeforeDispatch(m: Frame): Generator<AdapterOutput>;
28
28
  /** A6 `turn_usage`:发 end 度量 + 压下一轮基线。 */
29
29
  emitTurnUsageEnd(outputTokens: number): Generator<AdapterOutput>;
30
+ /**
31
+ * A6 `turn_usage` 的**无数字那一支**(0.68.0;异源对抗复审轮五实抓)——
32
+ * **只压下一轮基线,不发任何帧**。
33
+ *
34
+ * 🔴 为什么必须把它拆出来:`emitTurnUsageEnd` 一直在做**两件事** —— 发 end 度量 **与** 把
35
+ * `messageStartEmitted` 复位(下一个 model round 重新压 TTFT 基线)。而 #711 之后,一轮没量出账时
36
+ * 这一拍**没有数字可发**(占位零不许当读数,见 `adapter/runStream.ts` 的那条判据)⇒ 若整条跳过
37
+ * `emitTurnUsageEnd`,**连模型轮边界也一起跳过了**:下一轮不再发 `response_metrics{start}`,
38
+ * 它的用量于是按**上一轮**的基线对账(实测:首轮 400 字符缺账、次轮 4 字符 + 50 tokens,
39
+ * 计数从 150 掉到 101,且下一轮的 TTFT 记录一并丢失)。
40
+ * 🔴 判据分家:**「这一轮结束了」是一件事,「这一轮花了多少」是另一件事** —— 前者与有没有数字
41
+ * 无关,后者才受占位零那条判据管。合成一个动作,过滤数字就会顺手把边界也过滤掉。
42
+ */
43
+ noteModelRoundBoundary(): void;
30
44
  /** A7 `result`:只有还没发过 end 且这帧真带了 outputTokens 时才补发(`!endEmitted` 门)。 */
31
45
  /** `readOutputTokens` 惰性:只在真的要发 end 度量时才读 usage(与拆分前「`!endEmitted`
32
46
  * 门内才读 `m.usage`」逐字同语义 —— 对抗复审 #3 收货修)。 */
@@ -44,7 +44,10 @@ export function createTurnFlags(ctx, turnStartAt) {
44
44
  *emitTurnUsageEnd(outputTokens) {
45
45
  yield chrome({ kind: 'response_metrics', laneProof: MAIN, phase: 'end', outputTokens });
46
46
  endEmitted = true;
47
- messageStartEmitted = false; // 下一 model round 重新压基线
47
+ messageStartEmitted = false; // 下一 model round 重新压基线(与 noteModelRoundBoundary 同一件事)
48
+ },
49
+ noteModelRoundBoundary: () => {
50
+ messageStartEmitted = false;
48
51
  },
49
52
  *emitResultEndIfNeeded(readOutputTokens) {
50
53
  if (endEmitted)
@@ -85,11 +85,33 @@ export declare const CLAIM_HELD_STATES: readonly string[];
85
85
  * 判成「无公共属性」而拒绝赋值)。三处 `unknown` 是边界职责形(本包不拥有 run 记录的类型,
86
86
  * 消费点 `readStatus` 就是窄化动作本身;typeshape 门 RATCHET 逐条登记)。
87
87
  */
88
+ /**
89
+ * `GET /v1/runs/:id` 的行,**只到本层读得到的那几格**(0.68.0 / L-230)。
90
+ *
91
+ * 🔴 具名而不是内联匿名形:三格都是 wire 值(`unknown`),而端要按它们分支 —— 一个有名字的形让
92
+ * 宿主的适配层能声明自己返回什么,也让 typeshape 门的 B4 棘轮不为这一处升一格。
93
+ * 🔴 **三格全可选**:老引擎只发 `status`;`heldBy` / `cancelRequested` 是 server 7.73.0 S-122 P-45
94
+ * 的 additive 读面,缺席一律走既有路(见 `readClaimHolder` / `readCancelRequested` 的三态)。
95
+ * 🔴 行上其余键本层**一个都不读**(它们仍在 wire 上,只是这一层不需要);要读的端自己窄化。
96
+ */
97
+ export interface DurableRunRecordView {
98
+ /** run 行状态词(开集;本层按 {@link CLAIM_RELEASED_STATES} / {@link CLAIM_HELD_STATES} 白名单判)。 */
99
+ readonly status?: unknown;
100
+ /** 谁占着这个会话:非空串 = 那条 run 还占着;**`null` = 一句正面事实(交出来了)**;整键缺席 = 老引擎。 */
101
+ readonly heldBy?: unknown;
102
+ /** 这条 run 上**有没有受理过一次取消请求**(不是「已取消」)。整键缺席 = 老引擎不发。 */
103
+ readonly cancelRequested?: unknown;
104
+ }
88
105
  export interface DurableRunVerbs {
89
106
  events: unknown;
90
- get?: (taskId: string, opts?: DurableRunCallOpts) => Promise<{
91
- status?: unknown;
92
- } | null>;
107
+ /**
108
+ * `GET /v1/runs/:id`。
109
+ * 🔴 0.68.0(L-230;server 7.73.0 S-122 P-45 读面):型面补两个**可选**位 —— `heldBy` 与
110
+ * `cancelRequested`。两者都是 **additive**:老引擎不发,读点按缺席走既有路
111
+ * ({@link CLAIM_RELEASED_STATES}),所以拓型不是「要求宿主升级」,是「宿主已经透传的那两位
112
+ * 从此有读点」(L-230 ⑤ 逐字:缺的是读点不是通道)。
113
+ */
114
+ get?: (taskId: string, opts?: DurableRunCallOpts) => Promise<DurableRunRecordView | null>;
93
115
  cancel?: (taskId: string, opts?: DurableRunCallOpts) => Promise<unknown>;
94
116
  /** 三选卡的 steer 腿(`POST /v1/runs/:id/steer`)。🔴 **AT-MOST-ONCE**:steer 非幂等,契约逐字
95
117
  * 「it is NEVER retried」(本包 `controlRouter` 同律)—— 本层调它恰一次,失败如实上屏由人重发。
@@ -396,9 +418,11 @@ export type SelfHealOutcome =
396
418
  taskId: string;
397
419
  waitedMs: number;
398
420
  aborted: boolean;
399
- /** 🔴 最近一次探测**读到了一个确实代表持锁的词**吗(判据 = {@link CLAIM_HELD_STATES},不是
400
- * 「lastStatus 非空」—— 表外的新状态词非空却证明不了持锁)。false = 读不出来 / 读到一个不认识
401
- * 的词 ⇒ 文案只许说「确认不了」,绝不说「它还占着」。 */
421
+ /** 🔴 最近一次探测**拿到了持锁的正面证据**吗。判据**两条取并**(0.68.0 / L-230):
422
+ * ① 引擎直说的 `heldBy`(非空串 = 有一条 run 的名字在锁上);② 回落到状态词白名单
423
+ * ({@link CLAIM_HELD_STATES},老引擎不发 `heldBy` 时的既有路)。**不是**「lastStatus 非空」——
424
+ * 表外的新状态词非空却证明不了持锁。false = 两条都读不出 / 读到一个不认识的词 ⇒ 文案只许说
425
+ * 「确认不了」,绝不说「它还占着」。 */
402
426
  confirmedHeld: boolean;
403
427
  }
404
428
  /**
@@ -489,6 +513,16 @@ export type SelfHealSubmissionDisposition = 'held-for-decision' | 'resending' |
489
513
  * 「已解锁」证不出,自动重投可能撞回仍锁着的会话。
490
514
  */
491
515
  export declare function selfHealSubmissionDisposition(outcome: SelfHealOutcome): SelfHealSubmissionDisposition;
516
+ /**
517
+ * cancel 回执(202 体)/ run 行上的 `cancelRequested`(0.68.0 / L-230;server 7.73.0 S-122 P-45)。
518
+ *
519
+ * 🔴 **三态**同上:`true` / `false` 都是真读数,**读不出 ⇒ `undefined`**(老引擎不发这一位)。
520
+ * 绝不折 `false` —— 「没请求过取消」与「不知道有没有请求过」对下一步的含义不同:前者可以放心
521
+ * 再发一次 cancel,后者再发就可能是第二枪。
522
+ * 🔴 它**不是**「已经取消了」:上游逐字是 *requested*,一次**已受理的请求**,不是终局。
523
+ * 会话有没有交出来仍然只由 {@link readClaimHolder} 与 {@link CLAIM_RELEASED_STATES} 回答。
524
+ */
525
+ export declare function readCancelRequested(body: unknown): boolean | undefined;
492
526
  /**
493
527
  * 一次**至多一次**(non-idempotent)POST 失败之后:到底是「服务端明确拒了」还是「不知道有没有
494
528
  * 落地」。steer 与 cancel 两条腿共用这一把尺 —— 它们同属「这一枪不能盲发第二次」的族,而两类
@@ -534,6 +568,19 @@ export interface ClaimReleaseVerdict {
534
568
  * 已经不知道的事,恰好塌回本批要修的那一格。
535
569
  */
536
570
  lastStatus: string | null;
571
+ /**
572
+ * **最近一次**探测读到的 `heldBy` 三态(0.68.0 / L-230;见 `readClaimHolder`)。
573
+ *
574
+ * · `'held'` —— 引擎**直说**那条 run 还占着这个会话(比按状态词推断强得多的证据);
575
+ * · `'released'` —— 引擎**直说**已经交出来了;
576
+ * · `'unknown'` —— 这一位读不出(老引擎不发 / 本发没有回答)。
577
+ *
578
+ * 🔴 与 `lastStatus` 同律:**只记最近那一次有回答的探测**;截止掐断(连回答都没有)不改写它。
579
+ * 🔴 文案层的用法:`'held'` 是「那条 run 还占着这个会话」这句话的**直接证据**,比
580
+ * {@link CLAIM_HELD_STATES} 那条按词推断的路更强;两条都读不出时收口成「确认不了」,
581
+ * 绝不替引擎下一个证不出的断言。
582
+ */
583
+ lastHolder: 'held' | 'released' | 'unknown';
537
584
  }
538
585
  /** {@link waitForClaimRelease} 的注入口(时钟/等待/中止/窗口全部可注入 —— 门要能在零墙钟下判)。 */
539
586
  export interface ClaimReleaseWaitDeps {