@bitkyc08/opencodex 2.52.0-preview.20260912 → 2.53.0-preview.20260913

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/gui/dist/assets/index-BBOZWGB6.css +1 -0
  2. package/gui/dist/assets/index-D7ynYo2K.js +128 -0
  3. package/gui/dist/index.html +2 -2
  4. package/native/remote-workspace-helper/Cargo.lock +130 -0
  5. package/native/remote-workspace-helper/Cargo.toml +24 -0
  6. package/native/remote-workspace-helper/src/main.rs +49 -0
  7. package/native/remote-workspace-helper/src/protocol.rs +246 -0
  8. package/native/remote-workspace-helper/src/sandbox/macos.rs +19 -0
  9. package/native/remote-workspace-helper/src/sandbox/mod.rs +77 -0
  10. package/native/remote-workspace-helper/src/sandbox/windows.rs +15 -0
  11. package/package.json +6 -1
  12. package/src/adapters/anthropic-image-normalize.ts +30 -2
  13. package/src/adapters/anthropic.ts +1 -1
  14. package/src/adapters/base.ts +8 -2
  15. package/src/adapters/cursor/cursor-errors.ts +12 -0
  16. package/src/adapters/cursor/thread-continuity.ts +93 -0
  17. package/src/adapters/cursor.ts +104 -73
  18. package/src/adapters/devin/cloud-direct/chat.ts +312 -23
  19. package/src/adapters/devin/cloud-direct/metadata.ts +31 -2
  20. package/src/adapters/devin/live-models.ts +70 -3
  21. package/src/adapters/devin.ts +281 -21
  22. package/src/adapters/google-wire-compiler.ts +14 -6
  23. package/src/adapters/google.ts +22 -8
  24. package/src/adapters/kiro/adapter.ts +316 -0
  25. package/src/adapters/kiro/conversation.ts +136 -0
  26. package/src/adapters/kiro/payload.ts +432 -0
  27. package/src/adapters/kiro/reasoning.ts +56 -0
  28. package/src/adapters/kiro/stream.ts +1153 -0
  29. package/src/adapters/kiro/usage.ts +223 -0
  30. package/src/adapters/kiro/wire.ts +76 -0
  31. package/src/adapters/kiro.ts +8 -2319
  32. package/src/adapters/mimo-free.ts +1 -1
  33. package/src/adapters/openai-chat-images.ts +101 -0
  34. package/src/adapters/openai-chat.ts +201 -181
  35. package/src/adapters/openai-responses.ts +92 -224
  36. package/src/adapters/registry.ts +0 -7
  37. package/src/adapters/run-turn-queue.ts +13 -6
  38. package/src/bridge.ts +14 -15
  39. package/src/chat/inbound.ts +29 -4
  40. package/src/chat/outbound.ts +145 -107
  41. package/src/claude/desktop-profile.ts +4 -6
  42. package/src/cli/account-api.ts +14 -0
  43. package/src/cli/account-extended.ts +1 -1
  44. package/src/cli/account-history.ts +60 -0
  45. package/src/cli/account-main.ts +80 -0
  46. package/src/cli/account.ts +11 -3
  47. package/src/cli/capabilities.ts +113 -0
  48. package/src/cli/catalog.ts +109 -0
  49. package/src/cli/dispatch.ts +9 -0
  50. package/src/cli/help.ts +2 -0
  51. package/src/cli/index.ts +2 -2
  52. package/src/cli/observe.ts +28 -1
  53. package/src/cli/opencode.ts +42 -8
  54. package/src/cli/provider-runtime.ts +11 -1
  55. package/src/cli/provider.ts +22 -2
  56. package/src/cli/registry.ts +21 -0
  57. package/src/cli/remote-workspace.ts +154 -0
  58. package/src/cli/status.ts +39 -7
  59. package/src/cli/usage-report.ts +14 -2
  60. package/src/client/hub-client.ts +34 -0
  61. package/src/client/hub-state.ts +9 -1
  62. package/src/codex/account-store.ts +78 -0
  63. package/src/codex/auth-api.ts +81 -54
  64. package/src/codex/auth-context.ts +45 -16
  65. package/src/codex/catalog/effort.ts +1 -1
  66. package/src/codex/catalog/metadata.ts +3 -6
  67. package/src/codex/catalog/native-models.ts +4 -4
  68. package/src/codex/catalog/parsing.ts +2 -20
  69. package/src/codex/catalog/provider-fetch.ts +10 -1
  70. package/src/codex/catalog/remote.ts +233 -0
  71. package/src/codex/catalog/sync.ts +403 -35
  72. package/src/codex/convergence.ts +1 -1
  73. package/src/codex/history-manifest.ts +36 -0
  74. package/src/codex/history-provider.ts +32 -5
  75. package/src/codex/inject.ts +9 -0
  76. package/src/codex/main-account.ts +113 -0
  77. package/src/codex/main-device-reauth-api.ts +89 -0
  78. package/src/codex/main-device-reauth.ts +217 -0
  79. package/src/codex/native-residue.ts +9 -2
  80. package/src/codex/quota-auto-refresh.ts +3 -2
  81. package/src/codex/quota-capacity.ts +98 -0
  82. package/src/codex/quota-history.ts +160 -0
  83. package/src/codex/quota-types.ts +8 -0
  84. package/src/codex/quota.ts +118 -91
  85. package/src/codex/refresh.ts +2 -1
  86. package/src/codex/routing.ts +90 -17
  87. package/src/codex/sync.ts +33 -4
  88. package/src/combos/request.ts +19 -1
  89. package/src/config/multi-agent-surface.ts +61 -0
  90. package/src/config/provider-validation.ts +176 -0
  91. package/src/config.ts +213 -11
  92. package/src/generated/compatibility-version.json +436 -168
  93. package/src/images/loop.ts +119 -36
  94. package/src/lib/admission.ts +12 -6
  95. package/src/lib/redact.ts +7 -0
  96. package/src/lib/translator-budget.ts +4 -3
  97. package/src/lib/windows-atomic-replace.ts +1 -0
  98. package/src/lib/windows-elevation.ts +1 -1
  99. package/src/oauth/chatgpt-device.ts +62 -5
  100. package/src/oauth/devin/cli-import.ts +130 -0
  101. package/src/oauth/devin.ts +63 -8
  102. package/src/oauth/index.ts +29 -14
  103. package/src/oauth/kiro.ts +18 -6
  104. package/src/oauth/login-cli.ts +9 -1
  105. package/src/oauth/meta-muse-device.ts +464 -0
  106. package/src/oauth/meta-muse.ts +123 -32
  107. package/src/oauth/pool-kernel.ts +9 -0
  108. package/src/oauth/pool-settings-capability.ts +2 -2
  109. package/src/oauth/store.ts +57 -0
  110. package/src/oauth/types.ts +31 -0
  111. package/src/providers/derive.ts +13 -3
  112. package/src/providers/devin-cli-authmode-migration.ts +57 -35
  113. package/src/providers/devin-provider-merge-migration.ts +240 -0
  114. package/src/providers/muse-key-quota.ts +117 -0
  115. package/src/providers/muse-subscription-usage.ts +14 -2
  116. package/src/providers/openai-sidecar.ts +25 -3
  117. package/src/providers/opencode-zen-rate-limit.ts +58 -0
  118. package/src/providers/provider-id-rewrite.ts +20 -5
  119. package/src/providers/quota-types.ts +12 -0
  120. package/src/providers/quota.ts +143 -102
  121. package/src/providers/reasoning-metadata.ts +543 -0
  122. package/src/providers/registry.ts +80 -49
  123. package/src/reasoning-effort.ts +26 -2
  124. package/src/remote/hub-usage.ts +32 -0
  125. package/src/remote-control/index.ts +192 -41
  126. package/src/remote-control/workspace-activation.ts +9 -0
  127. package/src/remote-control/workspace-agent-connection.ts +366 -0
  128. package/src/remote-control/workspace-claude-runtime.ts +243 -0
  129. package/src/remote-control/workspace-codex-runtime.ts +531 -0
  130. package/src/remote-control/workspace-codex-sandbox.ts +115 -0
  131. package/src/remote-control/workspace-command-runner.ts +748 -0
  132. package/src/remote-control/workspace-coordinator.ts +231 -0
  133. package/src/remote-control/workspace-device.ts +585 -0
  134. package/src/remote-control/workspace-executable.ts +43 -0
  135. package/src/remote-control/workspace-executor.ts +397 -0
  136. package/src/remote-control/workspace-hub.ts +519 -0
  137. package/src/remote-control/workspace-pi-runtime.ts +382 -0
  138. package/src/remote-control/workspace-process.ts +129 -0
  139. package/src/remote-control/workspace-rpc.ts +304 -0
  140. package/src/remote-control/workspace-runtime.ts +60 -0
  141. package/src/remote-control/workspace-secret-store.ts +39 -0
  142. package/src/remote-control/workspace-sessions.ts +799 -0
  143. package/src/remote-control/workspace-tool-bridge.ts +192 -0
  144. package/src/responses/code-mode-helper-compat.ts +22 -3
  145. package/src/responses/hosted-tool-policy.ts +0 -1
  146. package/src/responses/muse-tool-name-alias.ts +379 -0
  147. package/src/responses/plaintext-v2-agent-messages.ts +902 -0
  148. package/src/router.ts +7 -0
  149. package/src/routing/compatibility/behavior.ts +0 -1
  150. package/src/server/audio-client.ts +64 -0
  151. package/src/server/audio-dictation.ts +91 -0
  152. package/src/server/audio-live.ts +185 -0
  153. package/src/server/audio-transcriptions.ts +183 -0
  154. package/src/server/audio-upstream.ts +153 -0
  155. package/src/server/auth-cors.ts +61 -2
  156. package/src/server/chat-completions.ts +1 -1
  157. package/src/server/chat-native-sse.ts +92 -48
  158. package/src/server/chat-native.ts +37 -15
  159. package/src/server/hub-usage.ts +57 -0
  160. package/src/server/images.ts +4 -0
  161. package/src/server/index.ts +722 -57
  162. package/src/server/lifecycle.ts +5 -6
  163. package/src/server/live-call-bindings.ts +60 -0
  164. package/src/server/live.ts +12 -1
  165. package/src/server/management/agent-settings-routes.ts +25 -4
  166. package/src/server/management/api-access.ts +37 -0
  167. package/src/server/management/api-key-usage.ts +7 -2
  168. package/src/server/management/config-routes.ts +1 -18
  169. package/src/server/management/context.ts +15 -0
  170. package/src/server/management/logs-usage-routes.ts +2 -0
  171. package/src/server/management/oauth-account-routes.ts +39 -12
  172. package/src/server/management/provider-routes.ts +125 -2
  173. package/src/server/management/remote-workspace-routes.ts +140 -0
  174. package/src/server/management/route-registry.ts +15 -0
  175. package/src/server/management/usage-aggregate-cache.ts +14 -15
  176. package/src/server/management/usage-summary-cache.ts +2 -0
  177. package/src/server/management-api.ts +23 -0
  178. package/src/server/ports.ts +17 -0
  179. package/src/server/relay-eager.ts +4 -1
  180. package/src/server/relay.ts +70 -10
  181. package/src/server/request-decompress.ts +6 -3
  182. package/src/server/responses/agent-task-recovery.ts +25 -32
  183. package/src/server/responses/codex-auth-error.ts +11 -0
  184. package/src/server/responses/codex-ws-exchange.ts +52 -3
  185. package/src/server/responses/codex-ws-wire.ts +55 -0
  186. package/src/server/responses/compact.ts +9 -1
  187. package/src/server/responses/core.ts +337 -73
  188. package/src/server/responses/encrypted-payload.ts +45 -2
  189. package/src/server/responses/ws-upstream.ts +4 -1
  190. package/src/server/responses-self-named-namespace-scrub.ts +1 -3
  191. package/src/server/responses-undeclared-tool-guard.ts +1 -1
  192. package/src/server/search.ts +3 -0
  193. package/src/server/sse-payload-rewrite.ts +136 -51
  194. package/src/server/ws-bridge.ts +35 -1
  195. package/src/service/cli.ts +372 -0
  196. package/src/service/diagnostics.ts +340 -0
  197. package/src/service/guards.ts +303 -0
  198. package/src/service/health.ts +222 -0
  199. package/src/service/launchd.ts +853 -0
  200. package/src/service/orchestration.ts +617 -0
  201. package/src/service/repair.ts +334 -0
  202. package/src/service/state.ts +363 -0
  203. package/src/service/systemd.ts +229 -0
  204. package/src/service/windows-ops.ts +690 -0
  205. package/src/service/windows-scheduler.ts +769 -0
  206. package/src/service/windows-taskxml.ts +613 -0
  207. package/src/service.ts +22 -5550
  208. package/src/storage/cleanup/db.ts +258 -0
  209. package/src/storage/cleanup/execute.ts +358 -0
  210. package/src/storage/cleanup/paths.ts +189 -0
  211. package/src/storage/cleanup/pending.ts +140 -0
  212. package/src/storage/cleanup/preview.ts +292 -0
  213. package/src/storage/cleanup/reconcile.ts +347 -0
  214. package/src/storage/cleanup/restore.ts +932 -0
  215. package/src/storage/cleanup/satellite.ts +474 -0
  216. package/src/storage/cleanup/staging.ts +129 -0
  217. package/src/storage/cleanup/types.ts +98 -0
  218. package/src/storage/cleanup.ts +49 -3127
  219. package/src/types/accounts.ts +2 -0
  220. package/src/types/config.ts +13 -12
  221. package/src/types/provider.ts +37 -0
  222. package/src/types/request.ts +2 -0
  223. package/src/types/tools.ts +17 -5
  224. package/src/types.ts +1 -0
  225. package/src/usage/expected-prices.ts +127 -0
  226. package/src/usage/log.ts +58 -1
  227. package/src/vision/eligibility.ts +13 -2
  228. package/src/web-search/loop.ts +56 -3
  229. package/gui/dist/assets/index-D_t6sCWs.js +0 -115
  230. package/gui/dist/assets/index-EdoPnm9_.css +0 -1
  231. package/src/adapters/devin-cli/acp.ts +0 -204
  232. package/src/adapters/devin-cli/adapter.ts +0 -345
  233. package/src/adapters/devin-cli/binary.ts +0 -69
  234. package/src/adapters/devin-cli/models.ts +0 -57
  235. package/src/oauth/devin-cli.ts +0 -149
  236. package/src/server/responses-reasoning-summary-rewrite.ts +0 -178
@@ -1,2319 +1,8 @@
1
- import { decodeEventStream } from "../lib/eventstream-decoder";
2
- import { estimateTokens } from "../lib/token-estimate";
3
- import { debugProviderDiagnostic } from "../lib/debug";
4
- import { isDebugEnabled } from "../lib/debug-settings";
5
- import { resolveKiroApiRegion, resolveKiroRequestProfile } from "../oauth/kiro";
6
- import { KIRO_MODEL_CONTEXT_WINDOWS, normalizeKiroModelId } from "../providers/kiro-models";
7
- import { modelRecordValue } from "../reasoning-effort";
8
- import { parseKiroEvent } from "./kiro-events";
9
- import { calibrateKiroEstimate, recordKiroCalibration, rekeyKiroCalibration } from "./kiro-calibration";
10
- import {
11
- classifyKiroEventError,
12
- classifyKiroHttpError,
13
- classifyKiroStreamError,
14
- safeKiroErrorMessage,
15
- safeKiroHttpErrorMessage,
16
- type KiroErrorClassification,
17
- } from "./kiro-errors";
18
- import { KiroThinkingParser } from "./kiro-thinking";
19
- import { isCompleteKiroToolInput, kiroTruncationErrorMessage } from "./kiro-truncation";
20
- import { createKiroToolNameRegistry, fallbackToolUseId, fingerprint, invocationId, isValidKiroConversationId, mapModelId, normalizeToolId, osTag, stableConversationId } from "./kiro-wire";
21
- import { namespacedToolName } from "../types";
22
- import {
23
- isTranslatorBudgetExceededError,
24
- releaseTranslatedEvent,
25
- retainTranslatedEvent,
26
- type TranslatorBudget,
27
- } from "../lib/translator-budget";
28
- import type {
29
- AdapterEvent,
30
- OcxAssistantMessage,
31
- OcxContentPart,
32
- OcxMessage,
33
- OcxParsedRequest,
34
- OcxProviderConfig,
35
- OcxTextContent,
36
- OcxToolCall,
37
- OcxToolResultMessage,
38
- OcxTool,
39
- OcxUsage,
40
- } from "../types";
41
- import { hasRecordedTrailingDeliveredFinalAnswer } from "../responses/turn-termination";
42
- import type { ProviderAdapter } from "./base";
43
- import type { AdapterFetchContext, AdapterRequest } from "./base";
44
- import { extractKiroImages, normalizeKiroImages, type KiroImage } from "./kiro-images";
45
- import { sniffImageDimensions } from "./anthropic-image-guard";
46
- import { fetchKiroWithRetry, noteKiroTransientThrottle } from "./kiro-retry";
47
- import { convertKiroToolContext } from "./kiro-tools";
48
- import { EMPTY_EXEC_OUTPUT_MESSAGE, annotateCodeModeHostFailure, normalizeEmptyExecToolResultText } from "./exec-tool-result-normalize";
49
- import { identifyRoutedModel } from "./identity";
50
- import { buildNonOpenAIToolCatalogNudgeFromNames, isBareShellBridgeTool, isCodexCodeModeExecTool } from "./tool-catalog-nudge";
51
- import {
52
- KIRO_ANSWER_DELIVERED_MESSAGE,
53
- KIRO_COMPLETION_INSTRUCTIONS,
54
- KIRO_COMPLETION_RETRY_MESSAGE,
55
- KIRO_COMPLETION_TOOL_NAME,
56
- KIRO_CONTINUATION_MESSAGE,
57
- KIRO_EMPTY_TOOL_RESULT_MESSAGE,
58
- KIRO_TOOL_RESULT_CARRIER_MESSAGE,
59
- MAX_KIRO_INJECTED_INSTRUCTION_CHARS,
60
- type KiroCompletionMode,
61
- } from "./kiro-constants";
62
-
63
- const AMZ_TARGET = "AmazonCodeWhispererStreamingService.GenerateAssistantResponse";
64
- const SDK_VERSION = "1.0.27";
65
- const NODE_VERSION = "22.21.1";
66
- const KIRO_IDE_VERSION = "1.0.0";
67
- const KIRO_FALLBACK_SERIALIZATION_ENVELOPE_BYTES = 64 * 1024;
68
- type KiroWireClient = "ide" | "cli";
69
-
70
- function kiroCliPlatform(): "linux" | "macos" | "windows" {
71
- return process.platform === "win32" ? "windows" : process.platform === "darwin" ? "macos" : "linux";
72
- }
73
-
74
- function kiroCliUserAgent(includeAppVersion: boolean): string {
75
- return [
76
- "aws-sdk-rust/1.3.15",
77
- "ua/2.1",
78
- "api/codewhispererstreaming/0.1.17975",
79
- `os/${kiroCliPlatform()}`,
80
- "lang/rust/1.92.0",
81
- ...(includeAppVersion ? ["md/appVersion-2.14.2"] : []),
82
- "m/F",
83
- "app/AmazonQ-For-CLI",
84
- ].join(" ");
85
- }
86
-
87
- // Payload construction (conversationState)
88
- interface KiroToolUse {
89
- name: string;
90
- input: Record<string, unknown>; // OBJECT, not stringified
91
- toolUseId: string;
92
- }
93
- interface KiroToolResult {
94
- content: Array<{ text: string }>;
95
- status: string;
96
- toolUseId: string;
97
- }
98
- interface KiroUserInputMessage {
99
- content: string;
100
- modelId?: string;
101
- origin?: string;
102
- userInputMessageContext?: {
103
- tools?: unknown[];
104
- toolResults?: KiroToolResult[];
105
- };
106
- images?: KiroImage[];
107
- }
108
- interface KiroHistoryEntry {
109
- userInputMessage?: KiroUserInputMessage;
110
- assistantResponseMessage?: {
111
- content: string;
112
- toolUses?: KiroToolUse[];
113
- reasoningContent?: { redactedContent: string };
114
- };
115
- }
116
-
117
- function kiroToolWireNames(tools: readonly unknown[]): string[] {
118
- return tools
119
- .map(tool => {
120
- const spec = (tool as { toolSpecification?: { name?: unknown } }).toolSpecification;
121
- return typeof spec?.name === "string" ? spec.name : undefined;
122
- })
123
- .filter((name): name is string => typeof name === "string");
124
- }
125
-
126
- function userContentText(content: string | OcxContentPart[]): string {
127
- if (typeof content === "string") return content;
128
- return content.map(p => (p.type === "text" ? p.text : "")).filter(Boolean).join("\n");
129
- }
130
-
131
- function usageContentText(content: string | OcxContentPart[]): string {
132
- if (typeof content === "string") return content;
133
- return content
134
- .map(p => {
135
- if (p.type === "text") return p.text;
136
- if (p.type === "image") return `[image:${p.detail ?? "auto"}]`;
137
- return "";
138
- })
139
- .filter(Boolean)
140
- .join("\n");
141
- }
142
- function serializeForUsage(value: unknown): string {
143
- try { return JSON.stringify(value); } catch { return String(value); }
144
- }
145
- function currentTurnUsageMessages(messages: OcxMessage[]): OcxMessage[] {
146
- return messages.slice(messages.map(m => m.role).lastIndexOf("assistant") + 1).filter(m => m.role !== "assistant");
147
- }
148
- function kiroPayloadMessages(parsed: OcxParsedRequest): OcxMessage[] {
149
- return parsed.context.messages;
150
- }
151
-
152
- function messageUsageText(msg: OcxMessage): string {
153
- switch (msg.role) {
154
- case "user":
155
- case "developer":
156
- return usageContentText(msg.content);
157
- case "toolResult":
158
- return [
159
- msg.toolName,
160
- msg.toolCallId,
161
- msg.isError ? "error" : "success",
162
- usageContentText(msg.content),
163
- ].filter(Boolean).join("\n");
164
- case "assistant":
165
- return "";
166
- }
167
- }
168
-
169
- function messageLogText(msg: OcxMessage): string {
170
- if (msg.role !== "assistant") return messageUsageText(msg);
171
- return msg.content.map(part => {
172
- if (part.type === "text") return part.text;
173
- if (part.type === "toolCall") return [part.name, part.id, serializeForUsage(part.arguments)].join("\n");
174
- return part.thinking;
175
- }).filter(Boolean).join("\n");
176
- }
177
-
178
- function estimateKiroImageTokens(image: KiroImage): number {
179
- const dimensions = sniffImageDimensions(image.source.bytes);
180
- if (dimensions) {
181
- return Math.max(256, Math.ceil(dimensions.width * dimensions.height / 750));
182
- }
183
- const decodedBytes = Math.floor(image.source.bytes.length * 3 / 4);
184
- return Math.max(256, Math.ceil(decodedBytes / 512));
185
- }
186
-
187
- function estimateKiroTokens(text: string, modelId?: string): number {
188
- return estimateTokens(text, modelId ? `kiro/${modelId}` : "kiro");
189
- }
190
-
191
- /** Hangul/Han/kana ranges, matching the shared estimator's own CJK classification. */
192
- function kiroCjkCount(text: string): number {
193
- let cjk = 0;
194
- for (let i = 0; i < text.length; i++) {
195
- const c = text.charCodeAt(i);
196
- if (
197
- (c >= 0xac00 && c <= 0xd7a3) || (c >= 0x1100 && c <= 0x11ff) || (c >= 0x3130 && c <= 0x318f)
198
- || (c >= 0x4e00 && c <= 0x9fff) || (c >= 0x3400 && c <= 0x4dbf) || (c >= 0x3040 && c <= 0x30ff)
199
- ) cjk++;
200
- }
201
- return cjk;
202
- }
203
-
204
- /**
205
- * Token estimate for walked payload text, with the wire expansion applied to the Latin portion
206
- * only. Splitting here rather than inside the shared estimator keeps that module pure and
207
- * provider-neutral: the expansion is a fact about Kiro's wire, not about tokenization.
208
- */
209
- function estimateKiroWireTokens(text: string, modelId: string): number {
210
- if (!text) return 0;
211
- const cjk = kiroCjkCount(text);
212
- if (cjk === 0) return Math.ceil(estimateKiroTokens(text, modelId) * KIRO_LATIN_WIRE_EXPANSION);
213
- const latinTokens = estimateKiroTokens("x".repeat(text.length - cjk), modelId);
214
- const cjkTokens = estimateKiroTokens("\uac00".repeat(cjk), modelId);
215
- return Math.ceil(latinTokens * KIRO_LATIN_WIRE_EXPANSION + cjkTokens);
216
- }
217
-
218
- /**
219
- * Structural cost of one conversation entry, in tokens.
220
- *
221
- * The walker below concatenates message TEXT, but the wire carries JSON: per-entry keys
222
- * (`userInputMessage`, `content`, `modelId`, `origin`) and role framing. That is charged
223
- * upstream and is invisible to a text-only count, so without it a long conversation drifts
224
- * further below the real charge with every turn added — an error proportional to entry COUNT,
225
- * which no per-character ratio can recover.
226
- *
227
- * Regressing serialized bodies against what the walker counts, over eleven payload sizes from
228
- * 3 to 701 entries:
229
- *
230
- * bodyBytes = 1.0422 * walkedChars + 66.7 * entries + 68
231
- *
232
- * 66.7 bytes at the measured 2.433 bytes per charged token is 27.4 tokens per entry. The
233
- * earlier value of 12 was a conservative hand-fit taken before that regression existed, and
234
- * being less than half the real cost is precisely why the estimate decayed with conversation
235
- * length: an under-charge of ~15 tokens per entry is invisible across four messages and
236
- * dominant across seven hundred.
237
- *
238
- * Cross-checked against 4,090 recorded requests, where real traffic averages 1,310 bytes per
239
- * message: 66.7 bytes is 5% of that, so this term charges framing and is not quietly absorbing
240
- * message content.
241
- */
242
- const KIRO_ENTRY_FRAMING_TOKENS = 27;
243
-
244
- /**
245
- * Multiplier reconciling the LATIN text estimate with what the wire charges for that same text.
246
- *
247
- * The shared estimator counts Latin text at 2.8 chars/token, while the wire charges 2.433 bytes
248
- * per token at 1.0422 bytes per walked character — an effective 2.334 chars/token, and
249
- * 2.8 / 2.334 = 1.199.
250
- *
251
- * The evidence that the split between this term and `KIRO_ENTRY_FRAMING_TOKENS` is right is its
252
- * stability: holding framing at 27, the multiplier the charge implies stays within 1.189-1.209
253
- * across a 230x range of conversation sizes. A mis-specified split drifts with size, and the
254
- * earlier 1.12/12 pair did — its accuracy fell from 0.92 at four messages to 0.87 at seven
255
- * hundred.
256
- *
257
- * LATIN ONLY, deliberately. 2.433 bytes/token is a property of this traffic mix, which is Latin
258
- * and code. A Hangul character is three UTF-8 bytes but roughly one token, so its bytes-per-token
259
- * is entirely different and a Latin-derived byte rate says nothing about it. Scaling CJK by this
260
- * factor bills Hangul at 1.25 chars/token, against recorded ground truth that already places the
261
- * shared 1.5 ratio at 0.90 of the authoritative count — an over-charge that would compact Korean
262
- * threads early.
263
- *
264
- * This is NOT JSON escaping, despite what an earlier version of this comment claimed. Measured
265
- * directly, `JSON.stringify` expands prose by 1.012 (Latin) to 1.019 (Korean), nowhere near 1.2.
266
- * Escaping is real but small, and is already inside the byte measurement this factor comes from.
267
- */
268
- const KIRO_LATIN_WIRE_EXPANSION = 1.2;
269
-
270
- function estimateKiroPayloadInputTokens(payload: Record<string, unknown>, modelId: string): number {
271
- const conversationState = (payload as {
272
- conversationState?: {
273
- history?: KiroHistoryEntry[];
274
- currentMessage?: KiroHistoryEntry;
275
- };
276
- }).conversationState;
277
- if (!conversationState) return 0;
278
-
279
- const parts: string[] = [];
280
- let imageTokens = 0;
281
- const entries = [
282
- ...(conversationState.history ?? []),
283
- ...(conversationState.currentMessage ? [conversationState.currentMessage] : []),
284
- ];
285
- for (const entry of entries) {
286
- const user = entry.userInputMessage;
287
- if (user) {
288
- if (user.content) parts.push(user.content);
289
- for (const image of user.images ?? []) imageTokens += estimateKiroImageTokens(image);
290
- const context = user.userInputMessageContext;
291
- if (context?.tools?.length) parts.push(serializeForUsage(context.tools));
292
- if (context?.toolResults?.length) parts.push(serializeForUsage(context.toolResults));
293
- }
294
- const assistant = entry.assistantResponseMessage;
295
- if (assistant) {
296
- if (assistant.content) parts.push(assistant.content);
297
- if (assistant.toolUses?.length) parts.push(serializeForUsage(assistant.toolUses));
298
- }
299
- }
300
- return estimateKiroWireTokens(parts.join("\n"), modelId)
301
- + imageTokens
302
- + entries.length * KIRO_ENTRY_FRAMING_TOKENS;
303
- }
304
-
305
- function shouldCountStablePromptOverhead(parsed: OcxParsedRequest): boolean {
306
- return !parsed.previousResponseId && !parsed.context.messages.some(m => m.role === "assistant");
307
- }
308
-
309
- function estimateKiroInputTokens(parsed: OcxParsedRequest): number {
310
- const parts = currentTurnUsageMessages(parsed.context.messages)
311
- .map(messageUsageText)
312
- .filter(Boolean);
313
-
314
- if (shouldCountStablePromptOverhead(parsed)) {
315
- if (parsed.context.systemPrompt?.length) parts.push(...parsed.context.systemPrompt);
316
- if (parsed.context.tools?.length) parts.push(serializeForUsage(parsed.context.tools));
317
- }
318
-
319
- return estimateKiroTokens(parts.join("\n"), parsed.modelId);
320
- }
321
-
322
- function estimateKiroLogInputTokens(parsed: OcxParsedRequest): number {
323
- const parts = parsed.context.messages.map(messageLogText).filter(Boolean);
324
- if (parsed.context.systemPrompt?.length) parts.push(...parsed.context.systemPrompt);
325
- if (parsed.context.tools?.length) parts.push(serializeForUsage(parsed.context.tools));
326
- return Math.max(estimateKiroInputTokens(parsed), estimateKiroTokens(parts.join("\n"), parsed.modelId));
327
- }
328
-
329
- function kiroUpstreamContextWindow(modelId: string | undefined): number | undefined {
330
- if (!modelId) return undefined;
331
- const normalizedModelId = normalizeKiroModelId(modelId);
332
- if (normalizedModelId === "auto") return undefined;
333
- const window = modelRecordValue(KIRO_MODEL_CONTEXT_WINDOWS, modelId)
334
- ?? modelRecordValue(KIRO_MODEL_CONTEXT_WINDOWS, normalizedModelId);
335
- return typeof window === "number" && Number.isFinite(window) && window > 0 ? window : undefined;
336
- }
337
-
338
- function kiroRuntimeEndpoint(provider: OcxProviderConfig, region: string): string {
339
- const configured = new URL(provider.baseUrl);
340
- if (
341
- /^runtime\.[a-z]{2}(?:-[a-z]+)+-\d\.kiro\.dev$/i.test(configured.hostname)
342
- && configured.pathname === "/"
343
- ) {
344
- return `https://runtime.${region}.kiro.dev/`;
345
- }
346
- return configured.toString();
347
- }
348
-
349
- export type KiroReasoningMode = "native" | "emulated";
350
-
351
- // Kiro takes a verified native effort field for these models, and each model family names it
352
- // differently: the Sol-only `reasoning.effort` versus the Claude-specific `output_config.effort`.
353
- // Models absent from this table fall back to emulated thinking instructions.
354
- const KIRO_NATIVE_EFFORT_FIELDS: Record<string, "reasoning" | "output_config"> = {
355
- "gpt-5.6-sol": "reasoning",
356
- "claude-opus-5": "output_config",
357
- };
358
-
359
- const KIRO_NATIVE_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
360
-
361
- function kiroNativeEffortField(modelId: string): "reasoning" | "output_config" | undefined {
362
- return KIRO_NATIVE_EFFORT_FIELDS[normalizeKiroModelId(modelId)];
363
- }
364
-
365
- export function kiroReasoningMode(modelId: string): KiroReasoningMode {
366
- return kiroNativeEffortField(modelId) ? "native" : "emulated";
367
- }
368
-
369
- function kiroThinkingBudget(parsed: OcxParsedRequest): number | undefined {
370
- const effort = parsed.options.reasoning;
371
- if (!effort || effort === "none") return undefined;
372
- const maxTokens = parsed.options.maxOutputTokens || 4096;
373
- const percent: Record<string, number> = {
374
- minimal: 0.10,
375
- low: 0.20,
376
- medium: 0.50,
377
- high: 0.80,
378
- xhigh: 0.90,
379
- max: 0.95,
380
- };
381
- const ratio = percent[effort];
382
- return ratio === undefined ? undefined : Math.max(1, Math.floor(maxTokens * ratio));
383
- }
384
-
385
- function injectKiroThinkingTags(content: string, parsed: OcxParsedRequest): string {
386
- if (kiroReasoningMode(parsed.modelId) !== "emulated") return content;
387
- const budget = kiroThinkingBudget(parsed);
388
- if (!budget) return content;
389
- const instruction = [
390
- "Think in English for better reasoning quality.",
391
- "Be thorough and systematic, consider edge cases, challenge assumptions, and verify reasoning before answering.",
392
- "After thinking, respond in the user's language.",
393
- ].join("\n");
394
- return [
395
- "<thinking_mode>enabled</thinking_mode>",
396
- `<max_thinking_length>${budget}</max_thinking_length>`,
397
- `<thinking_instruction>${instruction}</thinking_instruction>`,
398
- "",
399
- content,
400
- ].join("\n");
401
- }
402
-
403
- function validateKiroCapabilities(parsed: OcxParsedRequest): void {
404
- const choice = parsed.options.toolChoice;
405
- if (choice !== undefined && choice !== "auto" && choice !== "none") {
406
- throw new Error("Kiro supports only automatic tool choice or tool_choice:none");
407
- }
408
- if (parsed.options.serviceTier !== undefined) {
409
- throw new Error("Kiro does not support service tiers");
410
- }
411
- // Structured output is a real contract Kiro cannot honour: the wire has no
412
- // schema-constrained response mode, so a caller expecting parseable JSON would receive
413
- // prose and fail downstream. Refuse it.
414
- //
415
- // The rest of the Responses `text` object is not that. `text.verbosity` is a length
416
- // preference and `text.format: {type:"text"}` is ordinary prose — the default output
417
- // mode, which no capability flag governs and every correct client may send. Testing
418
- // `_rawBody.text !== undefined` refused those turns for the mere PRESENCE of the key,
419
- // the same mistake db040e70f removed one condition earlier where a permissive
420
- // `parallel_tool_calls` hint was read as a requirement.
421
- //
422
- // Nothing needs stripping the way openai-responses strips a no-op verbosity:
423
- // buildKiroPayload composes conversationState field by field from `parsed` and never
424
- // spreads `_rawBody`, so a tolerated control is dropped by construction. The test
425
- // asserts that absence so it stays true.
426
- if (parsed._structuredOutput) {
427
- throw new Error("Kiro does not support Responses structured output");
428
- }
429
- }
430
-
431
- type KiroTurn =
432
- | {
433
- kind: "user";
434
- content: string;
435
- images: KiroImage[];
436
- toolResults: KiroToolResult[];
437
- /**
438
- * True only for the proxy-generated acknowledgement that follows a delivered final answer.
439
- * A flag rather than a content comparison: a real user message may legitimately quote the
440
- * same sentence, and treating that as internal state would strip its thinking tags and
441
- * completion retry.
442
- */
443
- answerDeliveredAck?: boolean;
444
- }
445
- | {
446
- kind: "assistant";
447
- content: string;
448
- toolUses: KiroToolUse[];
449
- redactedReasoning?: string;
450
- /**
451
- * True when this assistant turn was the DELIVERED final answer (Responses
452
- * `phase: "final_answer"`). A trailing assistant turn normally means the model stopped
453
- * mid-task and needs a continuation prompt, but a delivered final answer already ended its
454
- * turn — prompting it again restarts finished work as if a goal were still open.
455
- */
456
- finalAnswer?: boolean;
457
- };
458
-
459
- /**
460
- * True when the LAST content-bearing message is an assistant final answer that closed its turn.
461
- *
462
- * Mirrors the turn-merge rule: a tool call in that message, or any later user/tool-result message,
463
- * means work continued, so the turn is no longer terminal. Empty assistant messages are skipped
464
- * rather than treated as continuation, since they carry no visible turn.
465
- */
466
- function hasTrailingDeliveredFinalAnswer(messages: readonly OcxMessage[], parsed?: OcxParsedRequest): boolean {
467
- for (let i = messages.length - 1; i >= 0; i--) {
468
- const msg = messages[i];
469
- if (msg.role !== "assistant") return false;
470
- const aMsg = msg as OcxAssistantMessage;
471
- const hasToolCall = (aMsg.content ?? []).some(part => part.type === "toolCall");
472
- if (hasToolCall) return false;
473
- const hasText = (aMsg.content ?? []).some(part => part.type === "text" && part.text.trim());
474
- if (!hasText) continue;
475
- return aMsg.phase === "final_answer"
476
- || (parsed !== undefined && hasRecordedTrailingDeliveredFinalAnswer(parsed, messages));
477
- }
478
- return false;
479
- }
480
-
481
- function appendTurnText(target: string, next: string): string {
482
- if (!next) return target;
483
- return target ? `${target}\n\n${next}` : next;
484
- }
485
-
486
- function validateKiroConversationState(history: KiroHistoryEntry[], currentMessage: KiroHistoryEntry): void {
487
- const entries = [...history, currentMessage];
488
- const pendingToolUses = new Set<string>();
489
- let previousRole: "user" | "assistant" | undefined;
490
-
491
- for (const entry of entries) {
492
- const user = entry.userInputMessage;
493
- const assistant = entry.assistantResponseMessage;
494
- if (Boolean(user) === Boolean(assistant)) {
495
- throw new Error("Kiro conversation entries must contain exactly one message role");
496
- }
497
- const role = user ? "user" : "assistant";
498
- if (role === previousRole) throw new Error("Kiro conversation roles must alternate");
499
- previousRole = role;
500
-
501
- if (user) {
502
- const hasPayload = Boolean(user.content.trim())
503
- || Boolean(user.images?.length)
504
- || Boolean(user.userInputMessageContext?.toolResults?.length);
505
- if (!hasPayload) throw new Error("Kiro user messages must not be empty");
506
- for (const result of user.userInputMessageContext?.toolResults ?? []) {
507
- if (!pendingToolUses.delete(result.toolUseId)) {
508
- throw new Error(`Kiro tool result has no matching tool use ${JSON.stringify(result.toolUseId)}`);
509
- }
510
- if (!result.content.some(part => part.text.trim())) {
511
- throw new Error(`Kiro tool result must not be empty ${JSON.stringify(result.toolUseId)}`);
512
- }
513
- }
514
- continue;
515
- }
516
-
517
- const toolUses = assistant?.toolUses ?? [];
518
- if (!assistant?.content.trim() && toolUses.length === 0) {
519
- throw new Error("Kiro assistant messages must not be empty");
520
- }
521
- for (const toolUse of toolUses) {
522
- if (pendingToolUses.has(toolUse.toolUseId)) {
523
- throw new Error(`Kiro conversation contains duplicate tool use ${JSON.stringify(toolUse.toolUseId)}`);
524
- }
525
- pendingToolUses.add(toolUse.toolUseId);
526
- }
527
- }
528
- if (pendingToolUses.size > 0) throw new Error("Kiro conversation contains an unanswered tool use");
529
- }
530
-
531
- function boundedInjectedInstruction(text: string, used: { value: number }): string | undefined {
532
- const remaining = MAX_KIRO_INJECTED_INSTRUCTION_CHARS - used.value;
533
- if (remaining <= 0 || !text) return undefined;
534
- let result = text.length <= remaining ? text : text.slice(0, remaining);
535
- // Never end the slice on a lone high surrogate: encoding it substitutes
536
- // U+FFFD into the injected instruction. One step back keeps a valid pair
537
- // out instead of a broken half.
538
- if (result.length > 0) {
539
- const last = result.charCodeAt(result.length - 1);
540
- if (last >= 0xd800 && last <= 0xdbff) result = result.slice(0, -1);
541
- }
542
- used.value += result.length;
543
- return result.length > 0 ? result : undefined;
544
- }
545
-
546
- /** Test-only: exercise the surrogate-safe instruction bound directly. */
547
- export function boundedInjectedInstructionForTests(text: string, used: { value: number }): string | undefined {
548
- return boundedInjectedInstruction(text, used);
549
- }
550
-
551
- function kiroCompletionTool(): Record<string, unknown> {
552
- return {
553
- toolSpecification: {
554
- name: KIRO_COMPLETION_TOOL_NAME,
555
- // The shared tool-catalog nudge enumerates this name next to ordinary tools and tells every
556
- // listed name to count a call only after its tool result returns. Nothing returns a result
557
- // here: a valid call becomes the turn's terminal. Left undescribed, the model reads one more
558
- // deferrable work tool and keeps calling tools with a finished answer already written as
559
- // commentary. So the description states the distinction, the obligation, and the terminality
560
- // where the model is actually choosing between tools.
561
- //
562
- // It also has to name the blocked-on-user state, for the same reason the prose contract does.
563
- // This is the surface the model reads while CHOOSING; if it admits only "fully complete", a
564
- // model holding a question that blocks progress reads this tool as unavailable and keeps
565
- // working instead, which is the measured defect. The two surfaces must not disagree.
566
- description: "Terminal completion channel, not an ordinary work tool. When the task is fully complete and no more work or tool calls are needed, you must call this tool exactly once instead of providing the final answer as ordinary assistant text. Call it the same way when you cannot continue until the user supplies a decision, information, or a clarification that only they can give: the question itself is the answer. Put the complete user-facing final answer in `answer`. The call is complete when issued: it ends the turn, returns no tool result, and no text or tool call may follow it.",
567
- inputSchema: {
568
- json: {
569
- type: "object",
570
- properties: {
571
- answer: {
572
- type: "string",
573
- description: "The complete final answer to show the user, or the blocking question you need the user to answer before you can continue.",
574
- },
575
- },
576
- required: ["answer"],
577
- },
578
- },
579
- },
580
- };
581
- }
582
-
583
- export function buildKiroPayload(
584
- parsed: OcxParsedRequest,
585
- profileArn: string | undefined,
586
- forcedCompletionMode?: KiroCompletionMode,
587
- wireClient: KiroWireClient = "ide",
588
- ): {
589
- payload: Record<string, unknown>;
590
- nameMap: Map<string, string>;
591
- conversationId: string;
592
- completionMode: KiroCompletionMode;
593
- } {
594
- validateKiroCapabilities(parsed);
595
- const modelId = mapModelId(parsed.modelId);
596
- const registry = createKiroToolNameRegistry();
597
- const toolContext = convertKiroToolContext(parsed, registry);
598
- const ordinaryTools = toolContext.tools;
599
- // A turn whose history already ENDS with a delivered final answer has nothing to complete.
600
- // Leaving completion "required" here would keep advertising codex_kiro_final_answer with its
601
- // instructions, so the model answers again, or replies with ordinary text and trips the
602
- // `needsFallback` retry, which ends its payload with KIRO_COMPLETION_RETRY_MESSAGE and reopens
603
- // the finished task. Suppressing the mode is what actually closes that loop; the neutral
604
- // acknowledgement below only stops the resume wording.
605
- //
606
- // Read from parsed messages because `completionMode` is needed to build the tool catalog, which
607
- // happens before the turn list exists. `forcedCompletionMode` still wins: the fallback retry
608
- // passes "text_fallback" explicitly and must not be silently downgraded.
609
- const trailingDeliveredAnswer = hasTrailingDeliveredFinalAnswer(kiroPayloadMessages(parsed), parsed);
610
- const completionMode: KiroCompletionMode = forcedCompletionMode
611
- ?? (ordinaryTools.length > 0 && !trailingDeliveredAnswer ? "required" : "disabled");
612
- const kiroTools = completionMode === "disabled"
613
- ? ordinaryTools
614
- : [...ordinaryTools, kiroCompletionTool()];
615
- const nameMap = toolContext.nameMap;
616
- const systemParts: string[] = [];
617
- const injectedChars = { value: 0 };
618
- // Name the Kiro model id actually sent on the wire without leaking the proxy identity upstream.
619
- if (parsed.context.systemPrompt?.length) {
620
- systemParts.push(identifyRoutedModel(parsed.context.systemPrompt.join("\n\n"), modelId));
621
- }
622
- for (const addition of toolContext.systemAdditions) {
623
- const boundedAddition = boundedInjectedInstruction(addition, injectedChars);
624
- if (boundedAddition) systemParts.push(boundedAddition);
625
- }
626
- // Kiro renames tools to satisfy its wire constraints, so resolve neighbor names through the
627
- // registry's existing aliases; a bare-name comparison would forbid tools this turn actually
628
- // advertises. Read the recorded mapping instead of calling `alias()`, which would REGISTER a
629
- // name for a tool that was never advertised and pollute the collision domain.
630
- const advertisedAlias = new Map<string, string>();
631
- for (const [alias, wireName] of registry.nameMap) advertisedAlias.set(wireName, alias);
632
- // Code mode is decided on the EMITTED catalog, not the requested list.
633
- //
634
- // `freeform` only exists on the requested tool objects -- `kiroToolWireNames` has already
635
- // reduced the emitted catalog to strings -- so the predicates must read the objects. But the
636
- // SHAPE that matters is the one the model receives: the count/byte budget can drop a requested
637
- // `exec_command` while `exec` survives, and scanning the requested list would then find a shell
638
- // bridge the model cannot call and suppress code mode for a catalog that is code-mode-shaped.
639
- // Intersecting the two keeps `tool_choice: "none"` and budget omission correct for free: both
640
- // empty the emitted set, so nothing can be named.
641
- const emittedToolNames = new Set(kiroToolWireNames(kiroTools));
642
- const emittedAlias = (tool: OcxTool): string | undefined => {
643
- const wireName = namespacedToolName(tool.namespace, tool.name);
644
- // Read the recorded mapping; `registry.alias()` would REGISTER a name here.
645
- const alias = advertisedAlias.get(wireName) ?? wireName;
646
- return emittedToolNames.has(alias) ? alias : undefined;
647
- };
648
- const requestedTools = parsed.context.tools ?? [];
649
- const emittedCodeModeExec = requestedTools.find(tool => isCodexCodeModeExecTool(tool) && emittedAlias(tool));
650
- const emittedShellBridge = requestedTools.some(tool => isBareShellBridgeTool(tool) && emittedAlias(tool));
651
- const codeModeExecName = emittedCodeModeExec && !emittedShellBridge
652
- ? emittedAlias(emittedCodeModeExec)
653
- : undefined;
654
- const toolCatalogNudge = buildNonOpenAIToolCatalogNudgeFromNames(
655
- kiroToolWireNames(kiroTools),
656
- name => advertisedAlias.get(name) ?? name,
657
- codeModeExecName,
658
- );
659
- const boundedNudge = toolCatalogNudge ? boundedInjectedInstruction(toolCatalogNudge, injectedChars) : undefined;
660
- if (boundedNudge) systemParts.push(boundedNudge);
661
- if (completionMode !== "disabled") {
662
- const boundedCompletion = boundedInjectedInstruction(KIRO_COMPLETION_INSTRUCTIONS, injectedChars);
663
- if (boundedCompletion) systemParts.push(boundedCompletion);
664
- }
665
- const systemPrefix = systemParts.length > 0 ? `${systemParts.join("\n\n")}\n\n` : "";
666
- const turns: KiroTurn[] = [];
667
- const priorCalls = new Map<string, { wireName: string; rawId: string }>();
668
- const pushUser = (content: string, images: KiroImage[] = [], toolResults: KiroToolResult[] = []): void => {
669
- const last = turns.at(-1);
670
- if (last?.kind === "user") {
671
- last.content = appendTurnText(last.content, content);
672
- last.images.push(...images);
673
- last.toolResults.push(...toolResults);
674
- } else {
675
- turns.push({ kind: "user", content, images: [...images], toolResults: [...toolResults] });
676
- }
677
- };
678
- const pushAssistant = (content: string, toolUses: KiroToolUse[], redactedReasoning?: string, finalAnswer?: boolean): void => {
679
- const last = turns.at(-1);
680
- if (last?.kind === "assistant") {
681
- last.content = appendTurnText(last.content, content);
682
- last.toolUses.push(...toolUses);
683
- // Merged turns keep the newest blob: it covers the reasoning up to the merged turn's end.
684
- if (redactedReasoning) last.redactedReasoning = redactedReasoning;
685
- // A merged turn is final only if its LAST component was: commentary appended after a final
686
- // answer means the model kept working, so the turn is no longer terminal.
687
- last.finalAnswer = finalAnswer === true;
688
- } else {
689
- turns.push({
690
- kind: "assistant",
691
- content,
692
- toolUses: [...toolUses],
693
- ...(redactedReasoning ? { redactedReasoning } : {}),
694
- ...(finalAnswer ? { finalAnswer: true } : {}),
695
- });
696
- }
697
- };
698
-
699
- let adjacentResult: {
700
- rawId: string;
701
- result: KiroToolResult;
702
- texts: string[];
703
- count: number;
704
- hasImages: boolean;
705
- } | undefined;
706
- const finishAdjacentResult = (): void => {
707
- if (adjacentResult && adjacentResult.count > 1) {
708
- if (adjacentResult.texts.some(text => text.trim())) {
709
- adjacentResult.result.content = adjacentResult.texts.map(text => ({ text }));
710
- } else if (adjacentResult.hasImages || adjacentResult.result.status === "error") {
711
- adjacentResult.result.content = [{ text: KIRO_EMPTY_TOOL_RESULT_MESSAGE }];
712
- }
713
- }
714
- adjacentResult = undefined;
715
- };
716
-
717
- for (const msg of kiroPayloadMessages(parsed)) {
718
- // Original-message adjacency matters even when a turn is collapsed or skipped below.
719
- if (msg.role !== "toolResult") finishAdjacentResult();
720
- if (msg.role === "user" || msg.role === "developer") {
721
- const text = userContentText((msg as { content: string | OcxContentPart[] }).content);
722
- const images = extractKiroImages((msg as { content: string | OcxContentPart[] }).content);
723
- pushUser(text, images);
724
- } else if (msg.role === "assistant") {
725
- const aMsg = msg as OcxAssistantMessage;
726
- const text = (aMsg.content || [])
727
- .filter((b): b is OcxTextContent => b.type === "text")
728
- .map(b => b.text)
729
- .join("");
730
- const toolCalls = (aMsg.content || [])
731
- .filter((b): b is OcxToolCall => b.type === "toolCall");
732
- const toolUses: KiroToolUse[] = toolCalls.map(tc => {
733
- const toolUseId = normalizeToolId(tc.id);
734
- if (!toolUseId) throw new Error("Kiro history contains a tool call with an empty id");
735
- if (priorCalls.has(toolUseId)) throw new Error(`Kiro history contains duplicate tool call id ${JSON.stringify(tc.id)}`);
736
- const wireName = namespacedToolName(tc.namespace, tc.name);
737
- const name = registry.alias(wireName);
738
- priorCalls.set(toolUseId, { wireName, rawId: tc.id });
739
- return { name, input: (tc.arguments ?? {}) as Record<string, unknown>, toolUseId };
740
- });
741
- if (!text && toolUses.length === 0) {
742
- const hasReasoning = aMsg.content.some(part => part.type === "thinking" && part.thinking.trim());
743
- if (hasReasoning) continue;
744
- }
745
- // `phase` survives the Responses round trip (parser.ts assistant branch), so a replayed
746
- // final answer is identifiable here rather than guessed from turn position.
747
- pushAssistant(text, toolUses, aMsg.kiroRedactedReasoning, aMsg.phase === "final_answer" && toolUses.length === 0);
748
- } else if (msg.role === "toolResult") {
749
- const tr = msg as OcxToolResultMessage;
750
- if (tr.containsEncryptedContent) {
751
- throw new Error(`Kiro cannot translate encrypted output for tool call ${JSON.stringify(tr.toolCallId)}`);
752
- }
753
- const text = userContentText(tr.content);
754
- // An empty code-mode exec result needs the SPECIFIC reason, not the generic fallback: the
755
- // model otherwise reads a blank result, concludes its earlier context was lost, and restarts
756
- // the task instead of calling text()/notify(). Checked before `text.trim()` because the
757
- // wrapper form ("Script completed\nWall time ...\nOutput:\n") is non-blank and would
758
- // otherwise pass through as if it were real output.
759
- const execOptions = { toolName: tr.toolName, toolNamespace: tr.toolNamespace };
760
- const normalizedExecText = normalizeEmptyExecToolResultText(text, execOptions);
761
- // A host failure string inside a non-empty exec result gets the rule it broke appended, but
762
- // only when this request's emitted catalog is genuinely code mode (`codeModeExecName` above):
763
- // a structured tool named exec, or exec beside a shell bridge, never ran the isolate. This is
764
- // the only substitution the grouping path below also carries: whitespace and empty/failed
765
- // wrappers keep their existing raw policy.
766
- const annotatedExecText = normalizedExecText === undefined && codeModeExecName !== undefined
767
- ? annotateCodeModeHostFailure(text, execOptions)
768
- : undefined;
769
- const resultText = normalizedExecText ?? annotatedExecText ?? (text.trim() ? text : KIRO_EMPTY_TOOL_RESULT_MESSAGE);
770
- const images = extractKiroImages(tr.content);
771
- const toolUseId = normalizeToolId(tr.toolCallId);
772
- const call = priorCalls.get(toolUseId);
773
- if (!call || call.rawId !== tr.toolCallId) {
774
- throw new Error(`Kiro history contains an orphaned tool result for call ${JSON.stringify(tr.toolCallId)}`);
775
- }
776
- // Keep real whitespace and failed wrappers, but no empty-success wrapper boilerplate.
777
- const rawGroupText = text.length > 0 && (!text.trim() || normalizedExecText !== EMPTY_EXEC_OUTPUT_MESSAGE)
778
- ? (annotatedExecText ?? text) : undefined;
779
- const last = turns.at(-1);
780
- if (
781
- adjacentResult?.rawId === tr.toolCallId
782
- && last?.kind === "user"
783
- && last.toolResults.at(-1) === adjacentResult.result
784
- ) {
785
- adjacentResult.count += 1;
786
- adjacentResult.hasImages ||= images.length > 0;
787
- if (rawGroupText !== undefined) adjacentResult.texts.push(rawGroupText);
788
- last.images.push(...images);
789
- if (tr.isError) adjacentResult.result.status = "error";
790
- continue;
791
- }
792
- finishAdjacentResult();
793
- // Carrier text is a placeholder for an OTHERWISE EMPTY tool-result turn, not a prefix.
794
- // Passing it here would push proxy filler AHEAD of a human instruction that Claude Code
795
- // sends in the same turn (mid-turn steering / queued_command, issue #543), burying the
796
- // newest user intent behind boilerplate. Backfill below only when nothing else speaks.
797
- const result: KiroToolResult = {
798
- content: [{ text: resultText }],
799
- status: tr.isError ? "error" : "success",
800
- toolUseId,
801
- };
802
- pushUser("", images, [result]);
803
- adjacentResult = {
804
- rawId: tr.toolCallId, result,
805
- texts: rawGroupText === undefined ? [] : [rawGroupText],
806
- count: 1, hasImages: images.length > 0,
807
- };
808
- }
809
- }
810
- finishAdjacentResult();
811
-
812
- if (turns.length === 0 || turns[0].kind === "assistant") {
813
- turns.unshift({ kind: "user", content: KIRO_CONTINUATION_MESSAGE, images: [], toolResults: [] });
814
- }
815
- // Kiro requires the request to end with a user turn, so a trailing assistant turn always gets
816
- // one appended (the pop below throws otherwise). What that turn SAYS is the load-bearing part.
817
- //
818
- // Normally a trailing assistant turn means the model stopped mid-task, and a continuation/retry
819
- // prompt is correct. A DELIVERED final answer is the exception: the turn already ended, and
820
- // telling that model to "continue" or to call the completion tool again reopens finished work —
821
- // the completed-task-behaves-like-an-open-goal loop. It gets a neutral acknowledgement instead:
822
- // structurally valid, but carrying no instruction to resume.
823
- const trailing = turns.at(-1);
824
- if (trailing?.kind === "assistant") {
825
- const resumeText = completionMode === "text_fallback" ? KIRO_COMPLETION_RETRY_MESSAGE : KIRO_CONTINUATION_MESSAGE;
826
- turns.push({
827
- kind: "user",
828
- content: trailing.finalAnswer ? KIRO_ANSWER_DELIVERED_MESSAGE : resumeText,
829
- images: [],
830
- toolResults: [],
831
- ...(trailing.finalAnswer ? { answerDeliveredAck: true } : {}),
832
- });
833
- }
834
-
835
- // Give tool-result turns a carrier sentence ONLY when they carry no other text. This runs
836
- // before the pop below so the current turn is covered too: skipping it there would ship an
837
- // empty current content, which validateKiroConversationState accepts (tool results count as
838
- // payload) and would therefore fail silently.
839
- for (const turn of turns) {
840
- if (turn.kind === "user" && !turn.content.trim() && turn.toolResults.length > 0) {
841
- turn.content = KIRO_TOOL_RESULT_CARRIER_MESSAGE;
842
- }
843
- }
844
-
845
- const currentTurn = turns.pop();
846
- if (!currentTurn || currentTurn.kind !== "user") throw new Error("Kiro request must end with a user turn");
847
- // Survives the pop as state, so the checks below never infer intent from user-supplied text.
848
- const answerDeliveredAck = currentTurn.answerDeliveredAck === true;
849
- const toEntry = (turn: KiroTurn): KiroHistoryEntry => turn.kind === "assistant"
850
- ? {
851
- assistantResponseMessage: {
852
- content: turn.content,
853
- ...(turn.toolUses.length > 0 ? { toolUses: turn.toolUses } : {}),
854
- ...(turn.redactedReasoning ? { reasoningContent: { redactedContent: turn.redactedReasoning } } : {}),
855
- },
856
- }
857
- : {
858
- userInputMessage: {
859
- content: turn.content,
860
- modelId,
861
- origin: wireClient === "cli" ? "KIRO_CLI" : "AI_EDITOR",
862
- ...(turn.images.length > 0 ? { images: turn.images } : {}),
863
- ...(turn.toolResults.length > 0 ? { userInputMessageContext: { toolResults: turn.toolResults } } : {}),
864
- },
865
- };
866
- const history = turns.map(toEntry);
867
- const currentEntry = toEntry(currentTurn);
868
- const currentUim = currentEntry.userInputMessage!;
869
-
870
- if (systemPrefix) {
871
- const firstUser = history.find(e => e.userInputMessage)?.userInputMessage;
872
- if (firstUser) firstUser.content = systemPrefix + firstUser.content;
873
- else currentUim.content = systemPrefix + currentUim.content;
874
- }
875
- if (kiroTools.length > 0) {
876
- currentUim.userInputMessageContext = { ...(currentUim.userInputMessageContext ?? {}), tools: kiroTools };
877
- }
878
- if (completionMode === "text_fallback") {
879
- // Never append the retry instruction onto the answer-delivered acknowledgement: it exists
880
- // precisely to avoid asking a finished turn for another completion call, and appending here
881
- // would reinstate the loop it prevents.
882
- if (currentUim.content !== KIRO_COMPLETION_RETRY_MESSAGE && !answerDeliveredAck) {
883
- currentUim.content = appendTurnText(currentUim.content, KIRO_COMPLETION_RETRY_MESSAGE);
884
- }
885
- } else if (
886
- !currentUim.userInputMessageContext?.toolResults
887
- && currentUim.content !== KIRO_CONTINUATION_MESSAGE
888
- && !answerDeliveredAck
889
- ) {
890
- currentUim.content = injectKiroThinkingTags(currentUim.content, parsed);
891
- }
892
-
893
- validateKiroConversationState(history, currentEntry);
894
- const conversationId = stableConversationId(parsed);
895
- const payload: Record<string, unknown> = {
896
- conversationState: {
897
- chatTriggerType: "MANUAL",
898
- ...(wireClient === "cli" ? {
899
- agentContinuationId: crypto.randomUUID(),
900
- agentTaskType: "vibe",
901
- } : {}),
902
- conversationId,
903
- currentMessage: { userInputMessage: currentUim },
904
- ...(history.length > 0 ? { history } : {}),
905
- },
906
- };
907
- const effort = parsed.options.reasoning;
908
- const effortField = kiroNativeEffortField(parsed.modelId);
909
- if (effortField && effort && effort !== "none") {
910
- if (!KIRO_NATIVE_EFFORTS.includes(effort)) {
911
- throw new Error(`Kiro ${normalizeKiroModelId(parsed.modelId)} does not support reasoning effort ${JSON.stringify(effort)}`);
912
- }
913
- payload.additionalModelRequestFields = { [effortField]: { effort } };
914
- }
915
- if (profileArn) payload.profileArn = profileArn;
916
- return { payload, nameMap, conversationId, completionMode };
917
- }
918
-
919
- // Stream parsing (shared by parseStream + parseResponse)
920
- // CodeWhisperer GenerateAssistantResponse ALWAYS returns an AWS eventstream body (there is no
921
- // non-streaming wire mode), so the streaming bridge and non-streaming Responses path decode the
922
- // same way — parseResponse just collects what parseStream yields.
923
- interface KiroAttemptParseResult {
924
- terminal?: AdapterEvent;
925
- needsFallback?: boolean;
926
- usage?: OcxUsage;
927
- providerState?: { kiro: { conversationId: string } };
928
- assistantText: string;
929
- sawReasoning: boolean;
930
- }
931
-
932
- interface KiroAttemptResult extends KiroAttemptParseResult {
933
- releaseRetained(): void;
934
- }
935
-
936
- interface KiroAttemptRetention {
937
- trackReplacement(previousBytes: number, nextBytes: number): void;
938
- retainEvent(event: AdapterEvent, bytes: number): void;
939
- releaseEvent(event: AdapterEvent): void;
940
- releaseAll(): void;
941
- }
942
-
943
- function createKiroAttemptRetention(budget: TranslatorBudget): KiroAttemptRetention {
944
- let retainedBytes = 0;
945
- const eventBytes = new Map<AdapterEvent, number>();
946
- return {
947
- trackReplacement(previousBytes, nextBytes) {
948
- retainedBytes = Math.max(0, retainedBytes - previousBytes) + nextBytes;
949
- },
950
- retainEvent(event, bytes) {
951
- retainedBytes += bytes;
952
- eventBytes.set(event, bytes);
953
- },
954
- releaseEvent(event) {
955
- const bytes = eventBytes.get(event);
956
- if (bytes === undefined) return;
957
- eventBytes.delete(event);
958
- retainedBytes = Math.max(0, retainedBytes - bytes);
959
- budget.releaseRetained(bytes, { kind: "retained_collectors" });
960
- },
961
- releaseAll() {
962
- if (retainedBytes > 0) budget.releaseRetained(retainedBytes, { kind: "retained_collectors" });
963
- retainedBytes = 0;
964
- eventBytes.clear();
965
- },
966
- };
967
- }
968
-
969
- interface KiroFallbackAttempt {
970
- response: Response;
971
- inputTokens: number;
972
- contextInputEstimate: number;
973
- nameMap: Map<string, string>;
974
- conversationId: string;
975
- releaseRequestBody?: () => void;
976
- }
977
-
978
- function appendedUtf8Bytes(previous: string, previousBytes: number, fragment: string): number {
979
- let nextBytes = previousBytes + Buffer.byteLength(fragment);
980
- const previousLast = previous.charCodeAt(previous.length - 1);
981
- const fragmentFirst = fragment.charCodeAt(0);
982
- if (previousLast >= 0xd800 && previousLast <= 0xdbff
983
- && fragmentFirst >= 0xdc00 && fragmentFirst <= 0xdfff) {
984
- nextBytes -= 2;
985
- }
986
- return nextBytes;
987
- }
988
-
989
- /** Exact UTF-8 size JSON.stringify() will use for a string, without materializing that copy. */
990
- function jsonStringSerializedUtf8Bytes(value: string): number {
991
- let bytes = 2; // Opening and closing quotes.
992
- for (let index = 0; index < value.length; index++) {
993
- const code = value.charCodeAt(index);
994
- if (code === 0x22 || code === 0x5c) {
995
- bytes += 2;
996
- } else if (code === 0x08 || code === 0x09 || code === 0x0a || code === 0x0c || code === 0x0d) {
997
- bytes += 2;
998
- } else if (code < 0x20) {
999
- bytes += 6;
1000
- } else if (code <= 0x7f) {
1001
- bytes += 1;
1002
- } else if (code <= 0x7ff) {
1003
- bytes += 2;
1004
- } else if (code >= 0xd800 && code <= 0xdbff) {
1005
- const next = value.charCodeAt(index + 1);
1006
- if (next >= 0xdc00 && next <= 0xdfff) {
1007
- bytes += 4;
1008
- index++;
1009
- } else {
1010
- bytes += 6;
1011
- }
1012
- } else if (code >= 0xdc00 && code <= 0xdfff) {
1013
- bytes += 6;
1014
- } else {
1015
- bytes += 3;
1016
- }
1017
- }
1018
- return bytes;
1019
- }
1020
-
1021
- interface KiroContextWindowState {
1022
- value?: number;
1023
- }
1024
-
1025
- type KiroFallbackFactory = (
1026
- conversationId: string | undefined,
1027
- assistantText: string,
1028
- sawReasoning: boolean,
1029
- budget: TranslatorBudget,
1030
- ) => Promise<KiroFallbackAttempt>;
1031
-
1032
- function mergeKiroUsage(
1033
- first: OcxUsage | undefined,
1034
- second: OcxUsage | undefined,
1035
- preserveFirstContextGrowth = false,
1036
- ): OcxUsage | undefined {
1037
- if (!first) return second;
1038
- if (!second) return first;
1039
- const sumOptional = (key: keyof OcxUsage): number | undefined => {
1040
- const a = first[key];
1041
- const b = second[key];
1042
- return typeof a === "number" || typeof b === "number"
1043
- ? (typeof a === "number" ? a : 0) + (typeof b === "number" ? b : 0)
1044
- : undefined;
1045
- };
1046
- const totalTokens = typeof first.totalTokens === "number" && typeof second.totalTokens === "number"
1047
- ? first.totalTokens + second.totalTokens
1048
- : undefined;
1049
- const carriedContextTotal = preserveFirstContextGrowth && typeof first.contextTotalTokens === "number"
1050
- ? first.contextTotalTokens + second.outputTokens
1051
- : undefined;
1052
- const combinedOutputTokens = first.outputTokens + second.outputTokens;
1053
- return {
1054
- inputTokens: first.inputTokens + second.inputTokens,
1055
- outputTokens: combinedOutputTokens,
1056
- ...(typeof first.contextTotalTokens === "number" || typeof second.contextTotalTokens === "number"
1057
- ? {
1058
- contextTotalTokens: Math.max(
1059
- first.contextTotalTokens ?? 0,
1060
- second.contextTotalTokens ?? 0,
1061
- carriedContextTotal ?? 0,
1062
- combinedOutputTokens,
1063
- ),
1064
- }
1065
- : {}),
1066
- ...(totalTokens !== undefined ? { totalTokens } : {}),
1067
- ...(sumOptional("cachedInputTokens") !== undefined ? { cachedInputTokens: sumOptional("cachedInputTokens") } : {}),
1068
- ...(sumOptional("cacheReadInputTokens") !== undefined ? { cacheReadInputTokens: sumOptional("cacheReadInputTokens") } : {}),
1069
- ...(sumOptional("cacheCreationInputTokens") !== undefined ? { cacheCreationInputTokens: sumOptional("cacheCreationInputTokens") } : {}),
1070
- ...(sumOptional("reasoningOutputTokens") !== undefined ? { reasoningOutputTokens: sumOptional("reasoningOutputTokens") } : {}),
1071
- ...(first.estimated || second.estimated ? { estimated: true } : {}),
1072
- };
1073
- }
1074
-
1075
- function retryableKiroIncomplete(
1076
- reason: string,
1077
- message: string,
1078
- usage: OcxUsage,
1079
- providerState: { kiro: { conversationId: string } } | undefined,
1080
- retryable = true,
1081
- ): AdapterEvent {
1082
- return {
1083
- type: "incomplete",
1084
- reason,
1085
- message,
1086
- usage,
1087
- retryable,
1088
- endTurn: false,
1089
- ...(providerState ? { providerState } : {}),
1090
- };
1091
- }
1092
-
1093
- /**
1094
- * Catch-path retryability for #519: only transport/socket failures with no emitted output
1095
- * are replay-safe. Malformed event payloads (`invalid Kiro …`) and any post-output failure
1096
- * stay terminal — same spirit as cursor's emittedOutput gate.
1097
- */
1098
- export function isRetryableKiroStreamCatchError(err: unknown, emittedOutput: boolean): boolean {
1099
- if (emittedOutput) return false;
1100
- const message = err instanceof Error ? err.message : String(err);
1101
- if (/^invalid Kiro\b/i.test(message)) return false;
1102
- // Include Smithy/eventstream truncation (`eventstream: truncated message at end of stream`):
1103
- // partial frame + clean EOF with zero output is the same replay-safe class as a socket close.
1104
- return /socket connection was closed|connection(?: was)? closed unexpectedly|ECONNRESET|EPIPE|UND_ERR_|fetch failed|decoder failed|premature close|other side closed|unexpected EOF|network connection lost|terminated|truncated message at end of stream|eventstream:\s*truncated/i
1105
- .test(message);
1106
- }
1107
-
1108
- /** Native clean-stop reason eligible for bounded private-completion validation. */
1109
- const KIRO_END_TURN_STOP_REASON = "END_TURN";
1110
-
1111
- async function* parseKiroAttempt(
1112
- response: Response,
1113
- budget: TranslatorBudget,
1114
- mode: KiroCompletionMode,
1115
- modelId: string | undefined,
1116
- inputTokens: number,
1117
- contextWindowState: KiroContextWindowState,
1118
- nameMap: Map<string, string> | undefined,
1119
- conversationId: string | undefined,
1120
- contextInputEstimate?: number,
1121
- /** True when an earlier attempt already flushed visible content to the client (#520). */
1122
- priorEmittedOutput = false,
1123
- ): AsyncGenerator<AdapterEvent, KiroAttemptResult> {
1124
- // `required` mode holds staged commentary until a real tool call or terminal metadata identifies
1125
- // the attempt boundary. Anything the inner parser leaves behind is flushed before the terminal.
1126
- const deferred: AdapterEvent[] = [];
1127
- const retention = createKiroAttemptRetention(budget);
1128
- // Shared box: the inner parser stages its calibration observation here on the completion path,
1129
- // and this wrapper decides whether the attempt was terminal enough to commit it. A box rather
1130
- // than a return field because the completion path has a dozen terminal returns and threading a
1131
- // field through every one of them is exactly the kind of edit that misses one.
1132
- const attemptCalibration: { value?: { conversationId: string; estimated: number; charged: number } } = {};
1133
- const attempt = parseKiroAttemptEvents(
1134
- response,
1135
- budget,
1136
- mode,
1137
- modelId,
1138
- inputTokens,
1139
- contextWindowState,
1140
- nameMap,
1141
- conversationId,
1142
- deferred,
1143
- retention,
1144
- attemptCalibration,
1145
- contextInputEstimate,
1146
- priorEmittedOutput,
1147
- );
1148
- let handedOff = false;
1149
- try {
1150
- const result = yield* attempt;
1151
- // A staged observation only counts when this attempt is the LAST one for the user turn. An
1152
- // attempt that asks for the bounded fallback streams again against a rebuilt payload, so
1153
- // committing here would move the factor twice for one turn and score the second observation
1154
- // against a payload the first had already inflated.
1155
- const staged = attemptCalibration.value;
1156
- attemptCalibration.value = undefined;
1157
- if (staged && !result.needsFallback) {
1158
- recordKiroCalibration(staged.conversationId, staged.estimated, staged.charged);
1159
- }
1160
- for (const event of deferred.splice(0)) {
1161
- try { yield event; } finally { retention.releaseEvent(event); }
1162
- }
1163
- handedOff = true;
1164
- return { ...result, releaseRetained: () => retention.releaseAll() };
1165
- } finally {
1166
- if (!handedOff) retention.releaseAll();
1167
- }
1168
- }
1169
-
1170
- async function* parseKiroAttemptEvents(
1171
- response: Response,
1172
- budget: TranslatorBudget,
1173
- mode: KiroCompletionMode,
1174
- modelId: string | undefined,
1175
- inputTokens: number,
1176
- contextWindowState: KiroContextWindowState,
1177
- nameMap: Map<string, string> | undefined,
1178
- conversationId: string | undefined,
1179
- deferred: AdapterEvent[],
1180
- retention: KiroAttemptRetention,
1181
- attemptCalibration: { value?: { conversationId: string; estimated: number; charged: number } },
1182
- contextInputEstimate?: number,
1183
- priorEmittedOutput = false,
1184
- ): AsyncGenerator<AdapterEvent, KiroAttemptParseResult> {
1185
- const emptyResult = (): KiroAttemptParseResult => ({ assistantText: "", sawReasoning: false });
1186
- // Every early return below is a failure path that stages nothing; only the completion path
1187
- // writes `attemptCalibration`, and the wrapper decides whether to commit it.
1188
- if (!response.body) {
1189
- return {
1190
- ...emptyResult(),
1191
- terminal: { type: "error", message: "Kiro response has no body", status: 502, errorType: "upstream_error" },
1192
- };
1193
- }
1194
-
1195
- let open: { id: string; name: string; chunks: string[]; completion: boolean } | null = null;
1196
- let openCallId: string | undefined;
1197
- const closeOpenCall = () => {
1198
- if (!openCallId) return;
1199
- budget.closeCall(openCallId);
1200
- openCallId = undefined;
1201
- };
1202
- let outputChars = "";
1203
- let outputCharsBytes = 0;
1204
- let contextUsagePercentage: number | undefined;
1205
- let returnedConversationId = conversationId;
1206
- let assistantText = "";
1207
- let assistantTextBytes = 0;
1208
- let sawText = false;
1209
- let sawReasoning = false;
1210
- let sawRealTool = false;
1211
- let completionAnswer: string | undefined;
1212
- let completionCalls = 0;
1213
- let authoritativeUsage: OcxUsage | undefined;
1214
- let stopReason: string | undefined;
1215
- const fallbackEvents: AdapterEvent[] = [];
1216
- const thinking = new KiroThinkingParser(budget);
1217
-
1218
- const retainedEventBytes = (event: AdapterEvent): number => Buffer.byteLength(JSON.stringify(event));
1219
- const retainEvent = (event: AdapterEvent): void => {
1220
- const bytes = retainedEventBytes(event);
1221
- budget.chargeRetained(bytes, { kind: "retained_collectors" });
1222
- retention.retainEvent(event, bytes);
1223
- };
1224
- const emitRetained = async function* (events: Iterable<AdapterEvent>): AsyncGenerator<AdapterEvent> {
1225
- for (const event of events) {
1226
- try { yield event; } finally { retention.releaseEvent(event); }
1227
- }
1228
- };
1229
- // A valid private completion answer supersedes the progress prose staged during the SAME
1230
- // inference: Kiro emits answer-like text and then calls the completion tool, so releasing both
1231
- // makes the bridge close the commentary message and open a second one with near-identical text
1232
- // (#2819 follow-up). Consume the collection instead — drop the redundant text, keep every
1233
- // non-text event, and release retention either way.
1234
- //
1235
- // This is deliberately the ONLY suppression site. The outer drain in `parseKiroAttempt` is also
1236
- // the leftover flush for early terminal returns (stream, protocol, and provider failures), so
1237
- // teaching it to discard text would hide the only commentary a failed turn ever produced.
1238
- // Splicing here leaves that drain empty on the completion path and untouched everywhere else.
1239
- const consumeSupersededByCompletion = async function* (
1240
- events: AdapterEvent[],
1241
- ): AsyncGenerator<AdapterEvent> {
1242
- for (const event of events.splice(0)) {
1243
- try {
1244
- if (event.type !== "text_delta") yield event;
1245
- } finally {
1246
- retention.releaseEvent(event);
1247
- }
1248
- }
1249
- };
1250
-
1251
- const providerState = (): { kiro: { conversationId: string } } | undefined =>
1252
- returnedConversationId ? { kiro: { conversationId: returnedConversationId } } : undefined;
1253
-
1254
- const contextUsageTotalFloor = (): number | undefined => {
1255
- if (contextUsagePercentage === undefined || !contextWindowState.value) return undefined;
1256
- const floor = Math.ceil(contextWindowState.value * Math.min(contextUsagePercentage, 100) / 100);
1257
- return Number.isFinite(floor) && floor > 0 ? floor : undefined;
1258
- };
1259
- const usage = (): OcxUsage => {
1260
- const base = authoritativeUsage ?? {
1261
- inputTokens,
1262
- outputTokens: estimateKiroTokens(outputChars, modelId),
1263
- estimated: true,
1264
- };
1265
- const estimatedContextTotal = contextInputEstimate !== undefined
1266
- ? contextInputEstimate + base.outputTokens
1267
- : undefined;
1268
- const authoritativeTurnTotal = base.inputTokens + base.outputTokens;
1269
- const contextTotal = Math.max(
1270
- estimatedContextTotal ?? 0,
1271
- contextUsageTotalFloor() ?? 0,
1272
- authoritativeTurnTotal,
1273
- );
1274
- return contextTotal > 0 ? { ...base, contextTotalTokens: contextTotal } : base;
1275
- };
1276
-
1277
- const classifiedTerminal = (failure: KiroErrorClassification): AdapterEvent => {
1278
- // Upstream exception/error frames can arrive after commentary was already staged (and will be
1279
- // flushed before this terminal is yielded). Replaying after that content would duplicate it.
1280
- const emittedOutput = priorEmittedOutput
1281
- || sawText
1282
- || sawReasoning
1283
- || sawRealTool
1284
- || assistantText.length > 0
1285
- || deferred.length > 0
1286
- || completionAnswer !== undefined
1287
- || completionCalls > 0
1288
- || open !== null
1289
- || fallbackEvents.length > 0;
1290
- if (failure.status === 429 && failure.retryable) noteKiroTransientThrottle();
1291
- return {
1292
- type: "error",
1293
- message: failure.message,
1294
- status: failure.status,
1295
- errorType: failure.errorType,
1296
- code: failure.code,
1297
- retryable: emittedOutput ? false : failure.retryable,
1298
- usage: usage(),
1299
- };
1300
- };
1301
-
1302
- const protocolTerminal = (message: string, malformedCompletion = false): AdapterEvent => {
1303
- if (mode === "text_fallback" && malformedCompletion) {
1304
- return retryableKiroIncomplete(
1305
- "malformed_kiro_completion",
1306
- message,
1307
- usage(),
1308
- providerState(),
1309
- // First-attempt progress was already flushed before this bounded fallback (#520).
1310
- !priorEmittedOutput,
1311
- );
1312
- }
1313
- return {
1314
- type: "error",
1315
- message,
1316
- status: 502,
1317
- errorType: "upstream_error",
1318
- code: malformedCompletion ? "invalid_kiro_completion" : "kiro_stream_protocol_error",
1319
- retryable: false,
1320
- usage: usage(),
1321
- };
1322
- };
1323
-
1324
- const classifyTool = (
1325
- tool: { id: string; name: string; chunks: string[]; completion: boolean },
1326
- ): AdapterEvent | undefined => {
1327
- if (tool.name !== KIRO_COMPLETION_TOOL_NAME) {
1328
- tool.completion = false;
1329
- return completionAnswer !== undefined || completionCalls > 0
1330
- ? protocolTerminal("Kiro returned a real tool call alongside a private final answer")
1331
- : undefined;
1332
- }
1333
- if (mode === "disabled") {
1334
- return protocolTerminal("Kiro returned the reserved private final-answer tool while explicit completion was disabled");
1335
- }
1336
- tool.completion = true;
1337
- if (completionAnswer !== undefined || completionCalls > 0) {
1338
- return protocolTerminal("Kiro returned more than one private final-answer tool call", true);
1339
- }
1340
- if (sawRealTool) {
1341
- return protocolTerminal("Kiro returned a private final answer alongside a real tool call");
1342
- }
1343
- return undefined;
1344
- };
1345
-
1346
- const beginTool = (
1347
- id: string,
1348
- name: string,
1349
- ): { tool?: { id: string; name: string; chunks: string[]; completion: boolean }; terminal?: AdapterEvent } => {
1350
- const next = { id, name, chunks: [], completion: false };
1351
- const terminal = classifyTool(next);
1352
- return terminal ? { terminal } : { tool: next };
1353
- };
1354
-
1355
- // In `required` mode Kiro's stop reason only arrives on the terminal metadata event, so staged
1356
- // commentary is held until either a real tool call proves the turn continues (flush as
1357
- // commentary) or the stream ends (relabel as the final answer when END_TURN says so). A heartbeat
1358
- // stands in for each held event so the bridge's stall watchdog stays armed.
1359
- const defer = (event: AdapterEvent): AdapterEvent[] => {
1360
- if (sawRealTool) return [...deferred.splice(0), event];
1361
- if (event.type !== "text_delta" && deferred.length === 0) return [event];
1362
- deferred.push(event);
1363
- retainEvent(event);
1364
- return [{ type: "heartbeat" }];
1365
- };
1366
-
1367
- const stage = (event: AdapterEvent): AdapterEvent[] => {
1368
- if (event.type === "text_delta") {
1369
- const nextAssistantTextBytes = appendedUtf8Bytes(assistantText, assistantTextBytes, event.text);
1370
- const assistantReservation = budget.reserveTransient(nextAssistantTextBytes, { kind: "retained_collectors" });
1371
- assistantText += event.text;
1372
- assistantReservation.commitRetained();
1373
- budget.releaseRetained(assistantTextBytes, { kind: "retained_collectors" });
1374
- retention.trackReplacement(assistantTextBytes, nextAssistantTextBytes);
1375
- assistantTextBytes = nextAssistantTextBytes;
1376
- if (event.text.trim()) sawText = true;
1377
- const nextOutputCharsBytes = appendedUtf8Bytes(outputChars, outputCharsBytes, event.text);
1378
- const outputReservation = budget.reserveTransient(nextOutputCharsBytes, { kind: "retained_collectors" });
1379
- outputChars += event.text;
1380
- outputReservation.commitRetained();
1381
- budget.releaseRetained(outputCharsBytes, { kind: "retained_collectors" });
1382
- retention.trackReplacement(outputCharsBytes, nextOutputCharsBytes);
1383
- outputCharsBytes = nextOutputCharsBytes;
1384
- const phased = mode === "disabled"
1385
- ? event
1386
- : { ...event, phase: "commentary" as const };
1387
- if (mode === "text_fallback") {
1388
- fallbackEvents.push(phased);
1389
- retainEvent(phased);
1390
- return [];
1391
- }
1392
- return mode === "required" ? defer(phased) : [phased];
1393
- }
1394
- if (event.type === "reasoning_raw_delta" || event.type === "thinking_delta") {
1395
- const text = event.type === "reasoning_raw_delta" ? event.text : event.thinking;
1396
- if (text.trim()) sawReasoning = true;
1397
- const nextOutputCharsBytes = appendedUtf8Bytes(outputChars, outputCharsBytes, text);
1398
- const reasoningReservation = budget.reserveTransient(nextOutputCharsBytes, { kind: "retained_collectors" });
1399
- outputChars += text;
1400
- reasoningReservation.commitRetained();
1401
- budget.releaseRetained(outputCharsBytes, { kind: "retained_collectors" });
1402
- retention.trackReplacement(outputCharsBytes, nextOutputCharsBytes);
1403
- outputCharsBytes = nextOutputCharsBytes;
1404
- }
1405
- if (mode === "text_fallback" && event.type !== "heartbeat") {
1406
- fallbackEvents.push(event);
1407
- retainEvent(event);
1408
- return [];
1409
- }
1410
- return mode === "required" ? defer(event) : [event];
1411
- };
1412
-
1413
- const parseCompletion = (chunks: string[]): string | Error => {
1414
- const raw = chunks.join("").trim();
1415
- let value: unknown;
1416
- try {
1417
- value = JSON.parse(raw || "{}");
1418
- } catch {
1419
- return new Error("Kiro returned invalid JSON for the private final-answer tool");
1420
- }
1421
- if (!value || typeof value !== "object" || Array.isArray(value)) {
1422
- return new Error("Kiro returned a non-object value for the private final-answer tool");
1423
- }
1424
- const answer = (value as { answer?: unknown }).answer;
1425
- if (typeof answer !== "string" || !answer.trim()) {
1426
- return new Error("Kiro returned an empty final answer");
1427
- }
1428
- return answer;
1429
- };
1430
-
1431
- const flushOpen = (): { events: AdapterEvent[]; terminal?: AdapterEvent } => {
1432
- if (!open) return { events: [] };
1433
- const tool = open;
1434
- open = null;
1435
- closeOpenCall();
1436
- const input = tool.chunks.join("");
1437
- if (!isCompleteKiroToolInput(input)) {
1438
- return { events: [], terminal: protocolTerminal(kiroTruncationErrorMessage("incomplete tool input JSON"), tool.completion) };
1439
- }
1440
- if (tool.completion) {
1441
- completionCalls++;
1442
- if (completionCalls > 1) {
1443
- return { events: [], terminal: protocolTerminal("Kiro returned more than one private final-answer tool call", true) };
1444
- }
1445
- if (sawRealTool) {
1446
- return { events: [], terminal: protocolTerminal("Kiro returned a private final answer alongside a real tool call") };
1447
- }
1448
- const answer = parseCompletion(tool.chunks);
1449
- if (answer instanceof Error) return { events: [], terminal: protocolTerminal(answer.message, true) };
1450
- completionAnswer = answer;
1451
- return { events: [] };
1452
- }
1453
- if (completionAnswer !== undefined || completionCalls > 0) {
1454
- return { events: [], terminal: protocolTerminal("Kiro returned a real tool call alongside a private final answer") };
1455
- }
1456
- sawRealTool = true;
1457
- const restored = nameMap?.get(tool.name) ?? tool.name;
1458
- return {
1459
- events: [
1460
- { type: "tool_call_start", id: tool.id, name: restored },
1461
- ...tool.chunks.filter(Boolean).map(argumentsChunk => ({ type: "tool_call_delta", arguments: argumentsChunk }) as AdapterEvent),
1462
- { type: "tool_call_end" },
1463
- ],
1464
- };
1465
- };
1466
-
1467
- try {
1468
- for await (const msg of decodeEventStream(response.body)) {
1469
- const mt = msg.headers[":message-type"];
1470
- if (mt === "exception" || mt === "error") {
1471
- open = null;
1472
- return {
1473
- assistantText,
1474
- sawReasoning,
1475
- terminal: classifiedTerminal(classifyKiroStreamError(msg.headers, new TextDecoder().decode(msg.payload))),
1476
- };
1477
- }
1478
- if (mt !== "event") {
1479
- open = null;
1480
- return {
1481
- assistantText,
1482
- sawReasoning,
1483
- terminal: protocolTerminal(`Kiro response protocol error: unsupported Smithy message type ${JSON.stringify(mt ?? "missing")}`),
1484
- };
1485
- }
1486
- const eventType = msg.headers[":event-type"];
1487
- if (!eventType) {
1488
- open = null;
1489
- return { assistantText, sawReasoning, terminal: protocolTerminal("Kiro response protocol error: event is missing :event-type") };
1490
- }
1491
- const ev = parseKiroEvent(eventType, msg.payload);
1492
- if (!ev) continue;
1493
- switch (ev.type) {
1494
- case "metadata":
1495
- if (ev.usage) authoritativeUsage = ev.usage;
1496
- if (ev.contextUsagePercentage !== undefined && ev.contextUsagePercentage > 0) {
1497
- contextUsagePercentage = ev.contextUsagePercentage;
1498
- }
1499
- if (ev.stopReason !== undefined) stopReason = ev.stopReason;
1500
- break;
1501
- case "message_metadata":
1502
- if (isValidKiroConversationId(ev.conversationId)) {
1503
- // Kiro can answer under a different conversation id than the request was built with.
1504
- // Carry the calibration entry across so the record below finds its own raw estimate
1505
- // instead of silently falling back to the already-corrected value.
1506
- rekeyKiroCalibration(returnedConversationId, ev.conversationId);
1507
- returnedConversationId = ev.conversationId;
1508
- }
1509
- break;
1510
- case "content":
1511
- if (ev.modelId) {
1512
- contextWindowState.value = kiroUpstreamContextWindow(ev.modelId) ?? contextWindowState.value;
1513
- }
1514
- if (open) {
1515
- open = null;
1516
- return { assistantText, sawReasoning, terminal: protocolTerminal(kiroTruncationErrorMessage("content arrived before tool stop")) };
1517
- }
1518
- if (ev.data) {
1519
- for (const contentEvent of thinking.feed(ev.data)) {
1520
- yield* emitRetained(stage(contentEvent));
1521
- }
1522
- }
1523
- break;
1524
- case "reasoning":
1525
- for (const contentEvent of thinking.flush()) {
1526
- yield* emitRetained(stage(contentEvent));
1527
- }
1528
- if (ev.data) {
1529
- yield* emitRetained(stage({ type: "reasoning_raw_delta", text: ev.data }));
1530
- }
1531
- if (ev.redactedContent) {
1532
- yield* emitRetained(stage({ type: "kiro_redacted_reasoning", data: ev.redactedContent }));
1533
- }
1534
- break;
1535
- case "context_usage":
1536
- if (ev.contextUsagePercentage > 0) contextUsagePercentage = ev.contextUsagePercentage;
1537
- break;
1538
- case "tool": {
1539
- for (const contentEvent of thinking.flush()) {
1540
- yield* emitRetained(stage(contentEvent));
1541
- }
1542
- if (!open) {
1543
- if (ev.stop === true) {
1544
- return { assistantText, sawReasoning, terminal: protocolTerminal("Kiro response protocol error: tool stop received without an open tool call") };
1545
- }
1546
- if (!ev.toolUseId || !ev.name) {
1547
- return { assistantText, sawReasoning, terminal: protocolTerminal("Kiro response protocol error: new tool event is missing toolUseId or name") };
1548
- }
1549
- const started = beginTool(ev.toolUseId, ev.name);
1550
- if (started.terminal) return { assistantText, sawReasoning, terminal: started.terminal };
1551
- open = started.tool!;
1552
- budget.openCall(open.id);
1553
- openCallId = open.id;
1554
- } else if (
1555
- (ev.toolUseId && ev.toolUseId !== open.id)
1556
- || (ev.name && open.name !== "unknown" && ev.name !== open.name)
1557
- ) {
1558
- closeOpenCall();
1559
- open = null;
1560
- return { assistantText, sawReasoning, terminal: protocolTerminal(kiroTruncationErrorMessage("tool input changed identity before stop")) };
1561
- }
1562
- if (open && open.name === "unknown" && ev.name) {
1563
- open.name = ev.name;
1564
- const terminal = classifyTool(open);
1565
- if (terminal) {
1566
- open = null;
1567
- return { assistantText, sawReasoning, terminal };
1568
- }
1569
- }
1570
- if (open && ev.input !== undefined) {
1571
- const previousCallBytes = open.chunks.reduce((total, chunk) => total + Buffer.byteLength(chunk), 0);
1572
- const nextCallBytes = previousCallBytes + Buffer.byteLength(ev.input);
1573
- const callReservation = budget.reserveTransient(nextCallBytes, { kind: "tool_args", callId: open.id });
1574
- open.chunks.push(ev.input);
1575
- callReservation.commitRetained();
1576
- budget.releaseRetained(previousCallBytes, { kind: "tool_args", callId: open.id });
1577
- const nextOutputCharsBytes = appendedUtf8Bytes(outputChars, outputCharsBytes, ev.input);
1578
- const toolOutputReservation = budget.reserveTransient(nextOutputCharsBytes, { kind: "retained_collectors" });
1579
- outputChars += ev.input;
1580
- toolOutputReservation.commitRetained();
1581
- budget.releaseRetained(outputCharsBytes, { kind: "retained_collectors" });
1582
- retention.trackReplacement(outputCharsBytes, nextOutputCharsBytes);
1583
- outputCharsBytes = nextOutputCharsBytes;
1584
- }
1585
- if (ev.stop === true) {
1586
- const flushed = flushOpen();
1587
- if (flushed.terminal) return { assistantText, sawReasoning, terminal: flushed.terminal };
1588
- for (const event of flushed.events) {
1589
- yield* emitRetained(stage(event));
1590
- }
1591
- } else {
1592
- yield { type: "heartbeat" };
1593
- }
1594
- break;
1595
- }
1596
- case "invalid_state":
1597
- open = null;
1598
- return { assistantText, sawReasoning, terminal: classifiedTerminal(classifyKiroEventError(undefined, ev.message ?? "Kiro entered an invalid state")) };
1599
- case "error":
1600
- open = null;
1601
- return { assistantText, sawReasoning, terminal: classifiedTerminal(classifyKiroEventError(ev.reason, ev.message)) };
1602
- case "truncation":
1603
- open = null;
1604
- return { assistantText, sawReasoning, terminal: protocolTerminal(kiroTruncationErrorMessage(ev.data)) };
1605
- }
1606
- }
1607
-
1608
- for (const contentEvent of thinking.flush()) {
1609
- yield* emitRetained(stage(contentEvent));
1610
- }
1611
- if (open) {
1612
- const input = open.chunks.join("");
1613
- if (!isCompleteKiroToolInput(input)) {
1614
- const privateTool = open.completion;
1615
- open = null;
1616
- return {
1617
- assistantText,
1618
- sawReasoning,
1619
- terminal: protocolTerminal(kiroTruncationErrorMessage("stream ended before tool stop"), privateTool),
1620
- };
1621
- }
1622
- const flushed = flushOpen();
1623
- if (flushed.terminal) return { assistantText, sawReasoning, terminal: flushed.terminal };
1624
- for (const event of flushed.events) {
1625
- yield* emitRetained(stage(event));
1626
- }
1627
- }
1628
-
1629
- const finalUsage = usage();
1630
- const finalProviderState = providerState();
1631
- if (contextUsagePercentage !== undefined) {
1632
- debugProviderDiagnostic("kiro", "context_usage", {
1633
- contextUsagePercentage,
1634
- ...(contextWindowState.value ? { upstreamContextWindow: contextWindowState.value } : {}),
1635
- });
1636
- }
1637
- // Upstream just told us what this payload cost. The ratio between that and our pre-request
1638
- // estimate is this conversation's own measured error, and it is the only feedback the
1639
- // estimator ever receives.
1640
- //
1641
- // Staged, not recorded. An attempt that sets `needsFallback` is not over: the adapter rebuilds
1642
- // the payload and streams a second time for the SAME user turn. Learning here would apply the
1643
- // fresh factor to that rebuild and then learn again from it, so one turn would move the factor
1644
- // twice and the second observation would score a payload the first had already inflated. Only
1645
- // the outer parser knows whether an attempt is terminal, so it commits.
1646
- //
1647
- // Subtract the output first. `contextUsageTotalFloor` is the absolute context size AFTER the
1648
- // response (`OcxUsage.contextTotalTokens`, types/request.ts), while `contextInputEstimate`
1649
- // covers the request payload alone. Dividing one by the other would charge generated tokens to
1650
- // prompt-tokenization error, so a short prompt answered at length would learn a large factor
1651
- // and inflate every later request in that conversation — the premature compaction this work
1652
- // exists to prevent.
1653
- const chargedTotal = contextUsageTotalFloor();
1654
- if (chargedTotal !== undefined && contextInputEstimate !== undefined) {
1655
- const chargedInput = chargedTotal - finalUsage.outputTokens;
1656
- if (chargedInput > 0 && returnedConversationId) {
1657
- attemptCalibration.value = { conversationId: returnedConversationId, estimated: contextInputEstimate, charged: chargedInput };
1658
- }
1659
- }
1660
- // Native stop metadata proves that this inference ended, but it does not prove that ordinary
1661
- // text is a final answer. Kiro has emitted END_TURN for progress prose, so tool-enabled turns
1662
- // still require the private completion call to distinguish commentary from completion (#531).
1663
- const normalizedStopReason = stopReason?.trim().toUpperCase();
1664
- const nativeCompletionStop = (normalizedStopReason === KIRO_END_TURN_STOP_REASON
1665
- || normalizedStopReason === "STOP_SEQUENCE")
1666
- && sawText
1667
- && !sawRealTool
1668
- && completionAnswer === undefined
1669
- && completionCalls === 0;
1670
-
1671
- debugProviderDiagnostic("kiro", "attempt_complete", {
1672
- mode,
1673
- sawText,
1674
- sawReasoning,
1675
- sawRealTool,
1676
- completionCalls,
1677
- nativeCompletionStop,
1678
- ...(stopReason !== undefined ? { stopReason } : {}),
1679
- assistantChars: assistantText.length,
1680
- });
1681
-
1682
- if (mode === "required") {
1683
- // A valid completion answer makes this inference's staged prose redundant; anything else
1684
- // still flushes exactly as before (bounded fallback, explicit stops, real tool calls).
1685
- if (completionAnswer !== undefined) yield* consumeSupersededByCompletion(deferred);
1686
- else yield* emitRetained(deferred.splice(0));
1687
- }
1688
-
1689
- if (mode === "text_fallback") {
1690
- if (completionAnswer !== undefined) {
1691
- yield* consumeSupersededByCompletion(fallbackEvents);
1692
- yield { type: "text_delta", text: completionAnswer, phase: "final_answer" };
1693
- return {
1694
- assistantText,
1695
- sawReasoning,
1696
- terminal: { type: "done", usage: finalUsage, endTurn: true, ...(finalProviderState ? { providerState: finalProviderState } : {}) },
1697
- };
1698
- }
1699
- if (sawRealTool) {
1700
- yield* emitRetained(fallbackEvents);
1701
- return {
1702
- assistantText,
1703
- sawReasoning,
1704
- terminal: { type: "done", usage: finalUsage, endTurn: false, ...(finalProviderState ? { providerState: finalProviderState } : {}) },
1705
- };
1706
- }
1707
- if (sawText) {
1708
- for (const event of fallbackEvents) {
1709
- try {
1710
- if (event.type !== "text_delta") yield event;
1711
- else yield { ...event, phase: "final_answer" };
1712
- } finally {
1713
- retention.releaseEvent(event);
1714
- }
1715
- }
1716
- return {
1717
- assistantText,
1718
- sawReasoning,
1719
- terminal: { type: "done", usage: finalUsage, endTurn: true, ...(finalProviderState ? { providerState: finalProviderState } : {}) },
1720
- };
1721
- }
1722
- yield* emitRetained(fallbackEvents);
1723
- return {
1724
- assistantText,
1725
- sawReasoning,
1726
- terminal: retryableKiroIncomplete(
1727
- sawReasoning ? "reasoning_only_kiro_fallback" : "empty_kiro_fallback",
1728
- sawReasoning
1729
- ? "Kiro produced reasoning but no final answer on its bounded completion retry"
1730
- : "Kiro produced no final answer on its bounded completion retry",
1731
- finalUsage,
1732
- finalProviderState,
1733
- // First-attempt progress was already flushed before this bounded fallback (#520).
1734
- !priorEmittedOutput,
1735
- ),
1736
- };
1737
- }
1738
-
1739
- if (completionAnswer !== undefined) {
1740
- yield { type: "text_delta", text: completionAnswer, phase: "final_answer" };
1741
- return {
1742
- assistantText,
1743
- sawReasoning,
1744
- terminal: { type: "done", usage: finalUsage, endTurn: true, ...(finalProviderState ? { providerState: finalProviderState } : {}) },
1745
- };
1746
- }
1747
- if (sawRealTool) {
1748
- return {
1749
- assistantText,
1750
- sawReasoning,
1751
- terminal: { type: "done", usage: finalUsage, endTurn: false, ...(finalProviderState ? { providerState: finalProviderState } : {}) },
1752
- };
1753
- }
1754
- if (mode === "required" && nativeCompletionStop) {
1755
- return {
1756
- assistantText,
1757
- sawReasoning,
1758
- needsFallback: true,
1759
- usage: finalUsage,
1760
- providerState: finalProviderState,
1761
- };
1762
- }
1763
-
1764
- // An explicit non-completion stop reason has already terminated this inference. Converting it into
1765
- // another model request would hide truncation behind a second paid call, and for context
1766
- // exhaustion it would resubmit a request that cannot fit. Only a MISSING stop reason falls
1767
- // through to the bounded compatibility fallback below.
1768
- //
1769
- // END_TURN and STOP_SEQUENCE with text take the bounded validation path above; reaching here
1770
- // with either means the turn produced no replayable text.
1771
- if (mode === "required" && normalizedStopReason !== undefined) {
1772
- const providerStateField = finalProviderState ? { providerState: finalProviderState } : {};
1773
- const incomplete = (reason: string, retryable: boolean) => ({
1774
- assistantText,
1775
- sawReasoning,
1776
- terminal: {
1777
- type: "incomplete" as const,
1778
- reason,
1779
- message: `Kiro stopped with ${normalizedStopReason} before an explicit final answer`,
1780
- usage: finalUsage,
1781
- retryable,
1782
- endTurn: false,
1783
- ...providerStateField,
1784
- },
1785
- });
1786
-
1787
- if (normalizedStopReason === "MODEL_CONTEXT_WINDOW_EXCEEDED") {
1788
- // Reuse the existing context-length contract (kiro-errors.ts) instead of inventing an
1789
- // incomplete reason: an unrecognized incomplete becomes a retryable 529 in Claude
1790
- // outbound, and `max_output_tokens` would make responses/state.ts cache this partial
1791
- // for continuation replay. Both invite a retry that cannot succeed.
1792
- return {
1793
- assistantText,
1794
- sawReasoning,
1795
- terminal: {
1796
- type: "error" as const,
1797
- message: "Kiro stopped because the model context window was exhausted",
1798
- status: 400,
1799
- errorType: "invalid_request_error",
1800
- code: "context_length_exceeded",
1801
- retryable: false,
1802
- usage: finalUsage,
1803
- },
1804
- };
1805
- }
1806
- if (normalizedStopReason === "MAX_TOKENS") return incomplete("max_output_tokens", true);
1807
- if (normalizedStopReason === "CONTENT_FILTERED" || normalizedStopReason === "GUARDRAIL_INTERVENED") {
1808
- return incomplete("content_filter", false);
1809
- }
1810
- if (normalizedStopReason === "MALFORMED_TOOL_USE") return incomplete("kiro_malformed_tool_use", false);
1811
- if (normalizedStopReason === "MALFORMED_MODEL_OUTPUT") return incomplete("kiro_malformed_model_output", false);
1812
- // TOOL_USE here means Kiro claimed a tool call it never emitted.
1813
- if (normalizedStopReason === "TOOL_USE") return incomplete("kiro_tool_use_without_call", false);
1814
- if (normalizedStopReason === KIRO_END_TURN_STOP_REASON || normalizedStopReason === "STOP_SEQUENCE") {
1815
- return incomplete(`kiro_${normalizedStopReason.toLowerCase()}_without_text`, false);
1816
- }
1817
- return incomplete(`kiro_${normalizedStopReason.toLowerCase() || "unknown_stop"}`, false);
1818
- }
1819
- // Kiro text has no trustworthy final/progress marker. When completion is required, ordinary
1820
- // text and reasoning remain unfinished until the one bounded fallback validates the turn.
1821
- if (mode === "required" && (sawText || sawReasoning)) {
1822
- return { assistantText, sawReasoning, needsFallback: true, usage: finalUsage, providerState: finalProviderState };
1823
- }
1824
- if (!sawText && !sawReasoning) {
1825
- return {
1826
- assistantText,
1827
- sawReasoning,
1828
- terminal: retryableKiroIncomplete(
1829
- "empty_kiro_stream",
1830
- "Kiro returned a successful but empty response stream",
1831
- finalUsage,
1832
- finalProviderState,
1833
- ),
1834
- };
1835
- }
1836
- return {
1837
- assistantText,
1838
- sawReasoning,
1839
- terminal: {
1840
- type: "done",
1841
- usage: finalUsage,
1842
- endTurn: mode === "disabled" ? sawText : false,
1843
- ...(finalProviderState ? { providerState: finalProviderState } : {}),
1844
- },
1845
- };
1846
- } catch (err) {
1847
- if (isTranslatorBudgetExceededError(err)) {
1848
- closeOpenCall();
1849
- return {
1850
- assistantText,
1851
- sawReasoning,
1852
- terminal: {
1853
- type: "error",
1854
- status: 502,
1855
- errorType: "upstream_error",
1856
- code: "translation_buffer_limit",
1857
- message: "upstream translation buffer exceeded the safe limit",
1858
- },
1859
- };
1860
- }
1861
- // Mid-stream socket closes after response.created / heartbeats only must stay retryable:
1862
- // nothing was relayed to the client, so a string-body replay is safe (see #519 / cursor's
1863
- // emittedOutput gate). Once any assistant text, reasoning, tool, or deferred content exists
1864
- // — including content flushed by a prior attempt before a bounded fallback — fail closed;
1865
- // the client may already have partial output. Protocol parse throws stay non-retryable even
1866
- // with zero output.
1867
- const emittedOutput = priorEmittedOutput
1868
- || sawText
1869
- || sawReasoning
1870
- || sawRealTool
1871
- || assistantText.length > 0
1872
- || deferred.length > 0
1873
- || completionAnswer !== undefined
1874
- || completionCalls > 0
1875
- || open !== null
1876
- || fallbackEvents.length > 0;
1877
- return {
1878
- assistantText,
1879
- sawReasoning,
1880
- terminal: {
1881
- type: "error",
1882
- message: safeKiroErrorMessage({}, err instanceof Error ? err.message : String(err)),
1883
- status: 502,
1884
- errorType: "server_error",
1885
- code: "kiro_stream_protocol_error",
1886
- retryable: isRetryableKiroStreamCatchError(err, emittedOutput),
1887
- usage: usage(),
1888
- },
1889
- };
1890
- } finally {
1891
- thinking.dispose();
1892
- closeOpenCall();
1893
- }
1894
- }
1895
-
1896
- export async function* parseKiroStream(
1897
- response: Response,
1898
- budget: TranslatorBudget,
1899
- modelId?: string,
1900
- inputTokens = 0,
1901
- contextWindow?: number,
1902
- nameMap?: Map<string, string>,
1903
- conversationId?: string,
1904
- completionMode: KiroCompletionMode = "disabled",
1905
- fallbackFactory?: KiroFallbackFactory,
1906
- contextInputEstimate?: number,
1907
- ): AsyncGenerator<AdapterEvent> {
1908
- const contextWindowState: KiroContextWindowState = { value: contextWindow };
1909
- const firstResult = yield* parseKiroAttempt(
1910
- response,
1911
- budget,
1912
- completionMode,
1913
- modelId,
1914
- inputTokens,
1915
- contextWindowState,
1916
- nameMap,
1917
- conversationId,
1918
- contextInputEstimate,
1919
- false,
1920
- );
1921
- try {
1922
- if (!firstResult.needsFallback) {
1923
- if (firstResult.terminal) yield firstResult.terminal;
1924
- return;
1925
- }
1926
- if (!fallbackFactory) {
1927
- yield retryableKiroIncomplete(
1928
- "uncompleted_kiro_response",
1929
- "Kiro produced progress without an explicit final answer and no bounded retry transport was available",
1930
- firstResult.usage ?? { inputTokens, outputTokens: 0, estimated: true },
1931
- firstResult.providerState,
1932
- );
1933
- return;
1934
- }
1935
-
1936
- yield { type: "heartbeat" };
1937
- // First attempt already flushed deferred progress before this point. Gate fallback
1938
- // setup/HTTP failures the same way as the second-stream catch so a replay cannot
1939
- // duplicate visible commentary (#520).
1940
- const priorEmittedOutput = Boolean(firstResult.assistantText.trim()) || firstResult.sawReasoning;
1941
- let firstAssistantText = firstResult.assistantText;
1942
- const firstHadAssistantText = firstAssistantText.length > 0;
1943
- let fallback: KiroFallbackAttempt;
1944
- try {
1945
- fallback = await fallbackFactory(
1946
- firstResult.providerState?.kiro.conversationId ?? conversationId,
1947
- firstAssistantText,
1948
- firstResult.sawReasoning,
1949
- budget,
1950
- );
1951
- } catch (err) {
1952
- firstAssistantText = "";
1953
- firstResult.assistantText = "";
1954
- firstResult.releaseRetained();
1955
- if (isTranslatorBudgetExceededError(err)) {
1956
- yield {
1957
- type: "error",
1958
- message: "upstream translation buffer exceeded the safe limit",
1959
- status: 502,
1960
- errorType: "upstream_error",
1961
- code: "translation_buffer_limit",
1962
- usage: firstResult.usage,
1963
- };
1964
- return;
1965
- }
1966
- yield {
1967
- type: "error",
1968
- message: safeKiroErrorMessage({}, err instanceof Error ? err.message : String(err)),
1969
- status: err instanceof Error && err.name === "TimeoutError" ? 504 : 502,
1970
- errorType: "upstream_error",
1971
- retryable: !priorEmittedOutput,
1972
- usage: firstResult.usage,
1973
- };
1974
- return;
1975
- }
1976
- // The factory has finished using the live first-attempt alias and has retained its own retry
1977
- // serialization through the fetch boundary. The discarded parser collectors can now release
1978
- // before the second attempt begins on the same turn budget.
1979
- firstAssistantText = "";
1980
- firstResult.assistantText = "";
1981
- firstResult.releaseRetained();
1982
- fallback.releaseRequestBody?.();
1983
- if (!fallback.response.ok) {
1984
- const payload = await fallback.response.text().catch(() => "");
1985
- const failure = classifyKiroHttpError(fallback.response.status, fallback.response.headers, payload);
1986
- yield {
1987
- type: "error",
1988
- message: failure.message,
1989
- status: failure.status,
1990
- errorType: failure.errorType,
1991
- code: failure.code,
1992
- retryable: priorEmittedOutput ? false : failure.retryable,
1993
- usage: firstResult.usage,
1994
- };
1995
- return;
1996
- }
1997
-
1998
- const secondResult = yield* parseKiroAttempt(
1999
- fallback.response,
2000
- budget,
2001
- "text_fallback",
2002
- modelId,
2003
- fallback.inputTokens,
2004
- contextWindowState,
2005
- fallback.nameMap,
2006
- fallback.conversationId,
2007
- fallback.contextInputEstimate,
2008
- // First attempt already flushed deferred progress to the client before this fallback.
2009
- // A zero-output transport failure here must stay non-retryable to avoid duplicating that text.
2010
- priorEmittedOutput,
2011
- );
2012
- try {
2013
- if (!secondResult.terminal) {
2014
- yield retryableKiroIncomplete(
2015
- "empty_kiro_fallback",
2016
- "Kiro's bounded completion retry ended without a terminal result",
2017
- mergeKiroUsage(firstResult.usage, secondResult.usage, firstHadAssistantText)
2018
- ?? { inputTokens, outputTokens: 0, estimated: true },
2019
- secondResult.providerState ?? firstResult.providerState,
2020
- !priorEmittedOutput,
2021
- );
2022
- return;
2023
- }
2024
- if (secondResult.terminal.type === "done" || secondResult.terminal.type === "incomplete") {
2025
- yield {
2026
- ...secondResult.terminal,
2027
- // Belt-and-suspenders: never advertise a replay-safe incomplete after flushed progress.
2028
- ...(secondResult.terminal.type === "incomplete" && priorEmittedOutput
2029
- ? { retryable: false as const }
2030
- : {}),
2031
- usage: mergeKiroUsage(firstResult.usage, secondResult.terminal.usage, firstHadAssistantText),
2032
- providerState: secondResult.terminal.providerState ?? firstResult.providerState,
2033
- };
2034
- return;
2035
- }
2036
- yield {
2037
- ...secondResult.terminal,
2038
- ...(secondResult.terminal.type === "error"
2039
- ? { usage: mergeKiroUsage(firstResult.usage, secondResult.terminal.usage, firstHadAssistantText) }
2040
- : {}),
2041
- };
2042
- } finally {
2043
- secondResult.releaseRetained();
2044
- }
2045
- } finally {
2046
- firstResult.releaseRetained();
2047
- }
2048
- }
2049
-
2050
- // Adapter
2051
- export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter {
2052
- // Per-request closure (resolveAdapter builds a fresh adapter per request — server.ts:440 — so this
2053
- // is race-free) carrying the heuristic input-token estimate from buildRequest into the stream.
2054
- let inputTokens = 0;
2055
- let contextInputEstimate = 0;
2056
- let modelId: string | undefined;
2057
- let contextWindow: number | undefined;
2058
- let toolNameMap: Map<string, string> | undefined;
2059
- let conversationId: string | undefined;
2060
- let completionMode: KiroCompletionMode = "disabled";
2061
- let requestSnapshot: OcxParsedRequest | undefined;
2062
- let firstRequestBodyBytes = 0;
2063
- let requestAbortSignal: AbortSignal | undefined;
2064
-
2065
- const build = async (
2066
- parsed: OcxParsedRequest,
2067
- forcedCompletionMode?: KiroCompletionMode,
2068
- ): Promise<{
2069
- request: AdapterRequest;
2070
- nameMap: Map<string, string>;
2071
- conversationId: string;
2072
- completionMode: KiroCompletionMode;
2073
- inputTokens: number;
2074
- contextInputEstimate: number;
2075
- }> => {
2076
- if (typeof provider.apiKey !== "string" || provider.apiKey.trim() === "") {
2077
- throw new Error("kiro token missing — run ocx login kiro");
2078
- }
2079
- const region = resolveKiroApiRegion(parsed._kiroAuthContext);
2080
- // Request-scoped: an AWS Builder ID account has no profile of its own and resolves to Kiro's
2081
- // fixed service profile here, without that value ever becoming the account's stored identity.
2082
- const requestProfile = resolveKiroRequestProfile(parsed._kiroAuthContext);
2083
- const resolvedProfileArn = requestProfile.profileArn;
2084
- const isApiKey = provider.apiKey.trim().startsWith("ksk_");
2085
- const profileArn = isApiKey ? undefined : resolvedProfileArn;
2086
- // Builder ID and Kiro API keys are accepted only on Kiro's CLI request path; enterprise
2087
- // profiles retain the IDE-shaped request. Builder ID now carries a profile ARN, so a truthy
2088
- // `profileArn` no longer implies "enterprise". The wire path reads the resolver's own verdict
2089
- // rather than re-deriving it, so the accountless path — where the auth type comes from the
2090
- // local import, not the request context — cannot send the fallback inside an IDE-shaped call.
2091
- const isBuilderId = requestProfile.builderIdFallback;
2092
- const wireClient: KiroWireClient = isApiKey || isBuilderId || !profileArn ? "cli" : "ide";
2093
- const fp = fingerprint().slice(0, 64);
2094
- const headers: Record<string, string> = wireClient === "cli" ? {
2095
- authorization: `Bearer ${provider.apiKey}`,
2096
- "content-type": "application/x-amz-json-1.0",
2097
- accept: "*/*",
2098
- "x-amz-target": AMZ_TARGET,
2099
- "user-agent": kiroCliUserAgent(true),
2100
- "x-amz-user-agent": kiroCliUserAgent(false),
2101
- "x-amzn-codewhisperer-optout": "true",
2102
- "amz-sdk-request": "attempt=1; max=3",
2103
- "amz-sdk-invocation-id": invocationId(),
2104
- ...(isApiKey ? { tokentype: "API_KEY" } : {}),
2105
- } : {
2106
- authorization: `Bearer ${provider.apiKey}`,
2107
- "content-type": "application/x-amz-json-1.0",
2108
- accept: "application/vnd.amazon.eventstream",
2109
- "x-amz-target": AMZ_TARGET,
2110
- "user-agent": `aws-sdk-js/${SDK_VERSION} ua/2.1 os/${osTag()} lang/js md/nodejs#${NODE_VERSION} api/codewhispererstreaming#${SDK_VERSION} m/E KiroIDE-${KIRO_IDE_VERSION}-${fp}`,
2111
- "x-amz-user-agent": `aws-sdk-js/${SDK_VERSION} KiroIDE-${KIRO_IDE_VERSION}-${fp}`,
2112
- "x-amzn-codewhisperer-optout": "true",
2113
- "x-amzn-kiro-agent-mode": "vibe",
2114
- "amz-sdk-invocation-id": invocationId(),
2115
- };
2116
- if (profileArn) headers["x-amzn-kiro-profile-arn"] = profileArn;
2117
- const built = buildKiroPayload(parsed, profileArn, forcedCompletionMode, wireClient);
2118
- await normalizeKiroImages(built.payload);
2119
- // Apply what earlier turns of THIS conversation measured. An unseen conversation is
2120
- // unchanged, so a first turn behaves exactly as it would without calibration.
2121
- const rawContextInputEstimate = estimateKiroPayloadInputTokens(built.payload, parsed.modelId);
2122
- const contextInputEstimate = calibrateKiroEstimate(built.conversationId, rawContextInputEstimate);
2123
- const body = JSON.stringify(built.payload);
2124
- // Every field below is evaluated before the call, so an unguarded call re-encodes the
2125
- // whole request body on each request even when provider debug is off. Gate the details.
2126
- if (isDebugEnabled()) {
2127
- debugProviderDiagnostic("kiro", "request", {
2128
- region,
2129
- requestedModel: parsed.modelId,
2130
- completionMode: built.completionMode,
2131
- bodyBytes: new TextEncoder().encode(body).length,
2132
- messageCount: kiroPayloadMessages(parsed).length,
2133
- toolCount: parsed.context.tools?.length ?? 0,
2134
- hasProfileArn: Boolean(profileArn),
2135
- wireClient,
2136
- hasPreviousResponseId: Boolean(parsed.previousResponseId),
2137
- });
2138
- }
2139
- return {
2140
- request: {
2141
- url: kiroRuntimeEndpoint(provider, region),
2142
- method: "POST",
2143
- headers,
2144
- body,
2145
- usageLog: { inputTokens: estimateKiroLogInputTokens(parsed), estimated: true },
2146
- },
2147
- nameMap: built.nameMap,
2148
- conversationId: built.conversationId,
2149
- completionMode: built.completionMode,
2150
- inputTokens: estimateKiroInputTokens(parsed),
2151
- contextInputEstimate,
2152
- };
2153
- };
2154
-
2155
- const fallbackFactory: KiroFallbackFactory = async (
2156
- returnedConversationId,
2157
- assistantText,
2158
- _sawReasoning,
2159
- budget,
2160
- ) => {
2161
- if (!requestSnapshot) throw new Error("Kiro completion retry lost its request state");
2162
- if (requestAbortSignal?.aborted) {
2163
- throw requestAbortSignal.reason instanceof Error
2164
- ? requestAbortSignal.reason
2165
- : new DOMException("Kiro request was cancelled", "AbortError");
2166
- }
2167
- const retryParsed = structuredClone(requestSnapshot);
2168
- retryParsed._providerContinuation = {
2169
- ...(retryParsed._providerContinuation ?? {}),
2170
- ...(returnedConversationId ? { kiro: { conversationId: returnedConversationId } } : {}),
2171
- };
2172
- // Reasoning is not replayable on the Kiro wire. Adding an empty assistant turn merely to mark
2173
- // that reasoning existed creates REQUEST_BODY_INVALID; only visible text earns a replay turn.
2174
- if (assistantText.trim()) {
2175
- retryParsed.context.messages.push({
2176
- role: "assistant",
2177
- content: [{ type: "text" as const, text: assistantText }],
2178
- phase: "commentary",
2179
- model: retryParsed.modelId,
2180
- timestamp: Date.now(),
2181
- });
2182
- }
2183
- // The retry starts from the already measured first wire body, adds one JSON-escaped replay
2184
- // string, and only changes bounded Kiro-owned fields (completion prompt/tool, history wrapper,
2185
- // and <=256-byte conversation id). 64 KiB is a conservative envelope for those fixed fields.
2186
- // Reserve that complete upper bound while the first-attempt collectors are still charged so a
2187
- // near-cap turn fails before build() can materialize the retry payload or serialized body.
2188
- const retryBodyUpperBound = firstRequestBodyBytes
2189
- + jsonStringSerializedUtf8Bytes(assistantText)
2190
- + KIRO_FALLBACK_SERIALIZATION_ENVELOPE_BYTES;
2191
- const retryBodyReservation = budget.reserveTransient(retryBodyUpperBound, { kind: "request_copies" });
2192
- let retryBodyBytes = 0;
2193
- let retryBodyRetained = false;
2194
- let requestBodyReleased = false;
2195
- const releaseRequestBody = () => {
2196
- if (requestBodyReleased) return;
2197
- requestBodyReleased = true;
2198
- if (retryBodyRetained) budget.releaseRetained(retryBodyBytes, { kind: "request_copies" });
2199
- else retryBodyReservation.release();
2200
- };
2201
- try {
2202
- const retry = await build(retryParsed, "text_fallback");
2203
- retryBodyBytes = Buffer.byteLength(retry.request.body);
2204
- if (retryBodyBytes > retryBodyUpperBound) {
2205
- throw new Error("Kiro retry serialization exceeded its pre-admitted upper bound");
2206
- }
2207
- retryBodyReservation.commitRetained();
2208
- retryBodyRetained = true;
2209
- budget.releaseRetained(retryBodyUpperBound - retryBodyBytes, { kind: "request_copies" });
2210
- const response = await fetchKiroWithRetry(retry.request, {
2211
- abortSignal: requestAbortSignal,
2212
- returnRawErrors: true,
2213
- stream: true,
2214
- });
2215
- return {
2216
- response,
2217
- inputTokens: retry.inputTokens,
2218
- contextInputEstimate: retry.contextInputEstimate,
2219
- nameMap: retry.nameMap,
2220
- conversationId: retry.conversationId,
2221
- releaseRequestBody,
2222
- };
2223
- } catch (error) {
2224
- releaseRequestBody();
2225
- throw error;
2226
- }
2227
- };
2228
-
2229
- return {
2230
- name: "kiro",
2231
- // A replayed history that already ENDS with a delivered final answer has nothing to ask Kiro.
2232
- // Before this hook the adapter still appended a trailing user turn — a neutral acknowledgement,
2233
- // but structurally still a prompt — and performed a real inference, so the model answered the
2234
- // closed task again and the finished turn behaved like a still-open goal.
2235
- //
2236
- // Suppressing the completion contract (above) removed the instruction to complete; it could not
2237
- // remove the inference. This is the boundary: no request is built, nothing is sent, and no token
2238
- // estimate is recorded.
2239
- //
2240
- // The forced-fallback build is deliberately NOT consulted here: this hook runs on the inbound
2241
- // turn only, and the adapter-owned bounded retry passes "text_fallback" through `build`
2242
- // directly, never through this path.
2243
- localTerminal(parsed: OcxParsedRequest) {
2244
- return hasTrailingDeliveredFinalAnswer(kiroPayloadMessages(parsed), parsed)
2245
- ? { reason: "kiro_final_answer_already_delivered" }
2246
- : undefined;
2247
- },
2248
-
2249
- async buildRequest(parsed: OcxParsedRequest, incoming) {
2250
- const built = await build(parsed);
2251
- modelId = parsed.modelId;
2252
- contextWindow = kiroUpstreamContextWindow(parsed.modelId);
2253
- inputTokens = built.inputTokens;
2254
- contextInputEstimate = built.contextInputEstimate;
2255
- toolNameMap = built.nameMap;
2256
- conversationId = built.conversationId;
2257
- completionMode = built.completionMode;
2258
- requestSnapshot = structuredClone(parsed);
2259
- firstRequestBodyBytes = Buffer.byteLength(built.request.body);
2260
- requestAbortSignal = incoming?.abortSignal;
2261
- return built.request;
2262
- },
2263
-
2264
- parseStream(response: Response, budget: TranslatorBudget): AsyncGenerator<AdapterEvent> {
2265
- return parseKiroStream(
2266
- response,
2267
- budget,
2268
- modelId,
2269
- inputTokens,
2270
- contextWindow,
2271
- toolNameMap,
2272
- conversationId,
2273
- completionMode,
2274
- completionMode === "required" ? fallbackFactory : undefined,
2275
- contextInputEstimate,
2276
- );
2277
- },
2278
-
2279
- fetchResponse(request: AdapterRequest, ctx?: AdapterFetchContext): Promise<Response> {
2280
- // The normal Responses path supplies cancellation at fetch time rather than build time.
2281
- // Keep it for the adapter-owned bounded continuation so cancelling the client turn aborts
2282
- // both the first Kiro request and its one allowed completion retry.
2283
- if (ctx?.abortSignal) requestAbortSignal = ctx.abortSignal;
2284
- return fetchKiroWithRetry(request, ctx);
2285
- },
2286
-
2287
- formatErrorBody(status: number, headers: Headers, payloadText: string): string {
2288
- return safeKiroHttpErrorMessage(status, headers, payloadText);
2289
- },
2290
-
2291
- // Kiro always returns an event stream, including for non-streaming Responses requests. Drain
2292
- // the decoder into a budget-owned batch so an upstream stream cannot grow this array without
2293
- // bound while the caller waits for the complete JSON response.
2294
- async parseResponse(response: Response, budget: TranslatorBudget): Promise<AdapterEvent[]> {
2295
- const events: AdapterEvent[] = [];
2296
- try {
2297
- for await (const e of parseKiroStream(
2298
- response,
2299
- budget,
2300
- modelId,
2301
- inputTokens,
2302
- contextWindow,
2303
- toolNameMap,
2304
- conversationId,
2305
- completionMode,
2306
- completionMode === "required" ? fallbackFactory : undefined,
2307
- contextInputEstimate,
2308
- )) {
2309
- retainTranslatedEvent(e, budget, events.at(-1));
2310
- events.push(e);
2311
- }
2312
- return events;
2313
- } catch (error) {
2314
- for (const event of events) releaseTranslatedEvent(event, budget);
2315
- throw error;
2316
- }
2317
- },
2318
- };
2319
- }
1
+ // Thin facade over the cohesive leaf modules under ./kiro/. Every name this file exported
2
+ // before the split is re-exported here, spelled identically, so existing consumers of
3
+ // "src/adapters/kiro" keep working unchanged.
4
+ export type { KiroReasoningMode } from "./kiro/reasoning";
5
+ export { kiroReasoningMode } from "./kiro/reasoning";
6
+ export { boundedInjectedInstructionForTests, buildKiroPayload } from "./kiro/payload";
7
+ export { isRetryableKiroStreamCatchError, parseKiroStream } from "./kiro/stream";
8
+ export { createKiroAdapter } from "./kiro/adapter";