@namzu/sdk 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/CHANGELOG.md +99 -0
  2. package/dist/compaction/__tests__/verifier-empty-reply.test.d.ts +12 -0
  3. package/dist/compaction/__tests__/verifier-empty-reply.test.d.ts.map +1 -0
  4. package/dist/compaction/__tests__/verifier-empty-reply.test.js +73 -0
  5. package/dist/compaction/__tests__/verifier-empty-reply.test.js.map +1 -0
  6. package/dist/compaction/verifier.d.ts.map +1 -1
  7. package/dist/compaction/verifier.js +10 -1
  8. package/dist/compaction/verifier.js.map +1 -1
  9. package/dist/config/__tests__/compaction-budget-schema.test.d.ts +18 -0
  10. package/dist/config/__tests__/compaction-budget-schema.test.d.ts.map +1 -0
  11. package/dist/config/__tests__/compaction-budget-schema.test.js +96 -0
  12. package/dist/config/__tests__/compaction-budget-schema.test.js.map +1 -0
  13. package/dist/config/runtime.d.ts.map +1 -1
  14. package/dist/config/runtime.js +20 -11
  15. package/dist/config/runtime.js.map +1 -1
  16. package/dist/manager/run/persistence.d.ts +2 -1
  17. package/dist/manager/run/persistence.d.ts.map +1 -1
  18. package/dist/manager/run/persistence.js +3 -1
  19. package/dist/manager/run/persistence.js.map +1 -1
  20. package/dist/provider/__tests__/errors.test.d.ts +2 -0
  21. package/dist/provider/__tests__/errors.test.d.ts.map +1 -0
  22. package/dist/provider/__tests__/errors.test.js +61 -0
  23. package/dist/provider/__tests__/errors.test.js.map +1 -0
  24. package/dist/provider/__tests__/registry.test.d.ts +5 -0
  25. package/dist/provider/__tests__/registry.test.d.ts.map +1 -1
  26. package/dist/provider/__tests__/registry.test.js +186 -1
  27. package/dist/provider/__tests__/registry.test.js.map +1 -1
  28. package/dist/provider/errors.d.ts +108 -0
  29. package/dist/provider/errors.d.ts.map +1 -0
  30. package/dist/provider/errors.js +311 -0
  31. package/dist/provider/errors.js.map +1 -0
  32. package/dist/provider/index.d.ts +3 -1
  33. package/dist/provider/index.d.ts.map +1 -1
  34. package/dist/provider/index.js +2 -1
  35. package/dist/provider/index.js.map +1 -1
  36. package/dist/provider/registry.d.ts +81 -1
  37. package/dist/provider/registry.d.ts.map +1 -1
  38. package/dist/provider/registry.js +173 -7
  39. package/dist/provider/registry.js.map +1 -1
  40. package/dist/public-runtime.d.ts +11 -11
  41. package/dist/public-runtime.d.ts.map +1 -1
  42. package/dist/public-runtime.js +11 -11
  43. package/dist/public-runtime.js.map +1 -1
  44. package/dist/registry/tool/execute.d.ts.map +1 -1
  45. package/dist/registry/tool/execute.js +5 -2
  46. package/dist/registry/tool/execute.js.map +1 -1
  47. package/dist/registry/tool/execute.test.js +11 -0
  48. package/dist/registry/tool/execute.test.js.map +1 -1
  49. package/dist/runtime/query/__tests__/stream-recovery.test.js +54 -1
  50. package/dist/runtime/query/__tests__/stream-recovery.test.js.map +1 -1
  51. package/dist/runtime/query/iteration/phases/compaction-safe-cut.test.d.ts +32 -0
  52. package/dist/runtime/query/iteration/phases/compaction-safe-cut.test.d.ts.map +1 -0
  53. package/dist/runtime/query/iteration/phases/compaction-safe-cut.test.js +152 -0
  54. package/dist/runtime/query/iteration/phases/compaction-safe-cut.test.js.map +1 -0
  55. package/dist/runtime/query/iteration/phases/compaction.d.ts.map +1 -1
  56. package/dist/runtime/query/iteration/phases/compaction.js +52 -12
  57. package/dist/runtime/query/iteration/phases/compaction.js.map +1 -1
  58. package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
  59. package/dist/runtime/query/iteration/stream-turn.js +11 -5
  60. package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
  61. package/dist/runtime/query/result.d.ts.map +1 -1
  62. package/dist/runtime/query/result.js +11 -1
  63. package/dist/runtime/query/result.js.map +1 -1
  64. package/dist/tools/builtins/__tests__/edit.test.js +54 -0
  65. package/dist/tools/builtins/__tests__/edit.test.js.map +1 -1
  66. package/dist/tools/builtins/edit.d.ts +1 -1
  67. package/dist/tools/builtins/edit.d.ts.map +1 -1
  68. package/dist/tools/builtins/edit.js +11 -13
  69. package/dist/tools/builtins/edit.js.map +1 -1
  70. package/dist/tools/coordinator/__tests__/task-list.test.js +14 -0
  71. package/dist/tools/coordinator/__tests__/task-list.test.js.map +1 -1
  72. package/dist/tools/coordinator/agent.d.ts +6 -8
  73. package/dist/tools/coordinator/agent.d.ts.map +1 -1
  74. package/dist/tools/coordinator/agent.js.map +1 -1
  75. package/dist/tools/coordinator/index.d.ts.map +1 -1
  76. package/dist/tools/coordinator/index.js +11 -32
  77. package/dist/tools/coordinator/index.js.map +1 -1
  78. package/dist/tools/defineTool.d.ts +1 -0
  79. package/dist/tools/defineTool.d.ts.map +1 -1
  80. package/dist/tools/defineTool.js +1 -0
  81. package/dist/tools/defineTool.js.map +1 -1
  82. package/dist/types/provider/config.d.ts +33 -0
  83. package/dist/types/provider/config.d.ts.map +1 -1
  84. package/dist/types/provider/error.d.ts +22 -0
  85. package/dist/types/provider/error.d.ts.map +1 -0
  86. package/dist/types/provider/error.js +2 -0
  87. package/dist/types/provider/error.js.map +1 -0
  88. package/dist/types/provider/index.d.ts +2 -1
  89. package/dist/types/provider/index.d.ts.map +1 -1
  90. package/dist/types/run/entity.d.ts +2 -0
  91. package/dist/types/run/entity.d.ts.map +1 -1
  92. package/dist/types/run/events.d.ts +2 -0
  93. package/dist/types/run/events.d.ts.map +1 -1
  94. package/dist/types/run/events.js.map +1 -1
  95. package/dist/types/tool/index.d.ts +6 -0
  96. package/dist/types/tool/index.d.ts.map +1 -1
  97. package/package.json +1 -1
  98. package/src/compaction/__tests__/verifier-empty-reply.test.ts +96 -0
  99. package/src/compaction/verifier.ts +10 -1
  100. package/src/config/__tests__/compaction-budget-schema.test.ts +114 -0
  101. package/src/config/runtime.ts +20 -11
  102. package/src/manager/run/persistence.ts +3 -1
  103. package/src/provider/__tests__/errors.test.ts +85 -0
  104. package/src/provider/__tests__/registry.test.ts +266 -1
  105. package/src/provider/errors.ts +344 -0
  106. package/src/provider/index.ts +22 -1
  107. package/src/provider/registry.ts +202 -7
  108. package/src/public-runtime.ts +55 -10
  109. package/src/registry/tool/execute.test.ts +14 -0
  110. package/src/registry/tool/execute.ts +5 -2
  111. package/src/runtime/query/__tests__/stream-recovery.test.ts +64 -1
  112. package/src/runtime/query/iteration/phases/compaction-safe-cut.test.ts +186 -0
  113. package/src/runtime/query/iteration/phases/compaction.ts +54 -12
  114. package/src/runtime/query/iteration/stream-turn.ts +11 -6
  115. package/src/runtime/query/result.ts +11 -1
  116. package/src/tools/builtins/__tests__/edit.test.ts +76 -0
  117. package/src/tools/builtins/edit.ts +12 -12
  118. package/src/tools/coordinator/__tests__/task-list.test.ts +16 -0
  119. package/src/tools/coordinator/agent.ts +6 -8
  120. package/src/tools/coordinator/index.ts +11 -34
  121. package/src/tools/defineTool.ts +2 -0
  122. package/src/types/provider/config.ts +36 -0
  123. package/src/types/provider/error.ts +29 -0
  124. package/src/types/provider/index.ts +8 -0
  125. package/src/types/run/entity.ts +2 -0
  126. package/src/types/run/events.ts +7 -1
  127. package/src/types/tool/index.ts +6 -0
@@ -47,11 +47,19 @@ export * from './utils/id.js'
47
47
 
48
48
  // ─── utility helpers ─────────────────────────────────────────────────────
49
49
 
50
- export { accumulateCost, calculateCost, formatCost, ZERO_COST } from './utils/cost.js'
50
+ export {
51
+ accumulateCost,
52
+ calculateCost,
53
+ formatCost,
54
+ ZERO_COST,
55
+ } from './utils/cost.js'
51
56
  export { toErrorMessage } from './utils/error.js'
52
57
  export { configureLogger, getRootLogger, Logger } from './utils/logger.js'
53
58
  export { buildToolResultHashes, hashToolResult } from './utils/hash.js'
54
- export { compressShellOutput, compressShellOutputFull } from './utils/shell-compress.js'
59
+ export {
60
+ compressShellOutput,
61
+ compressShellOutputFull,
62
+ } from './utils/shell-compress.js'
55
63
  export { createChildAbortController } from './utils/abort.js'
56
64
  export { memoizeAsync } from './utils/memoize.js'
57
65
  export { extractFinalResponse } from './utils/conversation.js'
@@ -61,7 +69,10 @@ export { extractFinalResponse } from './utils/conversation.js'
61
69
  export { resolveTaskModel } from './router/task-router.js'
62
70
  export { drainQuery, query } from './runtime/query/index.js'
63
71
  export { ContextCache } from './runtime/query/context-cache.js'
64
- export { CheckpointManager, projectEmergencyToCheckpoint } from './runtime/query/checkpoint.js'
72
+ export {
73
+ CheckpointManager,
74
+ projectEmergencyToCheckpoint,
75
+ } from './runtime/query/checkpoint.js'
65
76
  export { prepareReplayState } from './runtime/query/replay/prepare.js'
66
77
  export { listCheckpoints } from './runtime/query/replay/list.js'
67
78
  export { DecisionParser, FallbackResolver } from './runtime/decision/index.js'
@@ -73,8 +84,17 @@ export {
73
84
 
74
85
  // ─── personas, skills, advisory ──────────────────────────────────────────
75
86
 
76
- export { assembleSystemPrompt, mergePersonas, withSessionContext } from './persona/index.js'
77
- export { discoverSkills, loadSkill, resolveSkillChain, SkillRegistry } from './skills/index.js'
87
+ export {
88
+ assembleSystemPrompt,
89
+ mergePersonas,
90
+ withSessionContext,
91
+ } from './persona/index.js'
92
+ export {
93
+ discoverSkills,
94
+ loadSkill,
95
+ resolveSkillChain,
96
+ SkillRegistry,
97
+ } from './skills/index.js'
78
98
  export {
79
99
  AdvisorRegistry,
80
100
  AdvisoryContext,
@@ -144,17 +164,30 @@ export { LocalTaskGateway } from './gateway/local.js'
144
164
  // ─── providers, sandbox, vault ───────────────────────────────────────────
145
165
 
146
166
  export {
167
+ bodySaysContextOverflow,
168
+ classifyProviderHttpStatus,
147
169
  DuplicateProviderError,
170
+ isCallerAbortError,
171
+ isProviderRequestError,
172
+ LazyProviderLoadError,
173
+ LazyProviderSyncCreateError,
148
174
  MOCK_CAPABILITIES,
149
175
  MockLLMProvider,
176
+ parseRetryAfterMs,
150
177
  PERMISSIVE_PROVIDER_CAPABILITIES,
178
+ providerHttpError,
179
+ providerVendorError,
151
180
  ProviderRegistry,
181
+ ProviderRequestError,
152
182
  registerMock,
153
183
  resolveProviderCapabilities,
154
184
  UnknownProviderError,
155
185
  } from './provider/index.js'
156
186
 
157
- export { LocalSandboxProvider, SandboxProviderFactory } from './sandbox/index.js'
187
+ export {
188
+ LocalSandboxProvider,
189
+ SandboxProviderFactory,
190
+ } from './sandbox/index.js'
158
191
 
159
192
  export { InMemoryCredentialVault } from './vault/index.js'
160
193
 
@@ -220,7 +253,10 @@ export {
220
253
  runToA2ATask,
221
254
  } from './bridge/a2a/index.js'
222
255
 
223
- export { mapRunToStreamEvent, mapSessionToStreamEvent } from './bridge/sse/index.js'
256
+ export {
257
+ mapRunToStreamEvent,
258
+ mapSessionToStreamEvent,
259
+ } from './bridge/sse/index.js'
224
260
 
225
261
  // ─── bus, verification ───────────────────────────────────────────────────
226
262
 
@@ -353,10 +389,19 @@ export {
353
389
  // ─── runtime helpers colocated with shapes under `types/` (§1.5) ─────────
354
390
 
355
391
  export { A2AProtocolError } from './types/a2a/index.js'
356
- export { isTerminalActivityStatus, resolveActivityTracking } from './types/activity/index.js'
392
+ export {
393
+ isTerminalActivityStatus,
394
+ resolveActivityTracking,
395
+ } from './types/activity/index.js'
357
396
  export { isTerminalAgentTaskState } from './types/agent/task.js'
358
- export { accumulateTokenUsage, isTerminalStatus } from './types/common/index.js'
359
- export { assertComputerUseActionType, assertDisplayServer } from './types/computer-use/index.js'
397
+ export {
398
+ accumulateTokenUsage,
399
+ isTerminalStatus,
400
+ } from './types/common/index.js'
401
+ export {
402
+ assertComputerUseActionType,
403
+ assertDisplayServer,
404
+ } from './types/computer-use/index.js'
360
405
  export { isConnectorActive } from './types/connector/core.js'
361
406
  export { CONNECTOR_SCOPE_ORDER } from './types/connector/scope.js'
362
407
  export { RoutingResponseSchema } from './types/decision/index.js'
@@ -433,6 +433,20 @@ describe('ToolRegistry — execute', () => {
433
433
  expect(result.error).toContain('Expected string, received number')
434
434
  })
435
435
 
436
+ it('appends a tool-specific recovery hint to validation failures', async () => {
437
+ const r = new ToolRegistry()
438
+ r.register(
439
+ makeTool('strict', {
440
+ inputSchema: z.object({ required: z.string() }),
441
+ validationErrorHint: 'Retry with {"required":"value"}.',
442
+ }),
443
+ )
444
+ const result = await r.execute('strict', { required: 123 }, makeContext())
445
+ expect(result.success).toBe(false)
446
+ expect(result.error).toContain('Required: required: string.')
447
+ expect(result.error).toContain('Retry with {"required":"value"}.')
448
+ })
449
+
436
450
  it('empty-args validation lists required params with descriptions', async () => {
437
451
  const r = new ToolRegistry()
438
452
  r.register(
@@ -370,10 +370,13 @@ Executable tool names, descriptions, and JSON input schemas are attached through
370
370
  Object.keys(rawInput as Record<string, unknown>).length === 0)
371
371
 
372
372
  const requiredHint = describeRequiredInput(tool.inputSchema)
373
+ const recoveryHint = tool.validationErrorHint?.trim()
374
+ ? ` ${tool.validationErrorHint.trim()}`
375
+ : ''
373
376
 
374
377
  const enrichedMessage = isEmptyInput
375
- ? `Tool "${toolName}" was called with no arguments. ${requiredHint} Retry the call with the required parameters populated.`
376
- : `Validation failed for "${toolName}": ${errorMessage}. ${requiredHint}`
378
+ ? `Tool "${toolName}" was called with no arguments. ${requiredHint}${recoveryHint} Retry the call with the required parameters populated.`
379
+ : `Validation failed for "${toolName}": ${errorMessage}. ${requiredHint}${recoveryHint}`
377
380
 
378
381
  this.log.error(`Tool input validation failed: ${toolName}`, {
379
382
  errors: errorMessage,
@@ -4,6 +4,7 @@ import { join } from 'node:path'
4
4
  import { afterEach, describe, expect, it, vi } from 'vitest'
5
5
  import { z } from 'zod'
6
6
 
7
+ import { ProviderRequestError } from '../../../provider/errors.js'
7
8
  import { ToolRegistry } from '../../../registry/tool/execute.js'
8
9
  import type { SessionId, TenantId } from '../../../types/ids/index.js'
9
10
  import { createUserMessage } from '../../../types/message/index.js'
@@ -72,6 +73,22 @@ class IdleDuringToolInputProvider implements LLMProvider {
72
73
  }
73
74
  }
74
75
 
76
+ class ClassifiedFailureProvider implements LLMProvider {
77
+ readonly id = 'classified-failure'
78
+ readonly name = 'Classified Failure Provider'
79
+
80
+ async *chatStream(): AsyncIterable<StreamChunk> {
81
+ yield await Promise.reject(
82
+ new ProviderRequestError({
83
+ kind: 'throttle',
84
+ providerId: 'classified-failure',
85
+ status: 429,
86
+ retryAfterMs: 2000,
87
+ }),
88
+ )
89
+ }
90
+ }
91
+
75
92
  describe('query stream recovery', () => {
76
93
  let workdirs: string[] = []
77
94
 
@@ -82,7 +99,10 @@ describe('query stream recovery', () => {
82
99
 
83
100
  it('turns an idle stream with partial tool JSON into retryable tool feedback', async () => {
84
101
  const provider = new IdleDuringToolInputProvider()
85
- const actualWrite = vi.fn(async () => ({ success: true, output: 'should not run' }))
102
+ const actualWrite = vi.fn(async () => ({
103
+ success: true,
104
+ output: 'should not run',
105
+ }))
86
106
  const tools = new ToolRegistry()
87
107
  tools.register({
88
108
  name: 'write_file',
@@ -153,4 +173,47 @@ describe('query stream recovery', () => {
153
173
  'extend it with edit using insertLine',
154
174
  )
155
175
  })
176
+
177
+ it('preserves classified provider metadata through the primary run boundary', async () => {
178
+ const workingDirectory = await mkdtemp(join(tmpdir(), 'namzu-provider-error-'))
179
+ workdirs.push(workingDirectory)
180
+ const events: RunEvent[] = []
181
+
182
+ const run = await drainQuery(
183
+ {
184
+ provider: new ClassifiedFailureProvider(),
185
+ tools: new ToolRegistry(),
186
+ runConfig: {
187
+ model: 'mock-model',
188
+ timeoutMs: 5_000,
189
+ tokenBudget: 100_000,
190
+ maxIterations: 1,
191
+ maxResponseTokens: 256,
192
+ },
193
+ agentId: 'agent_test',
194
+ agentName: 'Test Agent',
195
+ messages: [createUserMessage('fail with classified metadata')],
196
+ workingDirectory,
197
+ sessionId: 'ses_provider_error' as SessionId,
198
+ threadId: 'thd_provider_error' as ThreadId,
199
+ projectId: 'prj_provider_error' as ProjectId,
200
+ tenantId: 'tnt_provider_error' as TenantId,
201
+ },
202
+ (event) => {
203
+ events.push(event)
204
+ },
205
+ )
206
+
207
+ expect(run.status).toBe('failed')
208
+ expect(run.lastProviderError).toEqual({
209
+ kind: 'throttle',
210
+ providerId: 'classified-failure',
211
+ status: 429,
212
+ retryAfterMs: 2000,
213
+ })
214
+ expect(events.find((event) => event.type === 'run_failed')).toMatchObject({
215
+ type: 'run_failed',
216
+ providerError: run.lastProviderError,
217
+ })
218
+ })
156
219
  })
@@ -0,0 +1,186 @@
1
+ /**
2
+ * The UNBOUNDED CUT.
3
+ *
4
+ * `runCompactionCheck` snaps the naive recent-window boundary
5
+ * (`messages.length - keepRecentMessages`) FORWARD via `findSafeTrimIndex` so a
6
+ * tool pair is never split. But `findSafeTrimIndex` only ever walks forward, and
7
+ * its leading-`tool`-message skip has no stop short of `messages.length` — so
8
+ * whenever the whole suffix from the naive index to the end is `tool` messages,
9
+ * the cut lands ON `messages.length`, `recentMessages` comes back EMPTY, and
10
+ * `[...preservedSystem, compactionMessage]` replaces the ENTIRE recent window.
11
+ * What is left is a transcript with no non-system message at all.
12
+ *
13
+ * The shape that produces it: ONE assistant turn that fans out
14
+ * `>= keepRecentMessages` parallel tool calls, measured at the START of the next
15
+ * iteration — which is exactly where `runCompactionCheck` runs (iteration/index.ts,
16
+ * after `refreshWorkingMemory`, before the model call), i.e. immediately after
17
+ * those results were appended.
18
+ *
19
+ * The `olderMessages.length < 1` guard at :117 cannot fire: `olderMessages` is
20
+ * the whole transcript in exactly this situation.
21
+ *
22
+ * Symmetrically, when the naive index lands BELOW `systemMessages.length` the
23
+ * cut sits inside the system prefix and the leading prompts are duplicated into
24
+ * `recentMessages`.
25
+ *
26
+ * The invariant these tests pin is the one a cut taken AT OR BELOW naive gives
27
+ * for free: a pass never removes more than the naive cut would, so at least
28
+ * `keepRecentMessages` original messages survive verbatim — or, when no safe cut
29
+ * exists at all, the pass is skipped and the transcript is untouched.
30
+ */
31
+
32
+ import { describe, expect, it, vi } from 'vitest'
33
+
34
+ import { WorkingStateManager } from '../../../../compaction/manager.js'
35
+ import { CompactionConfigSchema } from '../../../../config/runtime.js'
36
+ import type { RunId } from '../../../../types/ids/index.js'
37
+ import {
38
+ type Message,
39
+ createAssistantMessage,
40
+ createSystemMessage,
41
+ createToolMessage,
42
+ createUserMessage,
43
+ } from '../../../../types/message/index.js'
44
+ import type { Logger } from '../../../../utils/logger.js'
45
+ import { runCompactionCheck } from './compaction.js'
46
+ import type { IterationContext } from './context.js'
47
+
48
+ function makeLogger(): Logger {
49
+ const self = {
50
+ info: vi.fn(),
51
+ warn: vi.fn(),
52
+ error: vi.fn(),
53
+ debug: vi.fn(),
54
+ child: vi.fn(),
55
+ } as unknown as Logger
56
+ ;(self as { child: (ctx: unknown) => Logger }).child = vi.fn(() => self)
57
+ return self
58
+ }
59
+
60
+ /** Long enough that a handful of messages overflow a tiny budget. */
61
+ const FILLER = 'x'.repeat(200)
62
+
63
+ const KEEP_RECENT = 4
64
+
65
+ function makeCtx(opts: {
66
+ messages: Message[]
67
+ contextWindowTokens?: number
68
+ tokenBudget?: number
69
+ }): IterationContext {
70
+ const config = CompactionConfigSchema.parse({
71
+ strategy: 'structured',
72
+ llmVerification: false,
73
+ keepRecentMessages: KEEP_RECENT,
74
+ ...(opts.contextWindowTokens !== undefined
75
+ ? { contextWindowTokens: opts.contextWindowTokens }
76
+ : {}),
77
+ })
78
+ const manager = new WorkingStateManager(config)
79
+ manager.addDecision('built the report as .docx')
80
+
81
+ return {
82
+ runConfig: { tokenBudget: opts.tokenBudget ?? 0 },
83
+ compactionConfig: config,
84
+ workingStateManager: manager,
85
+ log: makeLogger(),
86
+ runMgr: {
87
+ id: 'run_1' as RunId,
88
+ currentIteration: 3,
89
+ messages: opts.messages,
90
+ },
91
+ } as unknown as IterationContext
92
+ }
93
+
94
+ function toolCall(id: string) {
95
+ return { id, type: 'function' as const, function: { name: 'read', arguments: '{}' } }
96
+ }
97
+
98
+ /** How many of the ORIGINAL messages survived the pass, by identity. */
99
+ function survivorCount(before: readonly Message[], after: readonly Message[]): number {
100
+ const kept = new Set<Message>(after)
101
+ return before.filter((m) => kept.has(m)).length
102
+ }
103
+
104
+ /**
105
+ * The exact shape the iteration loop holds when `runCompactionCheck` runs: the
106
+ * user's turn, one assistant that fanned out four parallel tool calls, and the
107
+ * four results that just landed. The next thing that happens is the model call —
108
+ * with whatever this function leaves behind.
109
+ */
110
+ function buildParallelFanOutTail(): Message[] {
111
+ return [
112
+ createSystemMessage(`STATIC SYSTEM PROMPT ${FILLER}`, 'cache'),
113
+ createUserMessage(`please rename the heading to Q3 ${FILLER}`),
114
+ createAssistantMessage(`reading the sources ${FILLER}`, [
115
+ toolCall('a'),
116
+ toolCall('b'),
117
+ toolCall('c'),
118
+ toolCall('d'),
119
+ ]),
120
+ createToolMessage(`result a ${FILLER}`, 'a'),
121
+ createToolMessage(`result b ${FILLER}`, 'b'),
122
+ createToolMessage(`result c ${FILLER}`, 'c'),
123
+ createToolMessage(`result d ${FILLER}`, 'd'),
124
+ ]
125
+ }
126
+
127
+ describe('compaction — the unbounded cut', () => {
128
+ it('keeps at least keepRecentMessages messages verbatim when the tail is a parallel tool fan-out', async () => {
129
+ const messages = buildParallelFanOutTail()
130
+ const before = [...messages]
131
+ const ctx = makeCtx({ messages, contextWindowTokens: 100 })
132
+
133
+ await runCompactionCheck(ctx)
134
+
135
+ expect(survivorCount(before, messages)).toBeGreaterThanOrEqual(KEEP_RECENT)
136
+ })
137
+
138
+ it('never leaves a system-only transcript (nothing for the next turn to answer)', async () => {
139
+ const messages = buildParallelFanOutTail()
140
+ const ctx = makeCtx({ messages, contextWindowTokens: 100 })
141
+
142
+ await runCompactionCheck(ctx)
143
+
144
+ const nonSystem = messages.filter((m) => m.role !== 'system')
145
+ expect(nonSystem.length).toBeGreaterThan(0)
146
+ })
147
+
148
+ it('skips the pass, leaving the transcript intact, when no safe cut exists at or below naive', async () => {
149
+ // System prefix, then a single assistant fanning out six parallel calls.
150
+ // Every candidate boundary at or below naive splits that one pair-set, so
151
+ // there is nothing safe to cut to — skipping beats deleting the turn.
152
+ const ids = ['a', 'b', 'c', 'd', 'e', 'f']
153
+ const messages: Message[] = [
154
+ createSystemMessage(`STATIC SYSTEM PROMPT ${FILLER}`, 'cache'),
155
+ createAssistantMessage(`fanning out ${FILLER}`, ids.map(toolCall)),
156
+ ...ids.map((id) => createToolMessage(`result ${id} ${FILLER}`, id)),
157
+ ]
158
+ const before = [...messages]
159
+ const ctx = makeCtx({ messages, contextWindowTokens: 100 })
160
+
161
+ await runCompactionCheck(ctx)
162
+
163
+ expect(messages).toEqual(before)
164
+ })
165
+
166
+ it('does not duplicate the leading system prompts when the naive cut lands inside them', async () => {
167
+ // Legacy (tokenBudget) path: five leading system messages and only three
168
+ // conversational ones, so naive = 8 - 4 = 4 < systemMessages.length = 5.
169
+ const messages: Message[] = [
170
+ createSystemMessage(`SYS-1 ${FILLER}`, 'cache'),
171
+ createSystemMessage(`SYS-2 ${FILLER}`),
172
+ createSystemMessage(`SYS-3 ${FILLER}`),
173
+ createSystemMessage(`SYS-4 ${FILLER}`),
174
+ createSystemMessage(`SYS-5 UNIQUE-MARKER ${FILLER}`),
175
+ createUserMessage(`user 0 ${FILLER}`),
176
+ createAssistantMessage(`assistant 0 ${FILLER}`),
177
+ createUserMessage(`user 1 ${FILLER}`),
178
+ ]
179
+ const ctx = makeCtx({ messages, tokenBudget: 100 })
180
+
181
+ await runCompactionCheck(ctx)
182
+
183
+ const marker = messages.filter((m) => m.content?.includes('UNIQUE-MARKER'))
184
+ expect(marker).toHaveLength(1)
185
+ })
186
+ })
@@ -94,18 +94,60 @@ export async function runCompactionCheck(ctx: IterationContext): Promise<void> {
94
94
  }
95
95
  if (systemMessages.length === 0) return
96
96
 
97
- // Tool-pair atomicity guard. A naive cut at `length - keepRecentMessages`
98
- // can land BETWEEN an assistant-with-toolCalls (which would be dropped into
99
- // `olderMessages`) and its `tool` results (kept in `recentMessages`),
100
- // leaving orphaned `tool_result` blocks at the head of the recent window.
101
- // The Anthropic provider then emits a `tool_result` with no matching
102
- // `tool_use` and the API rejects the next turn with a 400 so compaction,
103
- // whose whole job is to keep a long run alive, instead kills it. Snap the
104
- // boundary FORWARD to a safe point (existing `findSafeTrimIndex`, previously
105
- // only wired to the unused ConversationManager strategy classes) so no pair
106
- // is split. Any message this moves out of the recent window is already
107
- // represented in the extracted WorkingState the summary is built from.
108
- const keepStart = findSafeTrimIndex(messages, messages.length - config.keepRecentMessages)
97
+ // Tool-pair atomicity guard, taken DOWNWARD.
98
+ //
99
+ // A naive cut at `length - keepRecentMessages` can land BETWEEN an
100
+ // assistant-with-toolCalls (dropped into `olderMessages`) and its `tool`
101
+ // results (kept in `recentMessages`), leaving orphaned `tool_result` blocks at
102
+ // the head of the recent window. The Anthropic provider then emits a
103
+ // `tool_result` with no matching `tool_use` and the API rejects the next turn
104
+ // with a 400 — so compaction, whose whole job is to keep a long run alive,
105
+ // instead kills it.
106
+ //
107
+ // Snapping FORWARD via `findSafeTrimIndex` fixes the orphan and introduces a
108
+ // worse failure. That walk skips leading `tool` messages with no stop short of
109
+ // `messages.length`, so whenever the entire suffix from the naive index is
110
+ // `tool` messages, the boundary lands ON `messages.length`: `recentMessages`
111
+ // comes back EMPTY and `[...preservedSystem, compactionMessage]` replaces the
112
+ // whole recent window. The model is then asked to answer a conversation whose
113
+ // last turn — including the user's own message — was deleted, and it answers a
114
+ // question nobody asked. The shape is routine, not exotic: one assistant turn
115
+ // fanning out `>= keepRecentMessages` parallel tool calls, measured at the
116
+ // start of the very next iteration, which is exactly where this runs. The
117
+ // `olderMessages` floor guard below cannot catch it either — in that shape
118
+ // `olderMessages` is the whole transcript.
119
+ //
120
+ // So take the LARGEST safe boundary AT OR BELOW naive instead. A cut that low
121
+ // removes no more than the naive cut would, so at least `keepRecentMessages`
122
+ // original messages always survive verbatim, and the transcript can never be
123
+ // reduced to system messages alone. `findSafeTrimIndex(m, k) === k` is the
124
+ // safety predicate — reused rather than reimplemented, so the function itself
125
+ // (public API, other callers) is untouched.
126
+ const naiveKeepStart = messages.length - config.keepRecentMessages
127
+ let keepStart = -1
128
+ for (let candidate = naiveKeepStart; candidate > systemMessages.length; candidate--) {
129
+ if (findSafeTrimIndex(messages, candidate) === candidate) {
130
+ keepStart = candidate
131
+ break
132
+ }
133
+ }
134
+
135
+ // No safe boundary at or below naive. Either every candidate splits a pair-set
136
+ // (one assistant fanning out more calls than the recent window holds), or naive
137
+ // itself sits inside the leading system prefix — which would otherwise
138
+ // duplicate those prompts into `recentMessages`. Skipping costs one iteration's
139
+ // worth of context headroom; cutting anyway costs the live turn. The condition
140
+ // is self-clearing: the next assistant message moves naive past the tool block.
141
+ if (keepStart < 0) {
142
+ ctx.log.debug('Skipping compaction — no safe cut at or below the naive boundary', {
143
+ runId: ctx.runMgr.id,
144
+ naiveKeepStart,
145
+ systemMessages: systemMessages.length,
146
+ messageCount: messages.length,
147
+ })
148
+ return
149
+ }
150
+
109
151
  const recentMessages = messages.slice(keepStart)
110
152
  const olderMessages = messages.slice(systemMessages.length, keepStart)
111
153
 
@@ -1,3 +1,4 @@
1
+ import { isProviderRequestError } from '../../../provider/errors.js'
1
2
  import { mergeTokenUsage } from '../../../types/common/index.js'
2
3
  import type { ToolUseId } from '../../../types/ids/index.js'
3
4
  import type {
@@ -111,9 +112,12 @@ export async function* streamProviderTurn(
111
112
  inputTruncated: boolean
112
113
  }
113
114
  >()
114
- let streamError: string | undefined
115
+ let streamError: Error | undefined
115
116
 
116
- const stream = provider.chatStream({ ...params, stream: true }) as AsyncIterable<StreamChunk>
117
+ const stream = provider.chatStream({
118
+ ...params,
119
+ stream: true,
120
+ }) as AsyncIterable<StreamChunk>
117
121
 
118
122
  // Drive the stream manually so each `.next()` can be RACED against the run
119
123
  // abort: a Stop tears the in-flight model request down (the provider got
@@ -145,7 +149,7 @@ export async function* streamProviderTurn(
145
149
  if (res.done) break
146
150
  const chunk = res.value
147
151
  if (chunk.error) {
148
- streamError = chunk.error
152
+ streamError = new Error(`Provider stream error: ${chunk.error}`)
149
153
  break
150
154
  }
151
155
  if (!id && chunk.id) id = chunk.id
@@ -252,7 +256,7 @@ export async function* streamProviderTurn(
252
256
  // run as cancelled rather than recording a normal (errored) turn. Any
253
257
  // other stream error is captured into the synthesized response as before.
254
258
  if (signal?.aborted) throw err
255
- streamError = err instanceof Error ? err.message : String(err)
259
+ streamError = err instanceof Error ? err : new Error(String(err))
256
260
  } finally {
257
261
  if (onAbort) signal?.removeEventListener('abort', onAbort)
258
262
  // Release the underlying connection on every exit (natural end, error,
@@ -355,7 +359,7 @@ export async function* streamProviderTurn(
355
359
  log.warn('provider stream failed after tool input; surfacing tool call to executor', {
356
360
  runId,
357
361
  iteration,
358
- error: streamError,
362
+ error: streamError?.message ?? 'provider stream failed',
359
363
  toolCallCount: toolCalls.length,
360
364
  })
361
365
  }
@@ -378,7 +382,8 @@ export async function* streamProviderTurn(
378
382
  yield* drainPending()
379
383
 
380
384
  if (streamError && !recoveredToolInputFromStreamError) {
381
- throw new Error(`Provider stream error: ${streamError}`)
385
+ if (isProviderRequestError(streamError)) throw streamError
386
+ throw new Error(`Provider stream error: ${streamError.message}`)
382
387
  }
383
388
 
384
389
  const response: ChatCompletionResponse = {
@@ -1,6 +1,7 @@
1
1
  import { type Span, SpanStatusCode } from '@opentelemetry/api'
2
2
  import type { PlanManager } from '../../manager/plan/lifecycle.js'
3
3
  import type { RunPersistence } from '../../manager/run/persistence.js'
4
+ import { isProviderRequestError } from '../../provider/errors.js'
4
5
  import type { ActivityStore } from '../../store/activity/memory.js'
5
6
  import { GENAI, NAMZU } from '../../telemetry/attributes.js'
6
7
  import type { Run, RunEvent } from '../../types/run/index.js'
@@ -61,7 +62,15 @@ export class ResultAssembler {
61
62
  async *handleError(err: unknown, rootSpan: Span): AsyncGenerator<RunEvent> {
62
63
  const { runMgr, planManager, log, emitEvent, drainPending } = this.config
63
64
  const errorMessage = toErrorMessage(err)
64
- runMgr.markFailed(errorMessage)
65
+ const providerError = isProviderRequestError(err)
66
+ ? {
67
+ kind: err.kind,
68
+ providerId: err.providerId,
69
+ ...(err.status !== undefined ? { status: err.status } : {}),
70
+ ...(err.retryAfterMs !== undefined ? { retryAfterMs: err.retryAfterMs } : {}),
71
+ }
72
+ : undefined
73
+ runMgr.markFailed(errorMessage, providerError)
65
74
 
66
75
  if (planManager.isActive) {
67
76
  planManager.failPlan(errorMessage)
@@ -71,6 +80,7 @@ export class ResultAssembler {
71
80
  type: 'run_failed',
72
81
  runId: runMgr.id,
73
82
  error: errorMessage,
83
+ ...(providerError ? { providerError } : {}),
74
84
  })
75
85
  yield* drainPending()
76
86