@modelprofile.com/flexharness-agent 8.2.0 → 8.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -42,6 +42,8 @@ import {
42
42
  type TAgentSessionChangeListener,
43
43
  type TAgentCacheSetting,
44
44
  type TAgentGenerationPrepare,
45
+ type TAgentModelCallUsageEvent,
46
+ type TAgentModelCallUsageReporter,
45
47
  type TAgentPrompt,
46
48
  type TAgentToolCallFinishEvent,
47
49
  type TAgentToolExecutionReconciliationOptions,
@@ -50,6 +52,8 @@ import {
50
52
  isAgentEventStoreV2,
51
53
  validateAgentEventSnapshotV2,
52
54
  } from './smartagent.persistence.js';
55
+ import { MAX_RETRY_ATTEMPTS, planModelCallRetry } from './smartagent.retry.js';
56
+ import { AgentModelCallUsageRecorder } from './smartagent.usage.js';
53
57
  import type {
54
58
  IAgentEventArchive,
55
59
  IAgentEventArchiveV2,
@@ -64,11 +68,6 @@ import {
64
68
  type IAgentGenerationTransaction,
65
69
  } from './smartagent.transactions.js';
66
70
 
67
- const RETRY_INITIAL_DELAY = 2000;
68
- const RETRY_BACKOFF_FACTOR = 2;
69
- const RETRY_MAX_DELAY = 30_000;
70
- const MAX_RETRY_ATTEMPTS = 8;
71
-
72
71
  const isTransactionalControlEvent = (event: TAgentEvent): boolean =>
73
72
  event.type === 'generation-begun'
74
73
  || event.type === 'generation-execution-started'
@@ -139,27 +138,6 @@ interface IGenerationLeaseCleanupOwnership {
139
138
 
140
139
  const sessionHydrations = new WeakMap<IAgentSessionOptions, TAgentEventSnapshot | undefined>();
141
140
 
142
- const retryDelay = (attempt: number, headers?: Record<string, string>): number => {
143
- if (headers) {
144
- const milliseconds = headers['retry-after-ms'];
145
- if (milliseconds) {
146
- const parsed = Number.parseFloat(milliseconds);
147
- if (!Number.isNaN(parsed)) return parsed;
148
- }
149
- const retryAfter = headers['retry-after'];
150
- if (retryAfter) {
151
- const seconds = Number.parseFloat(retryAfter);
152
- if (!Number.isNaN(seconds)) return Math.ceil(seconds * 1000);
153
- const dateDelay = Date.parse(retryAfter) - Date.now();
154
- if (!Number.isNaN(dateDelay) && dateDelay > 0) return Math.ceil(dateDelay);
155
- }
156
- }
157
- return Math.min(
158
- RETRY_INITIAL_DELAY * Math.pow(RETRY_BACKOFF_FACTOR, attempt - 1),
159
- RETRY_MAX_DELAY,
160
- );
161
- };
162
-
163
141
  const sleep = async (milliseconds: number, signal?: AbortSignal): Promise<void> =>
164
142
  new Promise((resolve, reject) => {
165
143
  if (signal?.aborted) {
@@ -246,17 +224,6 @@ const waitWithAbort = <TValue>(
246
224
  });
247
225
  };
248
226
 
249
- const isRetryableError = (error: unknown): boolean => {
250
- const candidate = error as { status?: number; statusCode?: number };
251
- const status = candidate?.status ?? candidate?.statusCode;
252
- if (status === 429 || status === 529 || status === 503) return true;
253
- if (!(error instanceof Error)) return false;
254
- const message = error.message.toLowerCase();
255
- return message.includes('rate limit')
256
- || message.includes('overloaded')
257
- || message.includes('too many requests');
258
- };
259
-
260
227
  const isContextOverflow = (error: unknown): boolean => {
261
228
  if (!(error instanceof Error)) return false;
262
229
  const message = error.message.toLowerCase();
@@ -556,6 +523,19 @@ export class AgentSessionEngine implements IAgentSession {
556
523
  this.platform.reportListenerError(error);
557
524
  }
558
525
 
526
+ private reportModelCallUsage(event: TAgentModelCallUsageEvent): void {
527
+ try {
528
+ this.options.onUsage?.(event);
529
+ } catch (error) {
530
+ this.reportListenerError(error);
531
+ }
532
+ }
533
+
534
+ /** Reports the calls of a compaction that no generation caused: compact() and event retention. */
535
+ private readonly reportSessionCompactionUsage: TAgentModelCallUsageReporter = (call) => {
536
+ this.reportModelCallUsage({ ...call, source: 'compaction' });
537
+ };
538
+
559
539
  private queueNotification(change: IAgentSessionChange, durability: Promise<void>): void {
560
540
  const dispatch = (async () => {
561
541
  try {
@@ -1201,6 +1181,7 @@ export class AgentSessionEngine implements IAgentSession {
1201
1181
 
1202
1182
  private async performCompaction(
1203
1183
  reason: 'context-overflow' | 'retention' | 'manual',
1184
+ reportUsage: TAgentModelCallUsageReporter,
1204
1185
  abortSignal?: AbortSignal,
1205
1186
  ): Promise<IContextCompactionEvent | undefined> {
1206
1187
  abortSignal?.throwIfAborted();
@@ -1235,11 +1216,15 @@ export class AgentSessionEngine implements IAgentSession {
1235
1216
  const messages = this.buildContext(visibleCoveredEvents);
1236
1217
  const replacementMessages = this.options.contextCompactor
1237
1218
  ? await this.invokeGenerationCallback(
1238
- () => this.options.contextCompactor!(messages, visibleCoveredEvents, { abortSignal, reason }),
1219
+ () => this.options.contextCompactor!(
1220
+ messages,
1221
+ visibleCoveredEvents,
1222
+ { abortSignal, reason, reportUsage },
1223
+ ),
1239
1224
  )
1240
1225
  : reason === 'context-overflow' && this.options.onContextOverflow
1241
1226
  ? await this.invokeGenerationCallback(
1242
- () => this.options.onContextOverflow!(messages, { abortSignal }),
1227
+ () => this.options.onContextOverflow!(messages, { abortSignal, reportUsage }),
1243
1228
  )
1244
1229
  : undefined;
1245
1230
  if (!replacementMessages) {
@@ -1297,7 +1282,11 @@ export class AgentSessionEngine implements IAgentSession {
1297
1282
  const combinedAbort = combineAbortSignals([options.abort, this.sessionAbortController.signal]);
1298
1283
  const operation = this.queueExclusive(async () => {
1299
1284
  combinedAbort.signal.throwIfAborted();
1300
- await this.performCompaction(options.reason ?? 'manual', combinedAbort.signal);
1285
+ await this.performCompaction(
1286
+ options.reason ?? 'manual',
1287
+ this.reportSessionCompactionUsage,
1288
+ combinedAbort.signal,
1289
+ );
1301
1290
  }).finally(combinedAbort.cleanup);
1302
1291
  return waitWithAbort(operation, [options.abort, this.sessionAbortController.signal]);
1303
1292
  }
@@ -1365,6 +1354,7 @@ export class AgentSessionEngine implements IAgentSession {
1365
1354
  while (this.events.length > retention.maxEvents) {
1366
1355
  const compaction = await this.performCompaction(
1367
1356
  'retention',
1357
+ this.reportSessionCompactionUsage,
1368
1358
  this.sessionAbortController.signal,
1369
1359
  );
1370
1360
  if (!compaction) return;
@@ -1742,6 +1732,8 @@ export class AgentSessionEngine implements IAgentSession {
1742
1732
  return this.generateLocked(options);
1743
1733
  });
1744
1734
  this.generationQueue = run.then(() => undefined, () => undefined);
1735
+ // A transactional generation is not raced against its abort signal: it settles when its
1736
+ // execution has ended, after each of its model calls was reported through `onUsage`.
1745
1737
  return options.transaction
1746
1738
  ? run
1747
1739
  : waitWithAbort(run, [options.abort, this.sessionAbortController.signal]);
@@ -1848,11 +1840,27 @@ export class AgentSessionEngine implements IAgentSession {
1848
1840
 
1849
1841
  let stepCount = 0;
1850
1842
  let attempt = 0;
1843
+ let retriedMs = 0;
1851
1844
  let contextOverflowRetries = 0;
1852
1845
  let totalInput = 0;
1853
1846
  let totalOutput = 0;
1854
1847
  let totalCacheRead = 0;
1855
1848
  let totalCacheWrite = 0;
1849
+ const recordModelCallUsage = (event: TAgentModelCallUsageEvent): void => {
1850
+ if (event.status === 'reported') {
1851
+ totalInput += event.usage.inputTokens;
1852
+ totalOutput += event.usage.outputTokens;
1853
+ totalCacheRead += event.usage.cacheReadTokens;
1854
+ totalCacheWrite += event.usage.cacheWriteTokens;
1855
+ }
1856
+ this.reportModelCallUsage(event);
1857
+ };
1858
+ const modelCalls = new AgentModelCallUsageRecorder(
1859
+ runtime.model,
1860
+ (call) => recordModelCallUsage({ ...call, source: 'generation', generationId }),
1861
+ );
1862
+ const reportCompactionUsage: TAgentModelCallUsageReporter = (call) =>
1863
+ recordModelCallUsage({ ...call, source: 'compaction', generationId });
1856
1864
  let currentInferenceId: string | undefined;
1857
1865
  let latestInferenceVisibleIds = new Set(this.events.map((event) => event.id));
1858
1866
  const completedInferenceIds = new Set<string>();
@@ -1927,6 +1935,12 @@ export class AgentSessionEngine implements IAgentSession {
1927
1935
  onError: ({ error }) => {
1928
1936
  streamError = error;
1929
1937
  },
1938
+ onLanguageModelCallStart: () => {
1939
+ modelCalls.start();
1940
+ },
1941
+ onLanguageModelCallEnd: ({ modelId, usage }) => {
1942
+ modelCalls.end(modelId, usage);
1943
+ },
1930
1944
  repairToolCall: async ({ toolCall, tools: availableTools, error }) => {
1931
1945
  const lowerName = toolCall.toolName.toLowerCase();
1932
1946
  if (
@@ -2169,10 +2183,6 @@ export class AgentSessionEngine implements IAgentSession {
2169
2183
  completedInferenceIds.add(currentInferenceId);
2170
2184
  stepCount++;
2171
2185
  }
2172
- totalInput += step.usage.inputTokens ?? 0;
2173
- totalOutput += step.usage.outputTokens ?? 0;
2174
- totalCacheRead += step.usage.inputTokenDetails.cacheReadTokens ?? 0;
2175
- totalCacheWrite += step.usage.inputTokenDetails.cacheWriteTokens ?? 0;
2176
2186
  for (const toolCall of step.toolCalls) {
2177
2187
  recordToolCall(toolCalls, toolCallIndexes, toolCall);
2178
2188
  }
@@ -2192,6 +2202,7 @@ export class AgentSessionEngine implements IAgentSession {
2192
2202
  );
2193
2203
  await this.flushPersistence();
2194
2204
  attempt = 0;
2205
+ retriedMs = 0;
2195
2206
  contextOverflowRetries = 0;
2196
2207
  },
2197
2208
  });
@@ -2200,6 +2211,9 @@ export class AgentSessionEngine implements IAgentSession {
2200
2211
  const finishReason = await result.finishReason;
2201
2212
  await result.response;
2202
2213
  if (streamError) throw streamError;
2214
+ // A stream whose call was aborted can end without an error; its call was aborted, not
2215
+ // answered without usage.
2216
+ modelCalls.settle(combinedAbort.signal.aborted ? 'aborted' : 'missing');
2203
2217
  flushReasoning();
2204
2218
  await this.flushPersistence();
2205
2219
 
@@ -2227,6 +2241,7 @@ export class AgentSessionEngine implements IAgentSession {
2227
2241
  completedResult = generationResult;
2228
2242
  return generationResult;
2229
2243
  } catch (error) {
2244
+ modelCalls.settle(combinedAbort.signal.aborted ? 'aborted' : 'failed');
2230
2245
  flushReasoning();
2231
2246
  const effectiveError = streamError ?? error;
2232
2247
  if (combinedAbort.signal.aborted) {
@@ -2249,17 +2264,30 @@ export class AgentSessionEngine implements IAgentSession {
2249
2264
  stepCount++;
2250
2265
  }
2251
2266
 
2252
- if (isRetryableError(effectiveError) && attempt < MAX_RETRY_ATTEMPTS && stepCount < maxSteps) {
2253
- attempt++;
2254
- const errorWithHeaders = effectiveError as {
2255
- responseHeaders?: Record<string, string>;
2256
- headers?: Record<string, string>;
2257
- };
2258
- await sleep(
2259
- retryDelay(attempt, errorWithHeaders.responseHeaders ?? errorWithHeaders.headers),
2260
- combinedAbort.signal,
2261
- );
2262
- continue;
2267
+ const retry = stepCount < maxSteps
2268
+ ? planModelCallRetry(effectiveError, {
2269
+ attempt,
2270
+ retriedMs,
2271
+ provider: runtime.model.provider,
2272
+ now: Date.now(),
2273
+ })
2274
+ : { action: 'not-retryable' as const };
2275
+ switch (retry.action) {
2276
+ case 'fail':
2277
+ throw retry.error;
2278
+ case 'retry':
2279
+ attempt++;
2280
+ retriedMs += retry.delayMs;
2281
+ this.options.onRetry?.({
2282
+ attempt,
2283
+ maxAttempts: MAX_RETRY_ATTEMPTS,
2284
+ delayMs: retry.delayMs,
2285
+ reason: retry.reason,
2286
+ });
2287
+ await sleep(retry.delayMs, combinedAbort.signal);
2288
+ continue;
2289
+ case 'not-retryable':
2290
+ break;
2263
2291
  }
2264
2292
 
2265
2293
  if (isContextOverflow(effectiveError)) {
@@ -2273,7 +2301,11 @@ export class AgentSessionEngine implements IAgentSession {
2273
2301
  }
2274
2302
  combinedAbort.signal.throwIfAborted();
2275
2303
  contextOverflowRetries++;
2276
- await this.performCompaction('context-overflow', combinedAbort.signal);
2304
+ await this.performCompaction(
2305
+ 'context-overflow',
2306
+ reportCompactionUsage,
2307
+ combinedAbort.signal,
2308
+ );
2277
2309
  continue;
2278
2310
  }
2279
2311
  throw effectiveError;
package/ts_agent/index.ts CHANGED
@@ -14,6 +14,8 @@ export {
14
14
  modelMessagesToAgentEvents,
15
15
  } from './smartagent.events.js';
16
16
  export { ToolRegistry } from './smartagent.classes.toolregistry.js';
17
+ export { AgentModelCallUsageRecorder } from './smartagent.usage.js';
18
+ export type { IAgentModelCallUsageRecorderModel } from './smartagent.usage.js';
17
19
  export {
18
20
  AgentGenerationLeaseCleanupError,
19
21
  ContextOverflowError,
@@ -48,6 +50,7 @@ export type {
48
50
  IAgentToolCallRecord,
49
51
  IAgentToolCallStartEvent,
50
52
  IAgentToolCallUpdateEvent,
53
+ IAgentUsage,
51
54
  ProviderOptions,
52
55
  TAgentCacheRetention,
53
56
  TAgentCacheSetting,
@@ -55,6 +58,10 @@ export type {
55
58
  TAgentContextBuilder,
56
59
  TAgentContextCompactor,
57
60
  TAgentGenerationPrepare,
61
+ TAgentModelCallUnreportedReason,
62
+ TAgentModelCallUsage,
63
+ TAgentModelCallUsageEvent,
64
+ TAgentModelCallUsageReporter,
58
65
  TAgentSessionChangeListener,
59
66
  TAgentToolCallFinishEvent,
60
67
  TAgentToolExecutionReconciliationOptions,
@@ -110,6 +117,7 @@ export type {
110
117
  IAgentGenerationTransaction,
111
118
  TAgentGenerationTransactionState,
112
119
  } from './smartagent.transactions.js';
120
+ export type { IAgentRetryEvent, TAgentRetryReason } from './smartagent.retry.js';
113
121
  export * from './tool.contracts.js';
114
122
  export * from './tool.persistence.js';
115
123
  export * from './tool.adapter.js';
@@ -5,6 +5,7 @@ export { streamText, generateText, stepCountIs, wrapLanguageModel };
5
5
 
6
6
  export type {
7
7
  AssistantModelMessage,
8
+ LanguageModelUsage,
8
9
  ModelMessage,
9
10
  StepResult,
10
11
  SystemModelMessage,
@@ -18,15 +19,25 @@ export type {
18
19
  // model contracts and AI SDK
19
20
  import {
20
21
  applySmartAiCacheProviderOptions,
22
+ createModelLimitInfo,
21
23
  createSmartAiCachingMiddleware,
24
+ isModelLimitError,
25
+ isModelLimitInfo,
22
26
  jsonSchema,
27
+ ModelLimitError,
28
+ readRetryAfterMs,
23
29
  resolveSmartAiCacheProvider,
24
30
  tool,
25
31
  } from '@modelprofile.com/flexharness-models';
26
32
 
27
33
  export {
28
34
  applySmartAiCacheProviderOptions,
35
+ createModelLimitInfo,
29
36
  createSmartAiCachingMiddleware,
37
+ isModelLimitError,
38
+ isModelLimitInfo,
39
+ ModelLimitError,
40
+ readRetryAfterMs,
30
41
  resolveSmartAiCacheProvider,
31
42
  tool,
32
43
  jsonSchema,
@@ -88,7 +88,7 @@ The following Node examples reuse `setup` and, where needed, `tools` from the qu
88
88
  | `messages` | Current AI SDK message history after projection or compaction. Pass it into another run to continue. |
89
89
  | `steps` | Completed model steps, including steps from validation-triggered attempts. A step can call several tools. |
90
90
  | `finishReason` | The model's final finish reason; inspect this together with application validation. |
91
- | `usage` | Input, output and total tokens, plus cache-read and cache-write tokens. |
91
+ | `usage` | Input, output and total tokens, plus cache-read and cache-write tokens, summed over every model call of the run the provider reported: its model steps, retried calls included, and the reported calls of a compaction its context overflow caused. |
92
92
  | `toolCalls` | Tool-call IDs, names and inputs, with available outputs or errors. |
93
93
 
94
94
  ```typescript
@@ -153,9 +153,69 @@ Pass these callbacks to `runAgent` or `AgentSession.create()`:
153
153
  | `onToolCallStart(event)` | `toolCallId`, `toolName`, `input`. |
154
154
  | `onToolCallUpdate(event)` | The call identity and a transient streamed `output`. |
155
155
  | `onToolCallFinish(event)` | The call identity plus either `success: true, output` or `success: false, error`. |
156
+ | `onRetry(event)` | Before each wait to retry a model call: `attempt`, `maxAttempts`, `delayMs` and `reason` (`rate_limit`, `overloaded`, `unavailable`). |
157
+ | `onUsage(event)` | Once per model call, as soon as its usage is known; see [Count the usage of every run](#count-the-usage-of-every-run). |
156
158
 
157
159
  Tool updates are transient; the finish callback carries the authoritative final output. Use `subscribe()` for committed session changes (`committed`, `updated`, `archived`), and retain its returned unsubscribe function. Session listeners are delivered in order per listener, have bounded queues and timeouts, and are removed on failure. Streaming callbacks and session-change listeners serve different purposes.
158
160
 
161
+ ## Count the usage of every run
162
+
163
+ A run that throws or is aborted has used tokens too, and its promise carries no result. `onUsage` reports each model call once, as soon as its usage is known, whatever the run's outcome. Sum it to count a run's usage:
164
+
165
+ ```typescript
166
+ import type { IAgentUsage } from '@modelprofile.com/flexharness-agent';
167
+
168
+ const used: IAgentUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 };
169
+ let usageComplete = true;
170
+ try {
171
+ await runAgent({
172
+ ...setup,
173
+ tools,
174
+ prompt: 'Convert 10 km to miles.',
175
+ abort: AbortSignal.timeout(60_000),
176
+ onUsage: (event) => {
177
+ if (event.status === 'unreported') {
178
+ usageComplete = false;
179
+ return;
180
+ }
181
+ for (const key of Object.keys(used) as (keyof IAgentUsage)[]) used[key] += event.usage[key];
182
+ },
183
+ });
184
+ } finally {
185
+ console.log(used, usageComplete ? 'complete' : 'lower bound');
186
+ }
187
+ ```
188
+
189
+ | Event field | Meaning |
190
+ | --- | --- |
191
+ | `source` | `generation`: a model step of the generation. `compaction`: a model call of a context compaction. |
192
+ | `generationId` | The generation the call belongs to. A compaction carries the generation whose context overflow caused it; a compaction by `compact()` or event retention has none. |
193
+ | `provider`, `requestedModelId` | The model the call was made with: the language model's `provider` and `modelId`. Every event carries both, whatever its status, so key usage caps on them. |
194
+ | `status: 'reported'`, `usage`, `responseModelId` | The provider reported the call's usage. A count the provider leaves out is zero. `responseModelId` is the model id the provider's response named (a provider may answer with a dated model version), or `requestedModelId` when it named none. |
195
+ | `status: 'unreported'`, `reason` | The call ended before the provider reported its usage: `aborted` (the call was aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage). The provider may still have consumed tokens for it; their number is unknown. |
196
+
197
+ When a run returns, its reported calls sum to `result.usage`; count one or the other, not both. A call is reported when the provider's response ends, before its tool calls run, so a run aborted during a tool call still reports the call that requested it. Retried calls and validation retries are included. `onUsage` must not throw; an error it throws is reported like a session listener error and does not change the run's outcome. `runAgent` delivers every event of the run before its promise settles. `AgentSession.create()` accepts the same callback for every generation and compaction of the session; `generate()` and `scheduleGenerate()` without a `transaction` reject as soon as their abort signal fires and may report the interrupted call afterwards; `close()` waits for that report. With a `transaction`, they settle only after every call of the generation has been reported.
198
+
199
+ ### What is counted
200
+
201
+ - Every model step of a generation, whatever its outcome, including retried calls and validation retries.
202
+ - The model calls of a context compaction, when the compactor reports them. `contextCompactor` and `onContextOverflow` receive `reportUsage` in their options; it reports into `onUsage` with `source: 'compaction'`. `compactMessages()` from `@modelprofile.com/flexharness/compaction` reports each attempt of its model call when it receives `reportUsage`, so pass the handler's options through:
203
+
204
+ ```typescript
205
+ import { compactMessages } from '@modelprofile.com/flexharness/compaction';
206
+
207
+ const session = await AgentSession.create({
208
+ ...setup,
209
+ contextCompactor: (messages, _events, options) => compactMessages(setup.model, messages, options),
210
+ onContextOverflow: (messages, options) => compactMessages(setup.model, messages, options),
211
+ onUsage: (event) => console.log(event),
212
+ });
213
+ ```
214
+
215
+ A handler that makes its own model calls reports each one through `reportUsage` before its promise settles, a call that fails or is aborted as `unreported`. `AgentModelCallUsageRecorder` does the bookkeeping for one language model: call `start()` when a call begins, `end(responseModelId, usage)` with the AI SDK's `onLanguageModelCallEnd` values, and `settle(reason)` when a call ends otherwise.
216
+
217
+ Not counted: model calls a handler makes without reporting them through `reportUsage`, and model calls outside the session, such as tools that call models themselves. Report those in your own accounting.
218
+
159
219
  ## Validate an answer and request corrections
160
220
 
161
221
  `validateCompletion` returns `void` to accept a result or a string to add a corrective user message and generate again. `maxValidationRetries` defaults to `0`: a failed validation throws unless retries are configured.
@@ -242,9 +302,9 @@ Use an application-owned durable adapter for persistent sessions. Its `save(sess
242
302
  ### Bound model context and active events
243
303
 
244
304
  - `contextBuilder({ events })` controls the model-message projection.
245
- - `contextCompactor(messages, events, { reason, abortSignal })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention.
305
+ - `contextCompactor(messages, events, { reason, abortSignal, reportUsage })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention. Report the usage of its model calls through `reportUsage`; see [What is counted](#what-is-counted).
246
306
  - `eventRetention: { maxEvents }` triggers compaction and archival when the active event count exceeds the threshold. It also requires an event store with `archive()` support.
247
- - Context overflow invokes the configured compactor, or the `onContextOverflow` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
307
+ - Context overflow invokes the configured compactor, or the `onContextOverflow(messages, { abortSignal, reportUsage })` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
248
308
  - Open transactions and uncertain tool intents constrain when compaction and archival can proceed. Resolve them before manual compaction.
249
309
 
250
310
  Compaction changes the active model context; archival moves covered events out of the active event set. Implement archive retention in your chosen store when you need a complete audit history.
@@ -268,7 +328,7 @@ For generation-scoped resources, `generate({ prepare })` accepts a callback that
268
328
 
269
329
  Model transports, tools and resource callbacks must observe their supplied abort signals. Background jobs have separate lifecycles; closing a session does not automatically terminate them. Cleanup failures are observable. `closeCleanupCompleted` reports whether the session has released its retryable cleanup ownership.
270
330
 
271
- The runtime retries recognized transient provider failures with backoff and preserves completed tool results during a generation's retry sequence. A ledger reuses an already observed execution for the same tool-call identity. Applications still need their own idempotency and recovery rules for external side effects.
331
+ The runtime retries a model call that failed with a rate limit (429), an overloaded provider (529) or an unavailable one (503) at most 8 times, waiting for the provider's `retry-after` delay but at least the backoff that doubles from 2 s to 30 s. A requested delay beyond 60 s, or one that would take the call's retries past 150 s in total, fails the generation: a rate limit (429) as a `ModelLimitError` of kind `rate_limit` carrying that retry time, a 503 or 529 with the provider error. A `ModelLimitError` of kind `usage_limit`, which the provider adapters raise for a spent quota, is never retried. Completed tool results are preserved during a generation's retry sequence. A ledger reuses an already observed execution for the same tool-call identity. Applications still need their own idempotency and recovery rules for external side effects.
272
332
 
273
333
  `maxSteps` defaults to `20` and limits completed model steps per generation. Prompt-cache handling defaults to `cache: 'auto'`; use `cache: false` to disable the runtime's cache defaults, or pass an explicit cache policy and a stable `sessionId` for supported provider affinity.
274
334
 
@@ -37,4 +37,11 @@ export function runAgent(options: IAgentRunOptions): Promise<IAgentRunResult> {
37
37
  return runAgentWithSession(options, (sessionOptions) => RunnerSession.create(sessionOptions));
38
38
  }
39
39
 
40
- export type { IAgentRunOptions, IAgentRunResult } from './smartagent.interfaces.js';
40
+ export type {
41
+ IAgentRunOptions,
42
+ IAgentRunResult,
43
+ IAgentUsage,
44
+ TAgentModelCallUnreportedReason,
45
+ TAgentModelCallUsage,
46
+ TAgentModelCallUsageEvent,
47
+ } from './smartagent.interfaces.js';
@@ -51,6 +51,8 @@ export async function runAgentWithSession(
51
51
  onToolCallStart: options.onToolCallStart,
52
52
  onToolCallUpdate: options.onToolCallUpdate,
53
53
  onToolCallFinish: options.onToolCallFinish,
54
+ onRetry: options.onRetry,
55
+ onUsage: options.onUsage,
54
56
  onToolCall: options.onToolCall,
55
57
  onToolResult: options.onToolResult,
56
58
  onContextOverflow: options.onContextOverflow,
@@ -17,6 +17,7 @@ import type {
17
17
  TAgentToolResultOutput,
18
18
  } from './smartagent.events.js';
19
19
  import type { TAgentEventArchive, TAgentEventStore } from './smartagent.persistence.js';
20
+ import type { IAgentRetryEvent } from './smartagent.retry.js';
20
21
 
21
22
  export type { ProviderOptions };
22
23
  export interface IAgentCacheOptions extends ISmartAiCacheOptions {}
@@ -53,8 +54,69 @@ export type TAgentToolCallFinishEvent =
53
54
  error: string;
54
55
  });
55
56
 
57
+ /** Token usage of model calls. A count a provider leaves out counts as zero. */
58
+ export interface IAgentUsage {
59
+ inputTokens: number;
60
+ outputTokens: number;
61
+ totalTokens: number;
62
+ cacheReadTokens: number;
63
+ cacheWriteTokens: number;
64
+ }
65
+
66
+ interface IAgentModelCallIdentity {
67
+ /** The provider of the model that was called, as the language model names it. */
68
+ provider: string;
69
+ /**
70
+ * The model id the call was made with: the language model's `modelId`. Every call carries it,
71
+ * whatever its outcome, so key usage caps on `provider` and `requestedModelId`.
72
+ */
73
+ requestedModelId: string;
74
+ }
75
+
76
+ /** Why a model call ended without the provider reporting its usage. */
77
+ export type TAgentModelCallUnreportedReason = 'aborted' | 'failed' | 'missing';
78
+
79
+ /**
80
+ * The usage outcome of one model call, as the component that made the call reports it.
81
+ * `reported`: the provider reported the call's usage. `responseModelId` is the model id the
82
+ * provider's response named, or `requestedModelId` when the response named none.
83
+ * `unreported`: the call ended before the provider reported its usage: `aborted` (the call was
84
+ * aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage).
85
+ * The provider may still have consumed tokens for it; their number is unknown.
86
+ */
87
+ export type TAgentModelCallUsage =
88
+ | (IAgentModelCallIdentity & {
89
+ status: 'reported';
90
+ responseModelId: string;
91
+ usage: IAgentUsage;
92
+ })
93
+ | (IAgentModelCallIdentity & {
94
+ status: 'unreported';
95
+ reason: TAgentModelCallUnreportedReason;
96
+ });
97
+
98
+ /** Receives the usage outcome of each model call. Must not throw. */
99
+ export type TAgentModelCallUsageReporter = (call: TAgentModelCallUsage) => void;
100
+
101
+ /**
102
+ * The usage outcome of one model call of a session, reported once per call through `onUsage`.
103
+ * `source: 'generation'`: a model step of the generation `generationId`.
104
+ * `source: 'compaction'`: a model call of a context compaction. `generationId` names the generation
105
+ * whose context overflow caused it; it is absent for `compact()` and event retention.
106
+ */
107
+ export type TAgentModelCallUsageEvent = TAgentModelCallUsage & (
108
+ | { source: 'generation'; generationId: string }
109
+ | { source: 'compaction'; generationId?: string }
110
+ );
111
+
56
112
  export interface IAgentContextOverflowInvocationOptions {
57
113
  abortSignal?: AbortSignal;
114
+ /**
115
+ * Reports the usage of each model call the handler makes, into the session's `onUsage` as
116
+ * `source: 'compaction'`. The session always supplies it. Report every call before the returned
117
+ * promise settles, including calls that fail or are aborted.
118
+ */
119
+ reportUsage?: TAgentModelCallUsageReporter;
58
120
  }
59
121
 
60
122
  export interface IAgentContextBuildOptions {
@@ -68,6 +130,12 @@ export type TAgentContextBuilder = (
68
130
  export interface IAgentContextCompactionOptions {
69
131
  abortSignal?: AbortSignal;
70
132
  reason: 'context-overflow' | 'retention' | 'manual';
133
+ /**
134
+ * Reports the usage of each model call the compactor makes, into the session's `onUsage` as
135
+ * `source: 'compaction'`. The session always supplies it. Report every call before the returned
136
+ * promise settles, including calls that fail or are aborted.
137
+ */
138
+ reportUsage?: TAgentModelCallUsageReporter;
71
139
  }
72
140
 
73
141
  export type TAgentContextCompactor = (
@@ -150,6 +218,18 @@ export interface IAgentSessionOptions {
150
218
  onToolCallUpdate?: (event: IAgentToolCallUpdateEvent) => void;
151
219
  /** Called when a tool call finishes, with its stable AI SDK call id and success state. */
152
220
  onToolCallFinish?: (event: TAgentToolCallFinishEvent) => void;
221
+ /** Called before the engine waits to retry a rate-limited or unavailable model call. */
222
+ onRetry?: (event: IAgentRetryEvent) => void;
223
+ /**
224
+ * Called once for every model call of the session, as soon as its usage is known: when the
225
+ * provider reports it, or when the call ends without it. This covers the generation's model
226
+ * steps, the model call of `compactMessages()` when a compactor passes it `reportUsage`, and every
227
+ * call a `contextCompactor` or `onContextOverflow` handler reports through `reportUsage`. The
228
+ * reported calls of a generation, including those of a compaction its context overflow caused,
229
+ * sum to the generation's `usage` when it returns. Must not throw; an error it throws is reported
230
+ * like a session listener error and does not change the generation's outcome.
231
+ */
232
+ onUsage?: (event: TAgentModelCallUsageEvent) => void;
153
233
  /** @deprecated Use onToolCallStart instead. */
154
234
  onToolCall?: (toolName: string, input: unknown) => void;
155
235
  /** @deprecated Use onToolCallFinish instead. */
@@ -300,14 +380,11 @@ export interface IAgentRunResult {
300
380
  steps: number;
301
381
  /** Finish reason from the final step */
302
382
  finishReason: string;
303
- /** Accumulated token usage across all steps */
304
- usage: {
305
- inputTokens: number;
306
- outputTokens: number;
307
- totalTokens: number;
308
- cacheReadTokens: number;
309
- cacheWriteTokens: number;
310
- };
383
+ /**
384
+ * Token usage the provider reported for every model call of the run: its model steps, retried
385
+ * calls included, and the calls of a compaction its context overflow caused that were reported
386
+ */
387
+ usage: IAgentUsage;
311
388
  /** Tool calls observed during the run, including inputs and outputs/errors when available */
312
389
  toolCalls: IAgentToolCallRecord[];
313
390
  }