@modelprofile.com/flexharness-agent 8.2.0 → 8.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist_ts_agent/classes.sessionengine.d.ts +3 -0
- package/dist_ts_agent/classes.sessionengine.js +70 -52
- package/dist_ts_agent/index.d.ts +4 -1
- package/dist_ts_agent/index.js +2 -1
- package/dist_ts_agent/plugins.d.ts +3 -3
- package/dist_ts_agent/plugins.js +3 -3
- package/dist_ts_agent/runner.d.ts +1 -1
- package/dist_ts_agent/runtime.run.js +3 -1
- package/dist_ts_agent/smartagent.interfaces.d.ts +80 -8
- package/dist_ts_agent/smartagent.interfaces.js +1 -1
- package/dist_ts_agent/smartagent.retry.d.ts +39 -0
- package/dist_ts_agent/smartagent.retry.js +67 -0
- package/dist_ts_agent/smartagent.usage.d.ts +25 -0
- package/dist_ts_agent/smartagent.usage.js +53 -0
- package/package.json +2 -2
- package/readme.md +64 -4
- package/ts_agent/classes.sessionengine.ts +88 -56
- package/ts_agent/index.ts +8 -0
- package/ts_agent/plugins.ts +11 -0
- package/ts_agent/readme.md +64 -4
- package/ts_agent/runner.ts +8 -1
- package/ts_agent/runtime.run.ts +2 -0
- package/ts_agent/smartagent.interfaces.ts +85 -8
- package/ts_agent/smartagent.retry.ts +101 -0
- package/ts_agent/smartagent.usage.ts +66 -0
package/dist_ts_agent/index.d.ts
CHANGED
|
@@ -4,12 +4,15 @@ export { buildModelMessages } from './smartagent.context.js';
|
|
|
4
4
|
export { filterModelVisibleAgentEvents, getAgentGenerationTransactions, getOpenAgentGenerationTransactions, getUncertainToolExecutionIntents, isTransactionalGeneration, } from './smartagent.transactions.js';
|
|
5
5
|
export { createAgentEvent, createAgentEventId, modelMessagesToAgentEvents, } from './smartagent.events.js';
|
|
6
6
|
export { ToolRegistry } from './smartagent.classes.toolregistry.js';
|
|
7
|
+
export { AgentModelCallUsageRecorder } from './smartagent.usage.js';
|
|
8
|
+
export type { IAgentModelCallUsageRecorderModel } from './smartagent.usage.js';
|
|
7
9
|
export { AgentGenerationLeaseCleanupError, ContextOverflowError, } from './smartagent.interfaces.js';
|
|
8
10
|
export { AgentEventStoreConflictError, InMemoryAgentEventStore, isAgentEventStoreV2, validateAgentEventArchiveV2, validateAgentEventSnapshotV2, } from './smartagent.persistence.js';
|
|
9
|
-
export type { IAgentCacheOptions, IAgentContextOverflowInvocationOptions, IAgentContextBuildOptions, IAgentContextCompactionOptions, IAgentCompactOptions, IAgentEventRetentionOptions, IAgentBeginGenerationOptions, IAgentGenerateOptions, IAgentGenerateResult, IAgentGenerationHandle, IAgentGenerationLease, IAgentGenerationPreparationContext, IAgentRunOptions, IAgentRunResult, IAgentSession, IAgentSessionAbortOptions, IAgentSessionChange, IAgentSessionOptions, IAgentScheduleGenerateOptions, IAgentToolCallRecord, IAgentToolCallStartEvent, IAgentToolCallUpdateEvent, ProviderOptions, TAgentCacheRetention, TAgentCacheSetting, TAgentPrompt, TAgentContextBuilder, TAgentContextCompactor, TAgentGenerationPrepare, TAgentSessionChangeListener, TAgentToolCallFinishEvent, TAgentToolExecutionReconciliationOptions, } from './smartagent.interfaces.js';
|
|
11
|
+
export type { IAgentCacheOptions, IAgentContextOverflowInvocationOptions, IAgentContextBuildOptions, IAgentContextCompactionOptions, IAgentCompactOptions, IAgentEventRetentionOptions, IAgentBeginGenerationOptions, IAgentGenerateOptions, IAgentGenerateResult, IAgentGenerationHandle, IAgentGenerationLease, IAgentGenerationPreparationContext, IAgentRunOptions, IAgentRunResult, IAgentSession, IAgentSessionAbortOptions, IAgentSessionChange, IAgentSessionOptions, IAgentScheduleGenerateOptions, IAgentToolCallRecord, IAgentToolCallStartEvent, IAgentToolCallUpdateEvent, IAgentUsage, ProviderOptions, TAgentCacheRetention, TAgentCacheSetting, TAgentPrompt, TAgentContextBuilder, TAgentContextCompactor, TAgentGenerationPrepare, TAgentModelCallUnreportedReason, TAgentModelCallUsage, TAgentModelCallUsageEvent, TAgentModelCallUsageReporter, TAgentSessionChangeListener, TAgentToolCallFinishEvent, TAgentToolExecutionReconciliationOptions, } from './smartagent.interfaces.js';
|
|
10
12
|
export type { IAgentEventArchive, IAgentEventArchiveV2, IAgentEventSnapshot, IAgentEventSnapshotV2, IAgentEventStore, IAgentEventStoreV2, TAgentEventArchive, TAgentEventSnapshot, TAgentEventStore, } from './smartagent.persistence.js';
|
|
11
13
|
export type { IAgentEventBase, IAgentRuntimeEventPayload, IArchivedAgentGenerationTransaction, IAssistantMessageEvent, IContextCompactionEvent, IGenerationBegunEvent, IGenerationExecutionCompletedEvent, IGenerationExecutionStartedEvent, IGenerationOutcomeEvent, IModelMessageEvent, IModelMessagesToEventsOptions, IModelMessageEventIdentityCoordinates, IRuntimeEvent, ISystemMessageEvent, IToolCallEvent, IToolExecutionIntentEvent, IToolExecutionReconciliationEvent, IToolResultEvent, IUserMessageEvent, TAgentAssistantMessage, TAgentConversationEvent, TAgentControlEvent, TAgentEvent, TAgentEventInput, TAgentSystemMessage, TAgentGenerationOutcome, TAgentToolCallPart, TAgentToolMessage, TAgentToolResultOutput, TAgentToolResultPart, TAgentUserMessage, TModelMessageEmittedEventKind, TModelMessageEventIdentityFactory, TToolExecutionReconciliation, } from './smartagent.events.js';
|
|
12
14
|
export type { IAgentGenerationTransaction, TAgentGenerationTransactionState, } from './smartagent.transactions.js';
|
|
15
|
+
export type { IAgentRetryEvent, TAgentRetryReason } from './smartagent.retry.js';
|
|
13
16
|
export * from './tool.contracts.js';
|
|
14
17
|
export * from './tool.persistence.js';
|
|
15
18
|
export * from './tool.adapter.js';
|
package/dist_ts_agent/index.js
CHANGED
|
@@ -4,6 +4,7 @@ export { buildModelMessages } from './smartagent.context.js';
|
|
|
4
4
|
export { filterModelVisibleAgentEvents, getAgentGenerationTransactions, getOpenAgentGenerationTransactions, getUncertainToolExecutionIntents, isTransactionalGeneration, } from './smartagent.transactions.js';
|
|
5
5
|
export { createAgentEvent, createAgentEventId, modelMessagesToAgentEvents, } from './smartagent.events.js';
|
|
6
6
|
export { ToolRegistry } from './smartagent.classes.toolregistry.js';
|
|
7
|
+
export { AgentModelCallUsageRecorder } from './smartagent.usage.js';
|
|
7
8
|
export { AgentGenerationLeaseCleanupError, ContextOverflowError, } from './smartagent.interfaces.js';
|
|
8
9
|
export { AgentEventStoreConflictError, InMemoryAgentEventStore, isAgentEventStoreV2, validateAgentEventArchiveV2, validateAgentEventSnapshotV2, } from './smartagent.persistence.js';
|
|
9
10
|
export * from './tool.contracts.js';
|
|
@@ -12,4 +13,4 @@ export * from './tool.adapter.js';
|
|
|
12
13
|
export { tool, jsonSchema, stepCountIs } from 'ai';
|
|
13
14
|
export { z } from 'zod';
|
|
14
15
|
export { FileAgentEventStore } from './persistence.file.js';
|
|
15
|
-
//# sourceMappingURL=data:application/json;base64,
|
|
16
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoiaW5kZXguanMiLCJzb3VyY2VSb290IjoiIiwic291cmNlcyI6WyIuLi90c19hZ2VudC9pbmRleC50cyJdLCJuYW1lcyI6W10sIm1hcHBpbmdzIjoiQUFBQSxPQUFPLEVBQUUsUUFBUSxFQUFFLE1BQU0sK0JBQStCLENBQUM7QUFDekQsT0FBTyxFQUFFLFlBQVksRUFBRSxNQUFNLGlDQUFpQyxDQUFDO0FBQy9ELE9BQU8sRUFBRSxrQkFBa0IsRUFBRSxNQUFNLHlCQUF5QixDQUFDO0FBQzdELE9BQU8sRUFDTCw2QkFBNkIsRUFDN0IsOEJBQThCLEVBQzlCLGtDQUFrQyxFQUNsQyxnQ0FBZ0MsRUFDaEMseUJBQXlCLEdBQzFCLE1BQU0sOEJBQThCLENBQUM7QUFDdEMsT0FBTyxFQUNMLGdCQUFnQixFQUNoQixrQkFBa0IsRUFDbEIsMEJBQTBCLEdBQzNCLE1BQU0sd0JBQXdCLENBQUM7QUFDaEMsT0FBTyxFQUFFLFlBQVksRUFBRSxNQUFNLHNDQUFzQyxDQUFDO0FBQ3BFLE9BQU8sRUFBRSwyQkFBMkIsRUFBRSxNQUFNLHVCQUF1QixDQUFDO0FBRXBFLE9BQU8sRUFDTCxnQ0FBZ0MsRUFDaEMsb0JBQW9CLEdBQ3JCLE1BQU0sNEJBQTRCLENBQUM7QUFDcEMsT0FBTyxFQUNMLDRCQUE0QixFQUM1Qix1QkFBdUIsRUFDdkIsbUJBQW1CLEVBQ25CLDJCQUEyQixFQUMzQiw0QkFBNEIsR0FDN0IsTUFBTSw2QkFBNkIsQ0FBQztBQTRGckMsY0FBYyxxQkFBcUIsQ0FBQztBQUNwQyxjQUFjLHVCQUF1QixDQUFDO0FBQ3RDLGNBQWMsbUJBQW1CLENBQUM7QUFDbEMsT0FBTyxFQUFFLElBQUksRUFBRSxVQUFVLEVBQUUsV0FBVyxFQUFFLE1BQU0sSUFBSSxDQUFDO0FBQ25ELE9BQU8sRUFBRSxDQUFDLEVBQUUsTUFBTSxLQUFLLENBQUM7QUFFeEIsT0FBTyxFQUFFLG1CQUFtQixFQUFvQyxNQUFNLHVCQUF1QixDQUFDIn0=
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { streamText, generateText, stepCountIs, wrapLanguageModel } from 'ai';
|
|
2
2
|
export { streamText, generateText, stepCountIs, wrapLanguageModel };
|
|
3
|
-
export type { AssistantModelMessage, ModelMessage, StepResult, SystemModelMessage, ToolSet, ToolModelMessage, ToolExecutionOptions, StreamTextResult, UserModelMessage, } from 'ai';
|
|
4
|
-
import { applySmartAiCacheProviderOptions, createSmartAiCachingMiddleware, jsonSchema, resolveSmartAiCacheProvider, tool } from '@modelprofile.com/flexharness-models';
|
|
5
|
-
export { applySmartAiCacheProviderOptions, createSmartAiCachingMiddleware, resolveSmartAiCacheProvider, tool, jsonSchema, };
|
|
3
|
+
export type { AssistantModelMessage, LanguageModelUsage, ModelMessage, StepResult, SystemModelMessage, ToolSet, ToolModelMessage, ToolExecutionOptions, StreamTextResult, UserModelMessage, } from 'ai';
|
|
4
|
+
import { applySmartAiCacheProviderOptions, createModelLimitInfo, createSmartAiCachingMiddleware, isModelLimitError, isModelLimitInfo, jsonSchema, ModelLimitError, readRetryAfterMs, resolveSmartAiCacheProvider, tool } from '@modelprofile.com/flexharness-models';
|
|
5
|
+
export { applySmartAiCacheProviderOptions, createModelLimitInfo, createSmartAiCachingMiddleware, isModelLimitError, isModelLimitInfo, ModelLimitError, readRetryAfterMs, resolveSmartAiCacheProvider, tool, jsonSchema, };
|
|
6
6
|
export type { ISmartAiCacheOptions, TSmartAiLanguageModel, TSmartAiCacheRetention, TSmartAiCacheSetting, TSmartAiProviderOptions as ProviderOptions, } from '@modelprofile.com/flexharness-models';
|
|
7
7
|
export type { IToolExecutionContext, IToolJobState, } from './tool.contracts.js';
|
|
8
8
|
import { z } from 'zod';
|
package/dist_ts_agent/plugins.js
CHANGED
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
import { streamText, generateText, stepCountIs, wrapLanguageModel } from 'ai';
|
|
3
3
|
export { streamText, generateText, stepCountIs, wrapLanguageModel };
|
|
4
4
|
// model contracts and AI SDK
|
|
5
|
-
import { applySmartAiCacheProviderOptions, createSmartAiCachingMiddleware, jsonSchema, resolveSmartAiCacheProvider, tool, } from '@modelprofile.com/flexharness-models';
|
|
6
|
-
export { applySmartAiCacheProviderOptions, createSmartAiCachingMiddleware, resolveSmartAiCacheProvider, tool, jsonSchema, };
|
|
5
|
+
import { applySmartAiCacheProviderOptions, createModelLimitInfo, createSmartAiCachingMiddleware, isModelLimitError, isModelLimitInfo, jsonSchema, ModelLimitError, readRetryAfterMs, resolveSmartAiCacheProvider, tool, } from '@modelprofile.com/flexharness-models';
|
|
6
|
+
export { applySmartAiCacheProviderOptions, createModelLimitInfo, createSmartAiCachingMiddleware, isModelLimitError, isModelLimitInfo, ModelLimitError, readRetryAfterMs, resolveSmartAiCacheProvider, tool, jsonSchema, };
|
|
7
7
|
// zod
|
|
8
8
|
import { z } from 'zod';
|
|
9
9
|
export { z };
|
|
10
|
-
//# sourceMappingURL=data:application/json;base64,
|
|
10
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoicGx1Z2lucy5qcyIsInNvdXJjZVJvb3QiOiIiLCJzb3VyY2VzIjpbIi4uL3RzX2FnZW50L3BsdWdpbnMudHMiXSwibmFtZXMiOltdLCJtYXBwaW5ncyI6IkFBQUEsY0FBYztBQUNkLE9BQU8sRUFBRSxVQUFVLEVBQUUsWUFBWSxFQUFFLFdBQVcsRUFBRSxpQkFBaUIsRUFBRSxNQUFNLElBQUksQ0FBQztBQUU5RSxPQUFPLEVBQUUsVUFBVSxFQUFFLFlBQVksRUFBRSxXQUFXLEVBQUUsaUJBQWlCLEVBQUUsQ0FBQztBQWVwRSw2QkFBNkI7QUFDN0IsT0FBTyxFQUNMLGdDQUFnQyxFQUNoQyxvQkFBb0IsRUFDcEIsOEJBQThCLEVBQzlCLGlCQUFpQixFQUNqQixnQkFBZ0IsRUFDaEIsVUFBVSxFQUNWLGVBQWUsRUFDZixnQkFBZ0IsRUFDaEIsMkJBQTJCLEVBQzNCLElBQUksR0FDTCxNQUFNLHNDQUFzQyxDQUFDO0FBRTlDLE9BQU8sRUFDTCxnQ0FBZ0MsRUFDaEMsb0JBQW9CLEVBQ3BCLDhCQUE4QixFQUM5QixpQkFBaUIsRUFDakIsZ0JBQWdCLEVBQ2hCLGVBQWUsRUFDZixnQkFBZ0IsRUFDaEIsMkJBQTJCLEVBQzNCLElBQUksRUFDSixVQUFVLEdBQ1gsQ0FBQztBQWdCRixNQUFNO0FBQ04sT0FBTyxFQUFFLENBQUMsRUFBRSxNQUFNLEtBQUssQ0FBQztBQUV4QixPQUFPLEVBQUUsQ0FBQyxFQUFFLENBQUMifQ==
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
import type { IAgentRunOptions, IAgentRunResult } from './smartagent.interfaces.js';
|
|
2
2
|
/** Run in Node or a secure browser context, with internally owned cleanup. */
|
|
3
3
|
export declare function runAgent(options: IAgentRunOptions): Promise<IAgentRunResult>;
|
|
4
|
-
export type { IAgentRunOptions, IAgentRunResult } from './smartagent.interfaces.js';
|
|
4
|
+
export type { IAgentRunOptions, IAgentRunResult, IAgentUsage, TAgentModelCallUnreportedReason, TAgentModelCallUsage, TAgentModelCallUsageEvent, } from './smartagent.interfaces.js';
|
|
@@ -38,6 +38,8 @@ export async function runAgentWithSession(options, createSession) {
|
|
|
38
38
|
onToolCallStart: options.onToolCallStart,
|
|
39
39
|
onToolCallUpdate: options.onToolCallUpdate,
|
|
40
40
|
onToolCallFinish: options.onToolCallFinish,
|
|
41
|
+
onRetry: options.onRetry,
|
|
42
|
+
onUsage: options.onUsage,
|
|
41
43
|
onToolCall: options.onToolCall,
|
|
42
44
|
onToolResult: options.onToolResult,
|
|
43
45
|
onContextOverflow: options.onContextOverflow,
|
|
@@ -91,4 +93,4 @@ export async function runAgentWithSession(options, createSession) {
|
|
|
91
93
|
await session.close();
|
|
92
94
|
}
|
|
93
95
|
}
|
|
94
|
-
//# sourceMappingURL=data:application/json;base64,
|
|
96
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoicnVudGltZS5ydW4uanMiLCJzb3VyY2VSb290IjoiIiwic291cmNlcyI6WyIuLi90c19hZ2VudC9ydW50aW1lLnJ1bi50cyJdLCJuYW1lcyI6W10sIm1hcHBpbmdzIjoiQUFPQSxNQUFNLGNBQWMsR0FBRyxDQUNyQixNQUE4QixFQUM5QixNQUF1QyxFQUNqQyxFQUFFO0lBQ1IsTUFBTSxPQUFPLEdBQUcsSUFBSSxHQUFHLENBQUMsTUFBTSxDQUFDLEdBQUcsQ0FBQyxDQUFDLE1BQU0sRUFBRSxLQUFLLEVBQUUsRUFBRSxDQUFDLENBQUMsTUFBTSxDQUFDLFVBQVUsRUFBRSxLQUFLLENBQUMsQ0FBQyxDQUFDLENBQUM7SUFDbkYsS0FBSyxNQUFNLE1BQU0sSUFBSSxNQUFNLEVBQUUsQ0FBQztRQUM1QixNQUFNLGFBQWEsR0FBRyxPQUFPLENBQUMsR0FBRyxDQUFDLE1BQU0sQ0FBQyxVQUFVLENBQUMsQ0FBQztRQUNyRCxJQUFJLGFBQWEsS0FBSyxTQUFTLEVBQUUsQ0FBQztZQUNoQyxNQUFNLENBQUMsSUFBSSxDQUFDLEVBQUUsR0FBRyxNQUFNLEVBQUUsQ0FBQyxDQUFDO1lBQzNCLE9BQU8sQ0FBQyxHQUFHLENBQUMsTUFBTSxDQUFDLFVBQVUsRUFBRSxNQUFNLENBQUMsTUFBTSxHQUFHLENBQUMsQ0FBQyxDQUFDO1FBQ3BELENBQUM7YUFBTSxDQUFDO1lBQ04sTUFBTSxDQUFDLGFBQWEsQ0FBQyxHQUFHLEVBQUUsR0FBRyxNQUFNLENBQUMsYUFBYSxDQUFDLEVBQUUsR0FBRyxNQUFNLEVBQUUsQ0FBQztRQUNsRSxDQUFDO0lBQ0gsQ0FBQztBQUNILENBQUMsQ0FBQztBQUVGLE1BQU0sQ0FBQyxLQUFLLFVBQVUsbUJBQW1CLENBQ3ZDLE9BQXlCLEVBQ3pCLGFBQXdFO0lBRXhFLE1BQU0sT0FBTyxHQUFHLE1BQU0sYUFBYSxDQUFDO1FBQ2xDLEtBQUssRUFBRSxPQUFPLENBQUMsS0FBSztRQUNwQixNQUFNLEVBQUUsT0FBTyxDQUFDLE1BQU07UUFDdEIsS0FBSyxFQUFFLE9BQU8sQ0FBQyxLQUFLO1FBQ3BCLGVBQWUsRUFBRSxPQUFPLENBQUMsZUFBZTtRQUN4QyxTQUFTLEVBQUUsT0FBTyxDQUFDLFNBQVM7UUFDNUIsVUFBVSxFQUFFLE9BQU8sQ0FBQyxVQUFVO1FBQzlCLEtBQUssRUFBRSxPQUFPLENBQUMsS0FBSztRQUNwQixRQUFRLEVBQUUsT0FBTyxDQUFDLFFBQVE7UUFDMUIsUUFBUSxFQUFFLE9BQU8sQ0FBQyxRQUFRO1FBQzFCLE1BQU0sRUFBRSxPQUFPLENBQUMsTUFBTTtRQUN0QixnQkFBZ0IsRUFBRSxPQUFPLENBQUMsZ0JBQWdCO1FBQzFDLGNBQWMsRUFBRSxPQUFPLENBQUMsY0FBYztRQUN0QyxnQkFBZ0IsRUFBRSxPQUFPLENBQUMsZ0JBQWdCO1FBQzFDLGNBQWMsRUFBRSxPQUFPLENBQUMsY0FBYztRQUN0Qyx1QkFBdUIsRUFBRSxPQUFPLENBQUMsdUJBQXVCO1FBQ3hELHdCQUF3QixFQUFFLE9BQU8sQ0FBQyx3QkFBd0I7UUFDMUQsK0JBQStCLEVBQUUsT0FBTyxDQUFDLCtCQUErQjtRQUN4RSxnQ0FBZ0MsRUFBRSxPQUFPLENBQUMsZ0NBQWdDO1FBQzFFLE9BQU8sRUFBRSxPQUFPLENBQUMsT0FBTztRQUN4QixnQkFBZ0IsRUFBRSxPQUFPLENBQUMsZ0JBQWdCO1FBQzFDLGdCQUFnQixFQUFFLE9BQU8sQ0FBQyxnQkFBZ0I7UUFDMUMsY0FBYyxFQUFFLE9BQU8sQ0FBQyxjQUFjO1FBQ3RDLGVBQWUsRUFBRSxPQUFPLENBQUMsZUFBZTtRQUN4QyxnQkFBZ0IsRUFBRSxPQUFPLENBQUMsZ0JBQWdCO1FBQzFDLGdCQUFnQixFQUFFLE9BQU8sQ0FBQyxnQkFBZ0I7UUFDMUMsT0FBTyxFQUFFLE9BQU8sQ0FBQyxPQUFPO1FBQ3hCLE9BQU8sRUFBRSxPQUFPLENBQUMsT0FBTztRQUN4QixVQUFVLEVBQUUsT0FBTyxDQUFDLFVBQVU7UUFDOUIsWUFBWSxFQUFFLE9BQU8sQ0FBQyxZQUFZO1FBQ2xDLGlCQUFpQixFQUFFLE9BQU8sQ0FBQyxpQkFBaUI7UUFDNUMseUJBQXlCLEVBQUUsT0FBTyxDQUFDLHlCQUF5QjtLQUM3RCxDQUFDLENBQUM7SUFDSCxJQUFJLFVBQVUsR0FBRyxDQUFDLENBQUM7SUFDbkIsSUFBSSxVQUFVLEdBQUcsQ0FBQyxDQUFDO0lBQ25CLElBQUksV0FBVyxHQUFHLENBQUMsQ0FBQztJQUNwQixJQUFJLGNBQWMsR0FBRyxDQUFDLENBQUM7SUFDdkIsSUFBSSxlQUFlLEdBQUcsQ0FBQyxDQUFDO0lBQ3hCLElBQUksaUJBQWlCLEdBQUcsQ0FBQyxDQUFDO0lBQzFCLE1BQU0sU0FBUyxHQUEyQixFQUFFLENBQUM7SUFFN0MsSUFBSSxDQUFDO1FBQ0gsTUFBTSxPQUFPLENBQUMsZUFBZSxDQUFDLE9BQU8sQ0FBQyxNQUFNLENBQUMsQ0FBQztRQUM5QyxPQUFPLElBQUksRUFBRSxDQUFDO1lBQ1osTUFBTSxnQkFBZ0IsR0FBRyxNQUFNLE9BQU8sQ0FBQyxRQUFRLENBQUM7Z0JBQzlDLFFBQVEsRUFBRSxPQUFPLENBQUMsUUFBUTtnQkFDMUIsS0FBSyxFQUFFLE9BQU8sQ0FBQyxLQUFLO2FBQ3JCLENBQUMsQ0FBQztZQUNILFVBQVUsSUFBSSxnQkFBZ0IsQ0FBQyxLQUFLLENBQUM7WUFDckMsVUFBVSxJQUFJLGdCQUFnQixDQUFDLEtBQUssQ0FBQyxXQUFXLENBQUM7WUFDakQsV0FBVyxJQUFJLGdCQUFnQixDQUFDLEtBQUssQ0FBQyxZQUFZLENBQUM7WUFDbkQsY0FBYyxJQUFJLGdCQUFnQixDQUFDLEtBQUssQ0FBQyxlQUFlLENBQUM7WUFDekQsZUFBZSxJQUFJLGdCQUFnQixDQUFDLEtBQUssQ0FBQyxnQkFBZ0IsQ0FBQztZQUMzRCxjQUFjLENBQUMsU0FBUyxFQUFFLGdCQUFnQixDQUFDLFNBQVMsQ0FBQyxDQUFDO1lBRXRELE1BQU0sU0FBUyxHQUFvQjtnQkFDakMsSUFBSSxFQUFFLGdCQUFnQixDQUFDLElBQUk7Z0JBQzNCLFFBQVEsRUFBRSxPQUFPLENBQUMsZ0JBQWdCLEVBQUU7Z0JBQ3BDLEtBQUssRUFBRSxVQUFVO2dCQUNqQixZQUFZLEVBQUUsZ0JBQWdCLENBQUMsWUFBWTtnQkFDM0MsS0FBSyxFQUFFO29CQUNMLFdBQVcsRUFBRSxVQUFVO29CQUN2QixZQUFZLEVBQUUsV0FBVztvQkFDekIsV0FBVyxFQUFFLFVBQVUsR0FBRyxXQUFXO29CQUNyQyxlQUFlLEVBQUUsY0FBYztvQkFDL0IsZ0JBQWdCLEVBQUUsZUFBZTtpQkFDbEM7Z0JBQ0QsU0FBUzthQUNWLENBQUM7WUFFRixNQUFNLGdCQUFnQixHQUFHLE1BQU0sT0FBTyxDQUFDLGtCQUFrQixFQUFFLENBQUMsU0FBUyxDQUFDLENBQUM7WUFDdkUsSUFBSSxPQUFPLGdCQUFnQixLQUFLLFFBQVE7Z0JBQUUsT0FBTyxTQUFTLENBQUM7WUFDM0QsSUFBSSxpQkFBaUIsSUFBSSxDQUFDLE9BQU8sQ0FBQyxvQkFBb0IsSUFBSSxDQUFDLENBQUMsRUFBRSxDQUFDO2dCQUM3RCxNQUFNLElBQUksS0FBSyxDQUFDLHVDQUF1QyxnQkFBZ0IsRUFBRSxDQUFDLENBQUM7WUFDN0UsQ0FBQztZQUNELGlCQUFpQixFQUFFLENBQUM7WUFDcEIsTUFBTSxPQUFPLENBQUMsZUFBZSxDQUFDLGdCQUFnQixDQUFDLENBQUM7UUFDbEQsQ0FBQztJQUNILENBQUM7WUFBUyxDQUFDO1FBQ1QsTUFBTSxPQUFPLENBQUMsS0FBSyxFQUFFLENBQUM7SUFDeEIsQ0FBQztBQUNILENBQUMifQ==
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { ISmartAiCacheOptions, ToolSet, ModelMessage, TSmartAiLanguageModel, ProviderOptions, IToolExecutionContext, IToolJobState, TSmartAiCacheRetention, TSmartAiCacheSetting } from './plugins.js';
|
|
2
2
|
import type { IAgentRuntimeEventPayload, IToolExecutionIntentEvent, TAgentEvent, TAgentGenerationOutcome, TAgentToolResultOutput } from './smartagent.events.js';
|
|
3
3
|
import type { TAgentEventArchive, TAgentEventStore } from './smartagent.persistence.js';
|
|
4
|
+
import type { IAgentRetryEvent } from './smartagent.retry.js';
|
|
4
5
|
export type { ProviderOptions };
|
|
5
6
|
export interface IAgentCacheOptions extends ISmartAiCacheOptions {
|
|
6
7
|
}
|
|
@@ -32,8 +33,64 @@ export type TAgentToolCallFinishEvent = (IAgentToolCallStartEvent & {
|
|
|
32
33
|
success: false;
|
|
33
34
|
error: string;
|
|
34
35
|
});
|
|
36
|
+
/** Token usage of model calls. A count a provider leaves out counts as zero. */
|
|
37
|
+
export interface IAgentUsage {
|
|
38
|
+
inputTokens: number;
|
|
39
|
+
outputTokens: number;
|
|
40
|
+
totalTokens: number;
|
|
41
|
+
cacheReadTokens: number;
|
|
42
|
+
cacheWriteTokens: number;
|
|
43
|
+
}
|
|
44
|
+
interface IAgentModelCallIdentity {
|
|
45
|
+
/** The provider of the model that was called, as the language model names it. */
|
|
46
|
+
provider: string;
|
|
47
|
+
/**
|
|
48
|
+
* The model id the call was made with: the language model's `modelId`. Every call carries it,
|
|
49
|
+
* whatever its outcome, so key usage caps on `provider` and `requestedModelId`.
|
|
50
|
+
*/
|
|
51
|
+
requestedModelId: string;
|
|
52
|
+
}
|
|
53
|
+
/** Why a model call ended without the provider reporting its usage. */
|
|
54
|
+
export type TAgentModelCallUnreportedReason = 'aborted' | 'failed' | 'missing';
|
|
55
|
+
/**
|
|
56
|
+
* The usage outcome of one model call, as the component that made the call reports it.
|
|
57
|
+
* `reported`: the provider reported the call's usage. `responseModelId` is the model id the
|
|
58
|
+
* provider's response named, or `requestedModelId` when the response named none.
|
|
59
|
+
* `unreported`: the call ended before the provider reported its usage: `aborted` (the call was
|
|
60
|
+
* aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage).
|
|
61
|
+
* The provider may still have consumed tokens for it; their number is unknown.
|
|
62
|
+
*/
|
|
63
|
+
export type TAgentModelCallUsage = (IAgentModelCallIdentity & {
|
|
64
|
+
status: 'reported';
|
|
65
|
+
responseModelId: string;
|
|
66
|
+
usage: IAgentUsage;
|
|
67
|
+
}) | (IAgentModelCallIdentity & {
|
|
68
|
+
status: 'unreported';
|
|
69
|
+
reason: TAgentModelCallUnreportedReason;
|
|
70
|
+
});
|
|
71
|
+
/** Receives the usage outcome of each model call. Must not throw. */
|
|
72
|
+
export type TAgentModelCallUsageReporter = (call: TAgentModelCallUsage) => void;
|
|
73
|
+
/**
|
|
74
|
+
* The usage outcome of one model call of a session, reported once per call through `onUsage`.
|
|
75
|
+
* `source: 'generation'`: a model step of the generation `generationId`.
|
|
76
|
+
* `source: 'compaction'`: a model call of a context compaction. `generationId` names the generation
|
|
77
|
+
* whose context overflow caused it; it is absent for `compact()` and event retention.
|
|
78
|
+
*/
|
|
79
|
+
export type TAgentModelCallUsageEvent = TAgentModelCallUsage & ({
|
|
80
|
+
source: 'generation';
|
|
81
|
+
generationId: string;
|
|
82
|
+
} | {
|
|
83
|
+
source: 'compaction';
|
|
84
|
+
generationId?: string;
|
|
85
|
+
});
|
|
35
86
|
export interface IAgentContextOverflowInvocationOptions {
|
|
36
87
|
abortSignal?: AbortSignal;
|
|
88
|
+
/**
|
|
89
|
+
* Reports the usage of each model call the handler makes, into the session's `onUsage` as
|
|
90
|
+
* `source: 'compaction'`. The session always supplies it. Report every call before the returned
|
|
91
|
+
* promise settles, including calls that fail or are aborted.
|
|
92
|
+
*/
|
|
93
|
+
reportUsage?: TAgentModelCallUsageReporter;
|
|
37
94
|
}
|
|
38
95
|
export interface IAgentContextBuildOptions {
|
|
39
96
|
events: readonly TAgentEvent[];
|
|
@@ -42,6 +99,12 @@ export type TAgentContextBuilder = (options: IAgentContextBuildOptions) => Model
|
|
|
42
99
|
export interface IAgentContextCompactionOptions {
|
|
43
100
|
abortSignal?: AbortSignal;
|
|
44
101
|
reason: 'context-overflow' | 'retention' | 'manual';
|
|
102
|
+
/**
|
|
103
|
+
* Reports the usage of each model call the compactor makes, into the session's `onUsage` as
|
|
104
|
+
* `source: 'compaction'`. The session always supplies it. Report every call before the returned
|
|
105
|
+
* promise settles, including calls that fail or are aborted.
|
|
106
|
+
*/
|
|
107
|
+
reportUsage?: TAgentModelCallUsageReporter;
|
|
45
108
|
}
|
|
46
109
|
export type TAgentContextCompactor = (messages: ModelMessage[], events: readonly TAgentEvent[], options: IAgentContextCompactionOptions) => Promise<ModelMessage[]>;
|
|
47
110
|
export interface IAgentEventRetentionOptions {
|
|
@@ -113,6 +176,18 @@ export interface IAgentSessionOptions {
|
|
|
113
176
|
onToolCallUpdate?: (event: IAgentToolCallUpdateEvent) => void;
|
|
114
177
|
/** Called when a tool call finishes, with its stable AI SDK call id and success state. */
|
|
115
178
|
onToolCallFinish?: (event: TAgentToolCallFinishEvent) => void;
|
|
179
|
+
/** Called before the engine waits to retry a rate-limited or unavailable model call. */
|
|
180
|
+
onRetry?: (event: IAgentRetryEvent) => void;
|
|
181
|
+
/**
|
|
182
|
+
* Called once for every model call of the session, as soon as its usage is known: when the
|
|
183
|
+
* provider reports it, or when the call ends without it. This covers the generation's model
|
|
184
|
+
* steps, the model call of `compactMessages()` when a compactor passes it `reportUsage`, and every
|
|
185
|
+
* call a `contextCompactor` or `onContextOverflow` handler reports through `reportUsage`. The
|
|
186
|
+
* reported calls of a generation, including those of a compaction its context overflow caused,
|
|
187
|
+
* sum to the generation's `usage` when it returns. Must not throw; an error it throws is reported
|
|
188
|
+
* like a session listener error and does not change the generation's outcome.
|
|
189
|
+
*/
|
|
190
|
+
onUsage?: (event: TAgentModelCallUsageEvent) => void;
|
|
116
191
|
/** @deprecated Use onToolCallStart instead. */
|
|
117
192
|
onToolCall?: (toolName: string, input: unknown) => void;
|
|
118
193
|
/** @deprecated Use onToolCallFinish instead. */
|
|
@@ -226,14 +301,11 @@ export interface IAgentRunResult {
|
|
|
226
301
|
steps: number;
|
|
227
302
|
/** Finish reason from the final step */
|
|
228
303
|
finishReason: string;
|
|
229
|
-
/**
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
cacheReadTokens: number;
|
|
235
|
-
cacheWriteTokens: number;
|
|
236
|
-
};
|
|
304
|
+
/**
|
|
305
|
+
* Token usage the provider reported for every model call of the run: its model steps, retried
|
|
306
|
+
* calls included, and the calls of a compaction its context overflow caused that were reported
|
|
307
|
+
*/
|
|
308
|
+
usage: IAgentUsage;
|
|
237
309
|
/** Tool calls observed during the run, including inputs and outputs/errors when available */
|
|
238
310
|
toolCalls: IAgentToolCallRecord[];
|
|
239
311
|
}
|
|
@@ -21,4 +21,4 @@ export class ContextOverflowError extends Error {
|
|
|
21
21
|
this.name = 'ContextOverflowError';
|
|
22
22
|
}
|
|
23
23
|
}
|
|
24
|
-
//# sourceMappingURL=data:application/json;base64,
|
|
24
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoic21hcnRhZ2VudC5pbnRlcmZhY2VzLmpzIiwic291cmNlUm9vdCI6IiIsInNvdXJjZXMiOlsiLi4vdHNfYWdlbnQvc21hcnRhZ2VudC5pbnRlcmZhY2VzLnRzIl0sIm5hbWVzIjpbXSwibWFwcGluZ3MiOiJBQXNTQSxNQUFNLE9BQU8sZ0NBQWlDLFNBQVEsS0FBSztJQUN6QyxLQUFLLENBQXdCO0lBQzdCLFlBQVksQ0FBVTtJQUNyQixvQkFBb0IsQ0FBc0I7SUFFM0QsWUFDRSxLQUE0QixFQUM1QixZQUFxQixFQUNyQixvQkFBeUM7UUFFekMsS0FBSyxDQUNILG9DQUFvQyxZQUFZLFlBQVksS0FBSztZQUMvRCxDQUFDLENBQUMsWUFBWSxDQUFDLE9BQU87WUFDdEIsQ0FBQyxDQUFDLE1BQU0sQ0FBQyxZQUFZLENBQUMsRUFBRSxDQUMzQixDQUFDO1FBQ0YsSUFBSSxDQUFDLElBQUksR0FBRyxrQ0FBa0MsQ0FBQztRQUMvQyxJQUFJLENBQUMsS0FBSyxHQUFHLEtBQUssQ0FBQztRQUNuQixJQUFJLENBQUMsWUFBWSxHQUFHLFlBQVksQ0FBQztRQUNqQyxJQUFJLENBQUMsb0JBQW9CLEdBQUcsb0JBQW9CLENBQUM7SUFDbkQsQ0FBQztJQUVNLFlBQVk7UUFDakIsT0FBTyxJQUFJLENBQUMsb0JBQW9CLEVBQUUsQ0FBQztJQUNyQyxDQUFDO0NBQ0Y7QUFzSEQsTUFBTSxPQUFPLG9CQUFxQixTQUFRLEtBQUs7SUFDN0MsWUFBWSxPQUFPLEdBQUcsdUVBQXVFO1FBQzNGLEtBQUssQ0FBQyxPQUFPLENBQUMsQ0FBQztRQUNmLElBQUksQ0FBQyxJQUFJLEdBQUcsc0JBQXNCLENBQUM7SUFDckMsQ0FBQztDQUNGIn0=
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
export type TAgentRetryReason = 'rate_limit' | 'overloaded' | 'unavailable';
|
|
2
|
+
/** A failed model call the session engine retries after `delayMs`. */
|
|
3
|
+
export interface IAgentRetryEvent {
|
|
4
|
+
/** One-based number of this retry within the current model call. */
|
|
5
|
+
attempt: number;
|
|
6
|
+
maxAttempts: number;
|
|
7
|
+
delayMs: number;
|
|
8
|
+
reason: TAgentRetryReason;
|
|
9
|
+
}
|
|
10
|
+
export type TAgentRetryDecision = {
|
|
11
|
+
action: 'retry';
|
|
12
|
+
delayMs: number;
|
|
13
|
+
reason: TAgentRetryReason;
|
|
14
|
+
} | {
|
|
15
|
+
action: 'fail';
|
|
16
|
+
error: unknown;
|
|
17
|
+
} | {
|
|
18
|
+
action: 'not-retryable';
|
|
19
|
+
};
|
|
20
|
+
export interface IAgentRetryState {
|
|
21
|
+
/** Retries already made for the current model call. */
|
|
22
|
+
attempt: number;
|
|
23
|
+
/** Retry delay already spent on the current model call. */
|
|
24
|
+
retriedMs: number;
|
|
25
|
+
/** Provider id used when an untyped rate limit has to fail with a typed limit error. */
|
|
26
|
+
provider: string;
|
|
27
|
+
now: number;
|
|
28
|
+
}
|
|
29
|
+
export declare const MAX_RETRY_ATTEMPTS = 8;
|
|
30
|
+
/** The longest provider-requested delay honoured; the AI SDK uses the same bound. */
|
|
31
|
+
export declare const MAX_RETRY_AFTER_MS = 60000;
|
|
32
|
+
/** The total delay one model call may spend retrying, equal to the full default backoff. */
|
|
33
|
+
export declare const MAX_RETRY_WINDOW_MS = 150000;
|
|
34
|
+
/**
|
|
35
|
+
* Decides whether a failed model call is retried. Usage limits never are; transient limits wait for
|
|
36
|
+
* the provider's delay up to MAX_RETRY_AFTER_MS, but never less than the backoff, and all retries of
|
|
37
|
+
* one call stay within MAX_RETRY_ATTEMPTS and MAX_RETRY_WINDOW_MS.
|
|
38
|
+
*/
|
|
39
|
+
export declare function planModelCallRetry(error: unknown, state: IAgentRetryState): TAgentRetryDecision;
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import * as plugins from './plugins.js';
|
|
2
|
+
export const MAX_RETRY_ATTEMPTS = 8;
|
|
3
|
+
/** The longest provider-requested delay honoured; the AI SDK uses the same bound. */
|
|
4
|
+
export const MAX_RETRY_AFTER_MS = 60_000;
|
|
5
|
+
/** The total delay one model call may spend retrying, equal to the full default backoff. */
|
|
6
|
+
export const MAX_RETRY_WINDOW_MS = 150_000;
|
|
7
|
+
const RETRY_INITIAL_DELAY_MS = 2000;
|
|
8
|
+
const RETRY_BACKOFF_FACTOR = 2;
|
|
9
|
+
const RETRY_BACKOFF_MAX_DELAY_MS = 30_000;
|
|
10
|
+
function backoffDelayMs(attempt) {
|
|
11
|
+
return Math.min(RETRY_INITIAL_DELAY_MS * RETRY_BACKOFF_FACTOR ** (attempt - 1), RETRY_BACKOFF_MAX_DELAY_MS);
|
|
12
|
+
}
|
|
13
|
+
function untypedRetryReason(error) {
|
|
14
|
+
const candidate = error;
|
|
15
|
+
const status = candidate?.status ?? candidate?.statusCode;
|
|
16
|
+
const message = error instanceof Error ? error.message.toLowerCase() : '';
|
|
17
|
+
if (status === 429 || message.includes('rate limit') || message.includes('too many requests')) {
|
|
18
|
+
return 'rate_limit';
|
|
19
|
+
}
|
|
20
|
+
if (status === 529 || message.includes('overloaded'))
|
|
21
|
+
return 'overloaded';
|
|
22
|
+
if (status === 503)
|
|
23
|
+
return 'unavailable';
|
|
24
|
+
return undefined;
|
|
25
|
+
}
|
|
26
|
+
function readRetrySignal(error, now) {
|
|
27
|
+
if (plugins.isModelLimitError(error)) {
|
|
28
|
+
const { retryAfterMs, resetsAt } = error.limit;
|
|
29
|
+
const requestedDelayMs = retryAfterMs ?? (resetsAt === undefined ? undefined : Math.max(0, resetsAt - now));
|
|
30
|
+
return { reason: 'rate_limit', ...(requestedDelayMs === undefined ? {} : { requestedDelayMs }) };
|
|
31
|
+
}
|
|
32
|
+
const reason = untypedRetryReason(error);
|
|
33
|
+
if (reason === undefined)
|
|
34
|
+
return undefined;
|
|
35
|
+
const failure = error;
|
|
36
|
+
const requestedDelayMs = plugins.readRetryAfterMs(failure.responseHeaders ?? failure.headers, now);
|
|
37
|
+
return { reason, ...(requestedDelayMs === undefined ? {} : { requestedDelayMs }) };
|
|
38
|
+
}
|
|
39
|
+
/** The error a rate limit fails with when its retry time lies beyond the retry bounds. */
|
|
40
|
+
function overdueLimitError(error, signal, state) {
|
|
41
|
+
if (plugins.isModelLimitError(error) || signal.reason !== 'rate_limit' || signal.requestedDelayMs === undefined) {
|
|
42
|
+
return error;
|
|
43
|
+
}
|
|
44
|
+
const limit = plugins.createModelLimitInfo({ kind: 'rate_limit', provider: state.provider, retryAfterMs: signal.requestedDelayMs }, state.now);
|
|
45
|
+
// an empty or overlong provider id cannot describe a limit, so the provider's own error stays
|
|
46
|
+
return plugins.isModelLimitInfo(limit) ? new plugins.ModelLimitError(limit, { cause: error }) : error;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Decides whether a failed model call is retried. Usage limits never are; transient limits wait for
|
|
50
|
+
* the provider's delay up to MAX_RETRY_AFTER_MS, but never less than the backoff, and all retries of
|
|
51
|
+
* one call stay within MAX_RETRY_ATTEMPTS and MAX_RETRY_WINDOW_MS.
|
|
52
|
+
*/
|
|
53
|
+
export function planModelCallRetry(error, state) {
|
|
54
|
+
if (plugins.isModelLimitError(error) && error.limit.kind === 'usage_limit')
|
|
55
|
+
return { action: 'fail', error };
|
|
56
|
+
const signal = readRetrySignal(error, state.now);
|
|
57
|
+
if (signal === undefined)
|
|
58
|
+
return { action: 'not-retryable' };
|
|
59
|
+
if (state.attempt >= MAX_RETRY_ATTEMPTS)
|
|
60
|
+
return { action: 'fail', error };
|
|
61
|
+
const delayMs = Math.max(backoffDelayMs(state.attempt + 1), signal.requestedDelayMs ?? 0);
|
|
62
|
+
if (delayMs > MAX_RETRY_AFTER_MS || state.retriedMs + delayMs > MAX_RETRY_WINDOW_MS) {
|
|
63
|
+
return { action: 'fail', error: overdueLimitError(error, signal, state) };
|
|
64
|
+
}
|
|
65
|
+
return { action: 'retry', delayMs, reason: signal.reason };
|
|
66
|
+
}
|
|
67
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoic21hcnRhZ2VudC5yZXRyeS5qcyIsInNvdXJjZVJvb3QiOiIiLCJzb3VyY2VzIjpbIi4uL3RzX2FnZW50L3NtYXJ0YWdlbnQucmV0cnkudHMiXSwibmFtZXMiOltdLCJtYXBwaW5ncyI6IkFBQUEsT0FBTyxLQUFLLE9BQU8sTUFBTSxjQUFjLENBQUM7QUE0QnhDLE1BQU0sQ0FBQyxNQUFNLGtCQUFrQixHQUFHLENBQUMsQ0FBQztBQUNwQyxxRkFBcUY7QUFDckYsTUFBTSxDQUFDLE1BQU0sa0JBQWtCLEdBQUcsTUFBTSxDQUFDO0FBQ3pDLDRGQUE0RjtBQUM1RixNQUFNLENBQUMsTUFBTSxtQkFBbUIsR0FBRyxPQUFPLENBQUM7QUFFM0MsTUFBTSxzQkFBc0IsR0FBRyxJQUFJLENBQUM7QUFDcEMsTUFBTSxvQkFBb0IsR0FBRyxDQUFDLENBQUM7QUFDL0IsTUFBTSwwQkFBMEIsR0FBRyxNQUFNLENBQUM7QUFPMUMsU0FBUyxjQUFjLENBQUMsT0FBZTtJQUNyQyxPQUFPLElBQUksQ0FBQyxHQUFHLENBQUMsc0JBQXNCLEdBQUcsb0JBQW9CLElBQUksQ0FBQyxPQUFPLEdBQUcsQ0FBQyxDQUFDLEVBQUUsMEJBQTBCLENBQUMsQ0FBQztBQUM5RyxDQUFDO0FBRUQsU0FBUyxrQkFBa0IsQ0FBQyxLQUFjO0lBQ3hDLE1BQU0sU0FBUyxHQUFHLEtBQTZELENBQUM7SUFDaEYsTUFBTSxNQUFNLEdBQUcsU0FBUyxFQUFFLE1BQU0sSUFBSSxTQUFTLEVBQUUsVUFBVSxDQUFDO0lBQzFELE1BQU0sT0FBTyxHQUFHLEtBQUssWUFBWSxLQUFLLENBQUMsQ0FBQyxDQUFDLEtBQUssQ0FBQyxPQUFPLENBQUMsV0FBVyxFQUFFLENBQUMsQ0FBQyxDQUFDLEVBQUUsQ0FBQztJQUMxRSxJQUFJLE1BQU0sS0FBSyxHQUFHLElBQUksT0FBTyxDQUFDLFFBQVEsQ0FBQyxZQUFZLENBQUMsSUFBSSxPQUFPLENBQUMsUUFBUSxDQUFDLG1CQUFtQixDQUFDLEVBQUUsQ0FBQztRQUM5RixPQUFPLFlBQVksQ0FBQztJQUN0QixDQUFDO0lBQ0QsSUFBSSxNQUFNLEtBQUssR0FBRyxJQUFJLE9BQU8sQ0FBQyxRQUFRLENBQUMsWUFBWSxDQUFDO1FBQUUsT0FBTyxZQUFZLENBQUM7SUFDMUUsSUFBSSxNQUFNLEtBQUssR0FBRztRQUFFLE9BQU8sYUFBYSxDQUFDO0lBQ3pDLE9BQU8sU0FBUyxDQUFDO0FBQ25CLENBQUM7QUFFRCxTQUFTLGVBQWUsQ0FBQyxLQUFjLEVBQUUsR0FBVztJQUNsRCxJQUFJLE9BQU8sQ0FBQyxpQkFBaUIsQ0FBQyxLQUFLLENBQUMsRUFBRSxDQUFDO1FBQ3JDLE1BQU0sRUFBRSxZQUFZLEVBQUUsUUFBUSxFQUFFLEdBQUcsS0FBSyxDQUFDLEtBQUssQ0FBQztRQUMvQyxNQUFNLGdCQUFnQixHQUFHLFlBQVksSUFBSSxDQUFDLFFBQVEsS0FBSyxTQUFTLENBQUMsQ0FBQyxDQUFDLFNBQVMsQ0FBQyxDQUFDLENBQUMsSUFBSSxDQUFDLEdBQUcsQ0FBQyxDQUFDLEVBQUUsUUFBUSxHQUFHLEdBQUcsQ0FBQyxDQUFDLENBQUM7UUFDNUcsT0FBTyxFQUFFLE1BQU0sRUFBRSxZQUFZLEVBQUUsR0FBRyxDQUFDLGdCQUFnQixLQUFLLFNBQVMsQ0FBQyxDQUFDLENBQUMsRUFBRSxDQUFDLENBQUMsQ0FBQyxFQUFFLGdCQUFnQixFQUFFLENBQUMsRUFBRSxDQUFDO0lBQ25HLENBQUM7SUFDRCxNQUFNLE1BQU0sR0FBRyxrQkFBa0IsQ0FBQyxLQUFLLENBQUMsQ0FBQztJQUN6QyxJQUFJLE1BQU0sS0FBSyxTQUFTO1FBQUUsT0FBTyxTQUFTLENBQUM7SUFDM0MsTUFBTSxPQUFPLEdBQUcsS0FBdUYsQ0FBQztJQUN4RyxNQUFNLGdCQUFnQixHQUFHLE9BQU8sQ0FBQyxnQkFBZ0IsQ0FBQyxPQUFPLENBQUMsZUFBZSxJQUFJLE9BQU8sQ0FBQyxPQUFPLEVBQUUsR0FBRyxDQUFDLENBQUM7SUFDbkcsT0FBTyxFQUFFLE1BQU0sRUFBRSxHQUFHLENBQUMsZ0JBQWdCLEtBQUssU0FBUyxDQUFDLENBQUMsQ0FBQyxFQUFFLENBQUMsQ0FBQyxDQUFDLEVBQUUsZ0JBQWdCLEVBQUUsQ0FBQyxFQUFFLENBQUM7QUFDckYsQ0FBQztBQUVELDBGQUEwRjtBQUMxRixTQUFTLGlCQUFpQixDQUFDLEtBQWMsRUFBRSxNQUFvQixFQUFFLEtBQXVCO0lBQ3RGLElBQUksT0FBTyxDQUFDLGlCQUFpQixDQUFDLEtBQUssQ0FBQyxJQUFJLE1BQU0sQ0FBQyxNQUFNLEtBQUssWUFBWSxJQUFJLE1BQU0sQ0FBQyxnQkFBZ0IsS0FBSyxTQUFTLEVBQUUsQ0FBQztRQUNoSCxPQUFPLEtBQUssQ0FBQztJQUNmLENBQUM7SUFDRCxNQUFNLEtBQUssR0FBRyxPQUFPLENBQUMsb0JBQW9CLENBQ3hDLEVBQUUsSUFBSSxFQUFFLFlBQVksRUFBRSxRQUFRLEVBQUUsS0FBSyxDQUFDLFFBQVEsRUFBRSxZQUFZLEVBQUUsTUFBTSxDQUFDLGdCQUFnQixFQUFFLEVBQ3ZGLEtBQUssQ0FBQyxHQUFHLENBQ1YsQ0FBQztJQUNGLDhGQUE4RjtJQUM5RixPQUFPLE9BQU8sQ0FBQyxnQkFBZ0IsQ0FBQyxLQUFLLENBQUMsQ0FBQyxDQUFDLENBQUMsSUFBSSxPQUFPLENBQUMsZUFBZSxDQUFDLEtBQUssRUFBRSxFQUFFLEtBQUssRUFBRSxLQUFLLEVBQUUsQ0FBQyxDQUFDLENBQUMsQ0FBQyxLQUFLLENBQUM7QUFDeEcsQ0FBQztBQUVEOzs7O0dBSUc7QUFDSCxNQUFNLFVBQVUsa0JBQWtCLENBQUMsS0FBYyxFQUFFLEtBQXVCO0lBQ3hFLElBQUksT0FBTyxDQUFDLGlCQUFpQixDQUFDLEtBQUssQ0FBQyxJQUFJLEtBQUssQ0FBQyxLQUFLLENBQUMsSUFBSSxLQUFLLGFBQWE7UUFBRSxPQUFPLEVBQUUsTUFBTSxFQUFFLE1BQU0sRUFBRSxLQUFLLEVBQUUsQ0FBQztJQUM3RyxNQUFNLE1BQU0sR0FBRyxlQUFlLENBQUMsS0FBSyxFQUFFLEtBQUssQ0FBQyxHQUFHLENBQUMsQ0FBQztJQUNqRCxJQUFJLE1BQU0sS0FBSyxTQUFTO1FBQUUsT0FBTyxFQUFFLE1BQU0sRUFBRSxlQUFlLEVBQUUsQ0FBQztJQUM3RCxJQUFJLEtBQUssQ0FBQyxPQUFPLElBQUksa0JBQWtCO1FBQUUsT0FBTyxFQUFFLE1BQU0sRUFBRSxNQUFNLEVBQUUsS0FBSyxFQUFFLENBQUM7SUFDMUUsTUFBTSxPQUFPLEdBQUcsSUFBSSxDQUFDLEdBQUcsQ0FBQyxjQUFjLENBQUMsS0FBSyxDQUFDLE9BQU8sR0FBRyxDQUFDLENBQUMsRUFBRSxNQUFNLENBQUMsZ0JBQWdCLElBQUksQ0FBQyxDQUFDLENBQUM7SUFDMUYsSUFBSSxPQUFPLEdBQUcsa0JBQWtCLElBQUksS0FBSyxDQUFDLFNBQVMsR0FBRyxPQUFPLEdBQUcsbUJBQW1CLEVBQUUsQ0FBQztRQUNwRixPQUFPLEVBQUUsTUFBTSxFQUFFLE1BQU0sRUFBRSxLQUFLLEVBQUUsaUJBQWlCLENBQUMsS0FBSyxFQUFFLE1BQU0sRUFBRSxLQUFLLENBQUMsRUFBRSxDQUFDO0lBQzVFLENBQUM7SUFDRCxPQUFPLEVBQUUsTUFBTSxFQUFFLE9BQU8sRUFBRSxPQUFPLEVBQUUsTUFBTSxFQUFFLE1BQU0sQ0FBQyxNQUFNLEVBQUUsQ0FBQztBQUM3RCxDQUFDIn0=
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { LanguageModelUsage } from './plugins.js';
|
|
2
|
+
import type { TAgentModelCallUnreportedReason, TAgentModelCallUsageReporter } from './smartagent.interfaces.js';
|
|
3
|
+
/** The identity of the language model whose calls a recorder follows. */
|
|
4
|
+
export interface IAgentModelCallUsageRecorderModel {
|
|
5
|
+
readonly provider: string;
|
|
6
|
+
readonly modelId: string;
|
|
7
|
+
}
|
|
8
|
+
/**
|
|
9
|
+
* Follows the calls of one language model and reports each call's usage exactly once.
|
|
10
|
+
* Call `start()` when a model call begins, `end()` when its provider reports usage, and `settle()`
|
|
11
|
+
* when the call ends otherwise. A call that is still open when the next one starts ended without
|
|
12
|
+
* usage and is reported as `unreported` with reason `missing`.
|
|
13
|
+
*/
|
|
14
|
+
export declare class AgentModelCallUsageRecorder {
|
|
15
|
+
private readonly model;
|
|
16
|
+
private readonly report;
|
|
17
|
+
private open;
|
|
18
|
+
constructor(model: IAgentModelCallUsageRecorderModel, report: TAgentModelCallUsageReporter);
|
|
19
|
+
/** A model call begins. */
|
|
20
|
+
start(): void;
|
|
21
|
+
/** The provider reported the usage of the current call. A count it leaves out counts as zero. */
|
|
22
|
+
end(responseModelId: string, providerUsage: LanguageModelUsage): void;
|
|
23
|
+
/** The current call ended without usage. Does nothing when no call is open. */
|
|
24
|
+
settle(reason: TAgentModelCallUnreportedReason): void;
|
|
25
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Follows the calls of one language model and reports each call's usage exactly once.
|
|
3
|
+
* Call `start()` when a model call begins, `end()` when its provider reports usage, and `settle()`
|
|
4
|
+
* when the call ends otherwise. A call that is still open when the next one starts ended without
|
|
5
|
+
* usage and is reported as `unreported` with reason `missing`.
|
|
6
|
+
*/
|
|
7
|
+
export class AgentModelCallUsageRecorder {
|
|
8
|
+
model;
|
|
9
|
+
report;
|
|
10
|
+
open = false;
|
|
11
|
+
constructor(model, report) {
|
|
12
|
+
this.model = model;
|
|
13
|
+
this.report = report;
|
|
14
|
+
}
|
|
15
|
+
/** A model call begins. */
|
|
16
|
+
start() {
|
|
17
|
+
this.settle('missing');
|
|
18
|
+
this.open = true;
|
|
19
|
+
}
|
|
20
|
+
/** The provider reported the usage of the current call. A count it leaves out counts as zero. */
|
|
21
|
+
end(responseModelId, providerUsage) {
|
|
22
|
+
this.open = false;
|
|
23
|
+
const inputTokens = providerUsage.inputTokens ?? 0;
|
|
24
|
+
const outputTokens = providerUsage.outputTokens ?? 0;
|
|
25
|
+
const usage = {
|
|
26
|
+
inputTokens,
|
|
27
|
+
outputTokens,
|
|
28
|
+
totalTokens: inputTokens + outputTokens,
|
|
29
|
+
cacheReadTokens: providerUsage.inputTokenDetails.cacheReadTokens ?? 0,
|
|
30
|
+
cacheWriteTokens: providerUsage.inputTokenDetails.cacheWriteTokens ?? 0,
|
|
31
|
+
};
|
|
32
|
+
this.report({
|
|
33
|
+
status: 'reported',
|
|
34
|
+
provider: this.model.provider,
|
|
35
|
+
requestedModelId: this.model.modelId,
|
|
36
|
+
responseModelId,
|
|
37
|
+
usage,
|
|
38
|
+
});
|
|
39
|
+
}
|
|
40
|
+
/** The current call ended without usage. Does nothing when no call is open. */
|
|
41
|
+
settle(reason) {
|
|
42
|
+
if (!this.open)
|
|
43
|
+
return;
|
|
44
|
+
this.open = false;
|
|
45
|
+
this.report({
|
|
46
|
+
status: 'unreported',
|
|
47
|
+
provider: this.model.provider,
|
|
48
|
+
requestedModelId: this.model.modelId,
|
|
49
|
+
reason,
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
//# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoic21hcnRhZ2VudC51c2FnZS5qcyIsInNvdXJjZVJvb3QiOiIiLCJzb3VyY2VzIjpbIi4uL3RzX2FnZW50L3NtYXJ0YWdlbnQudXNhZ2UudHMiXSwibmFtZXMiOltdLCJtYXBwaW5ncyI6IkFBYUE7Ozs7O0dBS0c7QUFDSCxNQUFNLE9BQU8sMkJBQTJCO0lBSW5CO0lBQ0E7SUFKWCxJQUFJLEdBQUcsS0FBSyxDQUFDO0lBRXJCLFlBQ21CLEtBQXdDLEVBQ3hDLE1BQW9DO1FBRHBDLFVBQUssR0FBTCxLQUFLLENBQW1DO1FBQ3hDLFdBQU0sR0FBTixNQUFNLENBQThCO0lBQ3BELENBQUM7SUFFSiwyQkFBMkI7SUFDcEIsS0FBSztRQUNWLElBQUksQ0FBQyxNQUFNLENBQUMsU0FBUyxDQUFDLENBQUM7UUFDdkIsSUFBSSxDQUFDLElBQUksR0FBRyxJQUFJLENBQUM7SUFDbkIsQ0FBQztJQUVELGlHQUFpRztJQUMxRixHQUFHLENBQUMsZUFBdUIsRUFBRSxhQUFpQztRQUNuRSxJQUFJLENBQUMsSUFBSSxHQUFHLEtBQUssQ0FBQztRQUNsQixNQUFNLFdBQVcsR0FBRyxhQUFhLENBQUMsV0FBVyxJQUFJLENBQUMsQ0FBQztRQUNuRCxNQUFNLFlBQVksR0FBRyxhQUFhLENBQUMsWUFBWSxJQUFJLENBQUMsQ0FBQztRQUNyRCxNQUFNLEtBQUssR0FBZ0I7WUFDekIsV0FBVztZQUNYLFlBQVk7WUFDWixXQUFXLEVBQUUsV0FBVyxHQUFHLFlBQVk7WUFDdkMsZUFBZSxFQUFFLGFBQWEsQ0FBQyxpQkFBaUIsQ0FBQyxlQUFlLElBQUksQ0FBQztZQUNyRSxnQkFBZ0IsRUFBRSxhQUFhLENBQUMsaUJBQWlCLENBQUMsZ0JBQWdCLElBQUksQ0FBQztTQUN4RSxDQUFDO1FBQ0YsSUFBSSxDQUFDLE1BQU0sQ0FBQztZQUNWLE1BQU0sRUFBRSxVQUFVO1lBQ2xCLFFBQVEsRUFBRSxJQUFJLENBQUMsS0FBSyxDQUFDLFFBQVE7WUFDN0IsZ0JBQWdCLEVBQUUsSUFBSSxDQUFDLEtBQUssQ0FBQyxPQUFPO1lBQ3BDLGVBQWU7WUFDZixLQUFLO1NBQ04sQ0FBQyxDQUFDO0lBQ0wsQ0FBQztJQUVELCtFQUErRTtJQUN4RSxNQUFNLENBQUMsTUFBdUM7UUFDbkQsSUFBSSxDQUFDLElBQUksQ0FBQyxJQUFJO1lBQUUsT0FBTztRQUN2QixJQUFJLENBQUMsSUFBSSxHQUFHLEtBQUssQ0FBQztRQUNsQixJQUFJLENBQUMsTUFBTSxDQUFDO1lBQ1YsTUFBTSxFQUFFLFlBQVk7WUFDcEIsUUFBUSxFQUFFLElBQUksQ0FBQyxLQUFLLENBQUMsUUFBUTtZQUM3QixnQkFBZ0IsRUFBRSxJQUFJLENBQUMsS0FBSyxDQUFDLE9BQU87WUFDcEMsTUFBTTtTQUNQLENBQUMsQ0FBQztJQUNMLENBQUM7Q0FDRiJ9
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"author": "Task Venture Capital GmbH",
|
|
3
3
|
"license": "MIT",
|
|
4
4
|
"name": "@modelprofile.com/flexharness-agent",
|
|
5
|
-
"version": "8.
|
|
5
|
+
"version": "8.4.0",
|
|
6
6
|
"type": "module",
|
|
7
7
|
"description": "Standalone event-driven agent runtime with generation transactions, context and tool execution contracts.",
|
|
8
8
|
"main": "./dist_ts_agent/index.js",
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
}
|
|
19
19
|
},
|
|
20
20
|
"dependencies": {
|
|
21
|
-
"@modelprofile.com/flexharness-models": "8.
|
|
21
|
+
"@modelprofile.com/flexharness-models": "8.4.0",
|
|
22
22
|
"ai": "^7.0.79",
|
|
23
23
|
"zod": "^4.4.1",
|
|
24
24
|
"@noble/hashes": "^2.4.0",
|
package/readme.md
CHANGED
|
@@ -88,7 +88,7 @@ The following Node examples reuse `setup` and, where needed, `tools` from the qu
|
|
|
88
88
|
| `messages` | Current AI SDK message history after projection or compaction. Pass it into another run to continue. |
|
|
89
89
|
| `steps` | Completed model steps, including steps from validation-triggered attempts. A step can call several tools. |
|
|
90
90
|
| `finishReason` | The model's final finish reason; inspect this together with application validation. |
|
|
91
|
-
| `usage` | Input, output and total tokens, plus cache-read and cache-write tokens. |
|
|
91
|
+
| `usage` | Input, output and total tokens, plus cache-read and cache-write tokens, summed over every model call of the run the provider reported: its model steps, retried calls included, and the reported calls of a compaction its context overflow caused. |
|
|
92
92
|
| `toolCalls` | Tool-call IDs, names and inputs, with available outputs or errors. |
|
|
93
93
|
|
|
94
94
|
```typescript
|
|
@@ -153,9 +153,69 @@ Pass these callbacks to `runAgent` or `AgentSession.create()`:
|
|
|
153
153
|
| `onToolCallStart(event)` | `toolCallId`, `toolName`, `input`. |
|
|
154
154
|
| `onToolCallUpdate(event)` | The call identity and a transient streamed `output`. |
|
|
155
155
|
| `onToolCallFinish(event)` | The call identity plus either `success: true, output` or `success: false, error`. |
|
|
156
|
+
| `onRetry(event)` | Before each wait to retry a model call: `attempt`, `maxAttempts`, `delayMs` and `reason` (`rate_limit`, `overloaded`, `unavailable`). |
|
|
157
|
+
| `onUsage(event)` | Once per model call, as soon as its usage is known; see [Count the usage of every run](#count-the-usage-of-every-run). |
|
|
156
158
|
|
|
157
159
|
Tool updates are transient; the finish callback carries the authoritative final output. Use `subscribe()` for committed session changes (`committed`, `updated`, `archived`), and retain its returned unsubscribe function. Session listeners are delivered in order per listener, have bounded queues and timeouts, and are removed on failure. Streaming callbacks and session-change listeners serve different purposes.
|
|
158
160
|
|
|
161
|
+
## Count the usage of every run
|
|
162
|
+
|
|
163
|
+
A run that throws or is aborted has used tokens too, and its promise carries no result. `onUsage` reports each model call once, as soon as its usage is known, whatever the run's outcome. Sum it to count a run's usage:
|
|
164
|
+
|
|
165
|
+
```typescript
|
|
166
|
+
import type { IAgentUsage } from '@modelprofile.com/flexharness-agent';
|
|
167
|
+
|
|
168
|
+
const used: IAgentUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 };
|
|
169
|
+
let usageComplete = true;
|
|
170
|
+
try {
|
|
171
|
+
await runAgent({
|
|
172
|
+
...setup,
|
|
173
|
+
tools,
|
|
174
|
+
prompt: 'Convert 10 km to miles.',
|
|
175
|
+
abort: AbortSignal.timeout(60_000),
|
|
176
|
+
onUsage: (event) => {
|
|
177
|
+
if (event.status === 'unreported') {
|
|
178
|
+
usageComplete = false;
|
|
179
|
+
return;
|
|
180
|
+
}
|
|
181
|
+
for (const key of Object.keys(used) as (keyof IAgentUsage)[]) used[key] += event.usage[key];
|
|
182
|
+
},
|
|
183
|
+
});
|
|
184
|
+
} finally {
|
|
185
|
+
console.log(used, usageComplete ? 'complete' : 'lower bound');
|
|
186
|
+
}
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
| Event field | Meaning |
|
|
190
|
+
| --- | --- |
|
|
191
|
+
| `source` | `generation`: a model step of the generation. `compaction`: a model call of a context compaction. |
|
|
192
|
+
| `generationId` | The generation the call belongs to. A compaction carries the generation whose context overflow caused it; a compaction by `compact()` or event retention has none. |
|
|
193
|
+
| `provider`, `requestedModelId` | The model the call was made with: the language model's `provider` and `modelId`. Every event carries both, whatever its status, so key usage caps on them. |
|
|
194
|
+
| `status: 'reported'`, `usage`, `responseModelId` | The provider reported the call's usage. A count the provider leaves out is zero. `responseModelId` is the model id the provider's response named (a provider may answer with a dated model version), or `requestedModelId` when it named none. |
|
|
195
|
+
| `status: 'unreported'`, `reason` | The call ended before the provider reported its usage: `aborted` (the call was aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage). The provider may still have consumed tokens for it; their number is unknown. |
|
|
196
|
+
|
|
197
|
+
When a run returns, its reported calls sum to `result.usage`; count one or the other, not both. A call is reported when the provider's response ends, before its tool calls run, so a run aborted during a tool call still reports the call that requested it. Retried calls and validation retries are included. `onUsage` must not throw; an error it throws is reported like a session listener error and does not change the run's outcome. `runAgent` delivers every event of the run before its promise settles. `AgentSession.create()` accepts the same callback for every generation and compaction of the session; `generate()` and `scheduleGenerate()` without a `transaction` reject as soon as their abort signal fires and may report the interrupted call afterwards; `close()` waits for that report. With a `transaction`, they settle only after every call of the generation has been reported.
|
|
198
|
+
|
|
199
|
+
### What is counted
|
|
200
|
+
|
|
201
|
+
- Every model step of a generation, whatever its outcome, including retried calls and validation retries.
|
|
202
|
+
- The model calls of a context compaction, when the compactor reports them. `contextCompactor` and `onContextOverflow` receive `reportUsage` in their options; it reports into `onUsage` with `source: 'compaction'`. `compactMessages()` from `@modelprofile.com/flexharness/compaction` reports each attempt of its model call when it receives `reportUsage`, so pass the handler's options through:
|
|
203
|
+
|
|
204
|
+
```typescript
|
|
205
|
+
import { compactMessages } from '@modelprofile.com/flexharness/compaction';
|
|
206
|
+
|
|
207
|
+
const session = await AgentSession.create({
|
|
208
|
+
...setup,
|
|
209
|
+
contextCompactor: (messages, _events, options) => compactMessages(setup.model, messages, options),
|
|
210
|
+
onContextOverflow: (messages, options) => compactMessages(setup.model, messages, options),
|
|
211
|
+
onUsage: (event) => console.log(event),
|
|
212
|
+
});
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
A handler that makes its own model calls reports each one through `reportUsage` before its promise settles, a call that fails or is aborted as `unreported`. `AgentModelCallUsageRecorder` does the bookkeeping for one language model: call `start()` when a call begins, `end(responseModelId, usage)` with the AI SDK's `onLanguageModelCallEnd` values, and `settle(reason)` when a call ends otherwise.
|
|
216
|
+
|
|
217
|
+
Not counted: model calls a handler makes without reporting them through `reportUsage`, and model calls outside the session, such as tools that call models themselves. Report those in your own accounting.
|
|
218
|
+
|
|
159
219
|
## Validate an answer and request corrections
|
|
160
220
|
|
|
161
221
|
`validateCompletion` returns `void` to accept a result or a string to add a corrective user message and generate again. `maxValidationRetries` defaults to `0`: a failed validation throws unless retries are configured.
|
|
@@ -242,9 +302,9 @@ Use an application-owned durable adapter for persistent sessions. Its `save(sess
|
|
|
242
302
|
### Bound model context and active events
|
|
243
303
|
|
|
244
304
|
- `contextBuilder({ events })` controls the model-message projection.
|
|
245
|
-
- `contextCompactor(messages, events, { reason, abortSignal })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention.
|
|
305
|
+
- `contextCompactor(messages, events, { reason, abortSignal, reportUsage })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention. Report the usage of its model calls through `reportUsage`; see [What is counted](#what-is-counted).
|
|
246
306
|
- `eventRetention: { maxEvents }` triggers compaction and archival when the active event count exceeds the threshold. It also requires an event store with `archive()` support.
|
|
247
|
-
- Context overflow invokes the configured compactor, or the `onContextOverflow` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
|
|
307
|
+
- Context overflow invokes the configured compactor, or the `onContextOverflow(messages, { abortSignal, reportUsage })` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
|
|
248
308
|
- Open transactions and uncertain tool intents constrain when compaction and archival can proceed. Resolve them before manual compaction.
|
|
249
309
|
|
|
250
310
|
Compaction changes the active model context; archival moves covered events out of the active event set. Implement archive retention in your chosen store when you need a complete audit history.
|
|
@@ -268,7 +328,7 @@ For generation-scoped resources, `generate({ prepare })` accepts a callback that
|
|
|
268
328
|
|
|
269
329
|
Model transports, tools and resource callbacks must observe their supplied abort signals. Background jobs have separate lifecycles; closing a session does not automatically terminate them. Cleanup failures are observable. `closeCleanupCompleted` reports whether the session has released its retryable cleanup ownership.
|
|
270
330
|
|
|
271
|
-
The runtime retries
|
|
331
|
+
The runtime retries a model call that failed with a rate limit (429), an overloaded provider (529) or an unavailable one (503) at most 8 times, waiting for the provider's `retry-after` delay but at least the backoff that doubles from 2 s to 30 s. A requested delay beyond 60 s, or one that would take the call's retries past 150 s in total, fails the generation: a rate limit (429) as a `ModelLimitError` of kind `rate_limit` carrying that retry time, a 503 or 529 with the provider error. A `ModelLimitError` of kind `usage_limit`, which the provider adapters raise for a spent quota, is never retried. Completed tool results are preserved during a generation's retry sequence. A ledger reuses an already observed execution for the same tool-call identity. Applications still need their own idempotency and recovery rules for external side effects.
|
|
272
332
|
|
|
273
333
|
`maxSteps` defaults to `20` and limits completed model steps per generation. Prompt-cache handling defaults to `cache: 'auto'`; use `cache: false` to disable the runtime's cache defaults, or pass an explicit cache policy and a stable `sessionId` for supported provider affinity.
|
|
274
334
|
|