@modelprofile.com/flexharness-agent 8.3.0 → 8.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist_ts_agent/classes.sessionengine.d.ts +3 -0
- package/dist_ts_agent/classes.sessionengine.js +43 -11
- package/dist_ts_agent/index.d.ts +3 -1
- package/dist_ts_agent/index.js +2 -1
- package/dist_ts_agent/plugins.d.ts +1 -1
- package/dist_ts_agent/plugins.js +1 -1
- package/dist_ts_agent/runner.d.ts +1 -1
- package/dist_ts_agent/runtime.run.js +2 -1
- package/dist_ts_agent/smartagent.interfaces.d.ts +77 -8
- package/dist_ts_agent/smartagent.interfaces.js +1 -1
- package/dist_ts_agent/smartagent.usage.d.ts +25 -0
- package/dist_ts_agent/smartagent.usage.js +53 -0
- package/package.json +2 -2
- package/readme.md +62 -3
- package/ts_agent/classes.sessionengine.ts +61 -8
- package/ts_agent/index.ts +7 -0
- package/ts_agent/plugins.ts +1 -0
- package/ts_agent/readme.md +62 -3
- package/ts_agent/runner.ts +8 -1
- package/ts_agent/runtime.run.ts +1 -0
- package/ts_agent/smartagent.interfaces.ts +82 -8
- package/ts_agent/smartagent.usage.ts +66 -0
package/ts_agent/readme.md
CHANGED
|
@@ -88,7 +88,7 @@ The following Node examples reuse `setup` and, where needed, `tools` from the qu
|
|
|
88
88
|
| `messages` | Current AI SDK message history after projection or compaction. Pass it into another run to continue. |
|
|
89
89
|
| `steps` | Completed model steps, including steps from validation-triggered attempts. A step can call several tools. |
|
|
90
90
|
| `finishReason` | The model's final finish reason; inspect this together with application validation. |
|
|
91
|
-
| `usage` | Input, output and total tokens, plus cache-read and cache-write tokens. |
|
|
91
|
+
| `usage` | Input, output and total tokens, plus cache-read and cache-write tokens, summed over every model call of the run the provider reported: its model steps, retried calls included, and the reported calls of a compaction its context overflow caused. |
|
|
92
92
|
| `toolCalls` | Tool-call IDs, names and inputs, with available outputs or errors. |
|
|
93
93
|
|
|
94
94
|
```typescript
|
|
@@ -154,9 +154,68 @@ Pass these callbacks to `runAgent` or `AgentSession.create()`:
|
|
|
154
154
|
| `onToolCallUpdate(event)` | The call identity and a transient streamed `output`. |
|
|
155
155
|
| `onToolCallFinish(event)` | The call identity plus either `success: true, output` or `success: false, error`. |
|
|
156
156
|
| `onRetry(event)` | Before each wait to retry a model call: `attempt`, `maxAttempts`, `delayMs` and `reason` (`rate_limit`, `overloaded`, `unavailable`). |
|
|
157
|
+
| `onUsage(event)` | Once per model call, as soon as its usage is known; see [Count the usage of every run](#count-the-usage-of-every-run). |
|
|
157
158
|
|
|
158
159
|
Tool updates are transient; the finish callback carries the authoritative final output. Use `subscribe()` for committed session changes (`committed`, `updated`, `archived`), and retain its returned unsubscribe function. Session listeners are delivered in order per listener, have bounded queues and timeouts, and are removed on failure. Streaming callbacks and session-change listeners serve different purposes.
|
|
159
160
|
|
|
161
|
+
## Count the usage of every run
|
|
162
|
+
|
|
163
|
+
A run that throws or is aborted has used tokens too, and its promise carries no result. `onUsage` reports each model call once, as soon as its usage is known, whatever the run's outcome. Sum it to count a run's usage:
|
|
164
|
+
|
|
165
|
+
```typescript
|
|
166
|
+
import type { IAgentUsage } from '@modelprofile.com/flexharness-agent';
|
|
167
|
+
|
|
168
|
+
const used: IAgentUsage = { inputTokens: 0, outputTokens: 0, totalTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 };
|
|
169
|
+
let usageComplete = true;
|
|
170
|
+
try {
|
|
171
|
+
await runAgent({
|
|
172
|
+
...setup,
|
|
173
|
+
tools,
|
|
174
|
+
prompt: 'Convert 10 km to miles.',
|
|
175
|
+
abort: AbortSignal.timeout(60_000),
|
|
176
|
+
onUsage: (event) => {
|
|
177
|
+
if (event.status === 'unreported') {
|
|
178
|
+
usageComplete = false;
|
|
179
|
+
return;
|
|
180
|
+
}
|
|
181
|
+
for (const key of Object.keys(used) as (keyof IAgentUsage)[]) used[key] += event.usage[key];
|
|
182
|
+
},
|
|
183
|
+
});
|
|
184
|
+
} finally {
|
|
185
|
+
console.log(used, usageComplete ? 'complete' : 'lower bound');
|
|
186
|
+
}
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
| Event field | Meaning |
|
|
190
|
+
| --- | --- |
|
|
191
|
+
| `source` | `generation`: a model step of the generation. `compaction`: a model call of a context compaction. |
|
|
192
|
+
| `generationId` | The generation the call belongs to. A compaction carries the generation whose context overflow caused it; a compaction by `compact()` or event retention has none. |
|
|
193
|
+
| `provider`, `requestedModelId` | The model the call was made with: the language model's `provider` and `modelId`. Every event carries both, whatever its status, so key usage caps on them. |
|
|
194
|
+
| `status: 'reported'`, `usage`, `responseModelId` | The provider reported the call's usage. A count the provider leaves out is zero. `responseModelId` is the model id the provider's response named (a provider may answer with a dated model version), or `requestedModelId` when it named none. |
|
|
195
|
+
| `status: 'unreported'`, `reason` | The call ended before the provider reported its usage: `aborted` (the call was aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage). The provider may still have consumed tokens for it; their number is unknown. |
|
|
196
|
+
|
|
197
|
+
When a run returns, its reported calls sum to `result.usage`; count one or the other, not both. A call is reported when the provider's response ends, before its tool calls run, so a run aborted during a tool call still reports the call that requested it. Retried calls and validation retries are included. `onUsage` must not throw; an error it throws is reported like a session listener error and does not change the run's outcome. `runAgent` delivers every event of the run before its promise settles. `AgentSession.create()` accepts the same callback for every generation and compaction of the session; `generate()` and `scheduleGenerate()` without a `transaction` reject as soon as their abort signal fires and may report the interrupted call afterwards; `close()` waits for that report. With a `transaction`, they settle only after every call of the generation has been reported.
|
|
198
|
+
|
|
199
|
+
### What is counted
|
|
200
|
+
|
|
201
|
+
- Every model step of a generation, whatever its outcome, including retried calls and validation retries.
|
|
202
|
+
- The model calls of a context compaction, when the compactor reports them. `contextCompactor` and `onContextOverflow` receive `reportUsage` in their options; it reports into `onUsage` with `source: 'compaction'`. `compactMessages()` from `@modelprofile.com/flexharness/compaction` reports each attempt of its model call when it receives `reportUsage`, so pass the handler's options through:
|
|
203
|
+
|
|
204
|
+
```typescript
|
|
205
|
+
import { compactMessages } from '@modelprofile.com/flexharness/compaction';
|
|
206
|
+
|
|
207
|
+
const session = await AgentSession.create({
|
|
208
|
+
...setup,
|
|
209
|
+
contextCompactor: (messages, _events, options) => compactMessages(setup.model, messages, options),
|
|
210
|
+
onContextOverflow: (messages, options) => compactMessages(setup.model, messages, options),
|
|
211
|
+
onUsage: (event) => console.log(event),
|
|
212
|
+
});
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
A handler that makes its own model calls reports each one through `reportUsage` before its promise settles, a call that fails or is aborted as `unreported`. `AgentModelCallUsageRecorder` does the bookkeeping for one language model: call `start()` when a call begins, `end(responseModelId, usage)` with the AI SDK's `onLanguageModelCallEnd` values, and `settle(reason)` when a call ends otherwise.
|
|
216
|
+
|
|
217
|
+
Not counted: model calls a handler makes without reporting them through `reportUsage`, and model calls outside the session, such as tools that call models themselves. Report those in your own accounting.
|
|
218
|
+
|
|
160
219
|
## Validate an answer and request corrections
|
|
161
220
|
|
|
162
221
|
`validateCompletion` returns `void` to accept a result or a string to add a corrective user message and generate again. `maxValidationRetries` defaults to `0`: a failed validation throws unless retries are configured.
|
|
@@ -243,9 +302,9 @@ Use an application-owned durable adapter for persistent sessions. Its `save(sess
|
|
|
243
302
|
### Bound model context and active events
|
|
244
303
|
|
|
245
304
|
- `contextBuilder({ events })` controls the model-message projection.
|
|
246
|
-
- `contextCompactor(messages, events, { reason, abortSignal })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention.
|
|
305
|
+
- `contextCompactor(messages, events, { reason, abortSignal, reportUsage })` returns replacement model messages. Provide it to use `session.compact()` or automatic event retention. Report the usage of its model calls through `reportUsage`; see [What is counted](#what-is-counted).
|
|
247
306
|
- `eventRetention: { maxEvents }` triggers compaction and archival when the active event count exceeds the threshold. It also requires an event store with `archive()` support.
|
|
248
|
-
- Context overflow invokes the configured compactor, or the `onContextOverflow` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
|
|
307
|
+
- Context overflow invokes the configured compactor, or the `onContextOverflow(messages, { abortSignal, reportUsage })` handler. Without either, generation throws `ContextOverflowError`. `maxContextOverflowRetries` defaults to `3`.
|
|
249
308
|
- Open transactions and uncertain tool intents constrain when compaction and archival can proceed. Resolve them before manual compaction.
|
|
250
309
|
|
|
251
310
|
Compaction changes the active model context; archival moves covered events out of the active event set. Implement archive retention in your chosen store when you need a complete audit history.
|
package/ts_agent/runner.ts
CHANGED
|
@@ -37,4 +37,11 @@ export function runAgent(options: IAgentRunOptions): Promise<IAgentRunResult> {
|
|
|
37
37
|
return runAgentWithSession(options, (sessionOptions) => RunnerSession.create(sessionOptions));
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
export type {
|
|
40
|
+
export type {
|
|
41
|
+
IAgentRunOptions,
|
|
42
|
+
IAgentRunResult,
|
|
43
|
+
IAgentUsage,
|
|
44
|
+
TAgentModelCallUnreportedReason,
|
|
45
|
+
TAgentModelCallUsage,
|
|
46
|
+
TAgentModelCallUsageEvent,
|
|
47
|
+
} from './smartagent.interfaces.js';
|
package/ts_agent/runtime.run.ts
CHANGED
|
@@ -52,6 +52,7 @@ export async function runAgentWithSession(
|
|
|
52
52
|
onToolCallUpdate: options.onToolCallUpdate,
|
|
53
53
|
onToolCallFinish: options.onToolCallFinish,
|
|
54
54
|
onRetry: options.onRetry,
|
|
55
|
+
onUsage: options.onUsage,
|
|
55
56
|
onToolCall: options.onToolCall,
|
|
56
57
|
onToolResult: options.onToolResult,
|
|
57
58
|
onContextOverflow: options.onContextOverflow,
|
|
@@ -54,8 +54,69 @@ export type TAgentToolCallFinishEvent =
|
|
|
54
54
|
error: string;
|
|
55
55
|
});
|
|
56
56
|
|
|
57
|
+
/** Token usage of model calls. A count a provider leaves out counts as zero. */
|
|
58
|
+
export interface IAgentUsage {
|
|
59
|
+
inputTokens: number;
|
|
60
|
+
outputTokens: number;
|
|
61
|
+
totalTokens: number;
|
|
62
|
+
cacheReadTokens: number;
|
|
63
|
+
cacheWriteTokens: number;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
interface IAgentModelCallIdentity {
|
|
67
|
+
/** The provider of the model that was called, as the language model names it. */
|
|
68
|
+
provider: string;
|
|
69
|
+
/**
|
|
70
|
+
* The model id the call was made with: the language model's `modelId`. Every call carries it,
|
|
71
|
+
* whatever its outcome, so key usage caps on `provider` and `requestedModelId`.
|
|
72
|
+
*/
|
|
73
|
+
requestedModelId: string;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Why a model call ended without the provider reporting its usage. */
|
|
77
|
+
export type TAgentModelCallUnreportedReason = 'aborted' | 'failed' | 'missing';
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* The usage outcome of one model call, as the component that made the call reports it.
|
|
81
|
+
* `reported`: the provider reported the call's usage. `responseModelId` is the model id the
|
|
82
|
+
* provider's response named, or `requestedModelId` when the response named none.
|
|
83
|
+
* `unreported`: the call ended before the provider reported its usage: `aborted` (the call was
|
|
84
|
+
* aborted), `failed` (the call or its response failed) or `missing` (the response carried no usage).
|
|
85
|
+
* The provider may still have consumed tokens for it; their number is unknown.
|
|
86
|
+
*/
|
|
87
|
+
export type TAgentModelCallUsage =
|
|
88
|
+
| (IAgentModelCallIdentity & {
|
|
89
|
+
status: 'reported';
|
|
90
|
+
responseModelId: string;
|
|
91
|
+
usage: IAgentUsage;
|
|
92
|
+
})
|
|
93
|
+
| (IAgentModelCallIdentity & {
|
|
94
|
+
status: 'unreported';
|
|
95
|
+
reason: TAgentModelCallUnreportedReason;
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
/** Receives the usage outcome of each model call. Must not throw. */
|
|
99
|
+
export type TAgentModelCallUsageReporter = (call: TAgentModelCallUsage) => void;
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* The usage outcome of one model call of a session, reported once per call through `onUsage`.
|
|
103
|
+
* `source: 'generation'`: a model step of the generation `generationId`.
|
|
104
|
+
* `source: 'compaction'`: a model call of a context compaction. `generationId` names the generation
|
|
105
|
+
* whose context overflow caused it; it is absent for `compact()` and event retention.
|
|
106
|
+
*/
|
|
107
|
+
export type TAgentModelCallUsageEvent = TAgentModelCallUsage & (
|
|
108
|
+
| { source: 'generation'; generationId: string }
|
|
109
|
+
| { source: 'compaction'; generationId?: string }
|
|
110
|
+
);
|
|
111
|
+
|
|
57
112
|
export interface IAgentContextOverflowInvocationOptions {
|
|
58
113
|
abortSignal?: AbortSignal;
|
|
114
|
+
/**
|
|
115
|
+
* Reports the usage of each model call the handler makes, into the session's `onUsage` as
|
|
116
|
+
* `source: 'compaction'`. The session always supplies it. Report every call before the returned
|
|
117
|
+
* promise settles, including calls that fail or are aborted.
|
|
118
|
+
*/
|
|
119
|
+
reportUsage?: TAgentModelCallUsageReporter;
|
|
59
120
|
}
|
|
60
121
|
|
|
61
122
|
export interface IAgentContextBuildOptions {
|
|
@@ -69,6 +130,12 @@ export type TAgentContextBuilder = (
|
|
|
69
130
|
export interface IAgentContextCompactionOptions {
|
|
70
131
|
abortSignal?: AbortSignal;
|
|
71
132
|
reason: 'context-overflow' | 'retention' | 'manual';
|
|
133
|
+
/**
|
|
134
|
+
* Reports the usage of each model call the compactor makes, into the session's `onUsage` as
|
|
135
|
+
* `source: 'compaction'`. The session always supplies it. Report every call before the returned
|
|
136
|
+
* promise settles, including calls that fail or are aborted.
|
|
137
|
+
*/
|
|
138
|
+
reportUsage?: TAgentModelCallUsageReporter;
|
|
72
139
|
}
|
|
73
140
|
|
|
74
141
|
export type TAgentContextCompactor = (
|
|
@@ -153,6 +220,16 @@ export interface IAgentSessionOptions {
|
|
|
153
220
|
onToolCallFinish?: (event: TAgentToolCallFinishEvent) => void;
|
|
154
221
|
/** Called before the engine waits to retry a rate-limited or unavailable model call. */
|
|
155
222
|
onRetry?: (event: IAgentRetryEvent) => void;
|
|
223
|
+
/**
|
|
224
|
+
* Called once for every model call of the session, as soon as its usage is known: when the
|
|
225
|
+
* provider reports it, or when the call ends without it. This covers the generation's model
|
|
226
|
+
* steps, the model call of `compactMessages()` when a compactor passes it `reportUsage`, and every
|
|
227
|
+
* call a `contextCompactor` or `onContextOverflow` handler reports through `reportUsage`. The
|
|
228
|
+
* reported calls of a generation, including those of a compaction its context overflow caused,
|
|
229
|
+
* sum to the generation's `usage` when it returns. Must not throw; an error it throws is reported
|
|
230
|
+
* like a session listener error and does not change the generation's outcome.
|
|
231
|
+
*/
|
|
232
|
+
onUsage?: (event: TAgentModelCallUsageEvent) => void;
|
|
156
233
|
/** @deprecated Use onToolCallStart instead. */
|
|
157
234
|
onToolCall?: (toolName: string, input: unknown) => void;
|
|
158
235
|
/** @deprecated Use onToolCallFinish instead. */
|
|
@@ -303,14 +380,11 @@ export interface IAgentRunResult {
|
|
|
303
380
|
steps: number;
|
|
304
381
|
/** Finish reason from the final step */
|
|
305
382
|
finishReason: string;
|
|
306
|
-
/**
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
cacheReadTokens: number;
|
|
312
|
-
cacheWriteTokens: number;
|
|
313
|
-
};
|
|
383
|
+
/**
|
|
384
|
+
* Token usage the provider reported for every model call of the run: its model steps, retried
|
|
385
|
+
* calls included, and the calls of a compaction its context overflow caused that were reported
|
|
386
|
+
*/
|
|
387
|
+
usage: IAgentUsage;
|
|
314
388
|
/** Tool calls observed during the run, including inputs and outputs/errors when available */
|
|
315
389
|
toolCalls: IAgentToolCallRecord[];
|
|
316
390
|
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import type { LanguageModelUsage } from './plugins.js';
|
|
2
|
+
import type {
|
|
3
|
+
IAgentUsage,
|
|
4
|
+
TAgentModelCallUnreportedReason,
|
|
5
|
+
TAgentModelCallUsageReporter,
|
|
6
|
+
} from './smartagent.interfaces.js';
|
|
7
|
+
|
|
8
|
+
/** The identity of the language model whose calls a recorder follows. */
|
|
9
|
+
export interface IAgentModelCallUsageRecorderModel {
|
|
10
|
+
readonly provider: string;
|
|
11
|
+
readonly modelId: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Follows the calls of one language model and reports each call's usage exactly once.
|
|
16
|
+
* Call `start()` when a model call begins, `end()` when its provider reports usage, and `settle()`
|
|
17
|
+
* when the call ends otherwise. A call that is still open when the next one starts ended without
|
|
18
|
+
* usage and is reported as `unreported` with reason `missing`.
|
|
19
|
+
*/
|
|
20
|
+
export class AgentModelCallUsageRecorder {
|
|
21
|
+
private open = false;
|
|
22
|
+
|
|
23
|
+
constructor(
|
|
24
|
+
private readonly model: IAgentModelCallUsageRecorderModel,
|
|
25
|
+
private readonly report: TAgentModelCallUsageReporter,
|
|
26
|
+
) {}
|
|
27
|
+
|
|
28
|
+
/** A model call begins. */
|
|
29
|
+
public start(): void {
|
|
30
|
+
this.settle('missing');
|
|
31
|
+
this.open = true;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** The provider reported the usage of the current call. A count it leaves out counts as zero. */
|
|
35
|
+
public end(responseModelId: string, providerUsage: LanguageModelUsage): void {
|
|
36
|
+
this.open = false;
|
|
37
|
+
const inputTokens = providerUsage.inputTokens ?? 0;
|
|
38
|
+
const outputTokens = providerUsage.outputTokens ?? 0;
|
|
39
|
+
const usage: IAgentUsage = {
|
|
40
|
+
inputTokens,
|
|
41
|
+
outputTokens,
|
|
42
|
+
totalTokens: inputTokens + outputTokens,
|
|
43
|
+
cacheReadTokens: providerUsage.inputTokenDetails.cacheReadTokens ?? 0,
|
|
44
|
+
cacheWriteTokens: providerUsage.inputTokenDetails.cacheWriteTokens ?? 0,
|
|
45
|
+
};
|
|
46
|
+
this.report({
|
|
47
|
+
status: 'reported',
|
|
48
|
+
provider: this.model.provider,
|
|
49
|
+
requestedModelId: this.model.modelId,
|
|
50
|
+
responseModelId,
|
|
51
|
+
usage,
|
|
52
|
+
});
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** The current call ended without usage. Does nothing when no call is open. */
|
|
56
|
+
public settle(reason: TAgentModelCallUnreportedReason): void {
|
|
57
|
+
if (!this.open) return;
|
|
58
|
+
this.open = false;
|
|
59
|
+
this.report({
|
|
60
|
+
status: 'unreported',
|
|
61
|
+
provider: this.model.provider,
|
|
62
|
+
requestedModelId: this.model.modelId,
|
|
63
|
+
reason,
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
}
|