@modelprofile.com/flexharness-agent 8.2.0 → 8.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,101 @@
1
+ import * as plugins from './plugins.js';
2
+
3
+ export type TAgentRetryReason = 'rate_limit' | 'overloaded' | 'unavailable';
4
+
5
+ /** A failed model call the session engine retries after `delayMs`. */
6
+ export interface IAgentRetryEvent {
7
+ /** One-based number of this retry within the current model call. */
8
+ attempt: number;
9
+ maxAttempts: number;
10
+ delayMs: number;
11
+ reason: TAgentRetryReason;
12
+ }
13
+
14
+ export type TAgentRetryDecision =
15
+ | { action: 'retry'; delayMs: number; reason: TAgentRetryReason }
16
+ | { action: 'fail'; error: unknown }
17
+ | { action: 'not-retryable' };
18
+
19
+ export interface IAgentRetryState {
20
+ /** Retries already made for the current model call. */
21
+ attempt: number;
22
+ /** Retry delay already spent on the current model call. */
23
+ retriedMs: number;
24
+ /** Provider id used when an untyped rate limit has to fail with a typed limit error. */
25
+ provider: string;
26
+ now: number;
27
+ }
28
+
29
+ export const MAX_RETRY_ATTEMPTS = 8;
30
+ /** The longest provider-requested delay honoured; the AI SDK uses the same bound. */
31
+ export const MAX_RETRY_AFTER_MS = 60_000;
32
+ /** The total delay one model call may spend retrying, equal to the full default backoff. */
33
+ export const MAX_RETRY_WINDOW_MS = 150_000;
34
+
35
+ const RETRY_INITIAL_DELAY_MS = 2000;
36
+ const RETRY_BACKOFF_FACTOR = 2;
37
+ const RETRY_BACKOFF_MAX_DELAY_MS = 30_000;
38
+
39
+ interface IRetrySignal {
40
+ reason: TAgentRetryReason;
41
+ requestedDelayMs?: number;
42
+ }
43
+
44
+ function backoffDelayMs(attempt: number): number {
45
+ return Math.min(RETRY_INITIAL_DELAY_MS * RETRY_BACKOFF_FACTOR ** (attempt - 1), RETRY_BACKOFF_MAX_DELAY_MS);
46
+ }
47
+
48
+ function untypedRetryReason(error: unknown): TAgentRetryReason | undefined {
49
+ const candidate = error as { status?: number; statusCode?: number } | undefined;
50
+ const status = candidate?.status ?? candidate?.statusCode;
51
+ const message = error instanceof Error ? error.message.toLowerCase() : '';
52
+ if (status === 429 || message.includes('rate limit') || message.includes('too many requests')) {
53
+ return 'rate_limit';
54
+ }
55
+ if (status === 529 || message.includes('overloaded')) return 'overloaded';
56
+ if (status === 503) return 'unavailable';
57
+ return undefined;
58
+ }
59
+
60
+ function readRetrySignal(error: unknown, now: number): IRetrySignal | undefined {
61
+ if (plugins.isModelLimitError(error)) {
62
+ const { retryAfterMs, resetsAt } = error.limit;
63
+ const requestedDelayMs = retryAfterMs ?? (resetsAt === undefined ? undefined : Math.max(0, resetsAt - now));
64
+ return { reason: 'rate_limit', ...(requestedDelayMs === undefined ? {} : { requestedDelayMs }) };
65
+ }
66
+ const reason = untypedRetryReason(error);
67
+ if (reason === undefined) return undefined;
68
+ const failure = error as { responseHeaders?: Record<string, string>; headers?: Record<string, string> };
69
+ const requestedDelayMs = plugins.readRetryAfterMs(failure.responseHeaders ?? failure.headers, now);
70
+ return { reason, ...(requestedDelayMs === undefined ? {} : { requestedDelayMs }) };
71
+ }
72
+
73
+ /** The error a rate limit fails with when its retry time lies beyond the retry bounds. */
74
+ function overdueLimitError(error: unknown, signal: IRetrySignal, state: IAgentRetryState): unknown {
75
+ if (plugins.isModelLimitError(error) || signal.reason !== 'rate_limit' || signal.requestedDelayMs === undefined) {
76
+ return error;
77
+ }
78
+ const limit = plugins.createModelLimitInfo(
79
+ { kind: 'rate_limit', provider: state.provider, retryAfterMs: signal.requestedDelayMs },
80
+ state.now,
81
+ );
82
+ // an empty or overlong provider id cannot describe a limit, so the provider's own error stays
83
+ return plugins.isModelLimitInfo(limit) ? new plugins.ModelLimitError(limit, { cause: error }) : error;
84
+ }
85
+
86
+ /**
87
+ * Decides whether a failed model call is retried. Usage limits never are; transient limits wait for
88
+ * the provider's delay up to MAX_RETRY_AFTER_MS, but never less than the backoff, and all retries of
89
+ * one call stay within MAX_RETRY_ATTEMPTS and MAX_RETRY_WINDOW_MS.
90
+ */
91
+ export function planModelCallRetry(error: unknown, state: IAgentRetryState): TAgentRetryDecision {
92
+ if (plugins.isModelLimitError(error) && error.limit.kind === 'usage_limit') return { action: 'fail', error };
93
+ const signal = readRetrySignal(error, state.now);
94
+ if (signal === undefined) return { action: 'not-retryable' };
95
+ if (state.attempt >= MAX_RETRY_ATTEMPTS) return { action: 'fail', error };
96
+ const delayMs = Math.max(backoffDelayMs(state.attempt + 1), signal.requestedDelayMs ?? 0);
97
+ if (delayMs > MAX_RETRY_AFTER_MS || state.retriedMs + delayMs > MAX_RETRY_WINDOW_MS) {
98
+ return { action: 'fail', error: overdueLimitError(error, signal, state) };
99
+ }
100
+ return { action: 'retry', delayMs, reason: signal.reason };
101
+ }
@@ -0,0 +1,66 @@
1
+ import type { LanguageModelUsage } from './plugins.js';
2
+ import type {
3
+ IAgentUsage,
4
+ TAgentModelCallUnreportedReason,
5
+ TAgentModelCallUsageReporter,
6
+ } from './smartagent.interfaces.js';
7
+
8
+ /** The identity of the language model whose calls a recorder follows. */
9
+ export interface IAgentModelCallUsageRecorderModel {
10
+ readonly provider: string;
11
+ readonly modelId: string;
12
+ }
13
+
14
+ /**
15
+ * Follows the calls of one language model and reports each call's usage exactly once.
16
+ * Call `start()` when a model call begins, `end()` when its provider reports usage, and `settle()`
17
+ * when the call ends otherwise. A call that is still open when the next one starts ended without
18
+ * usage and is reported as `unreported` with reason `missing`.
19
+ */
20
+ export class AgentModelCallUsageRecorder {
21
+ private open = false;
22
+
23
+ constructor(
24
+ private readonly model: IAgentModelCallUsageRecorderModel,
25
+ private readonly report: TAgentModelCallUsageReporter,
26
+ ) {}
27
+
28
+ /** A model call begins. */
29
+ public start(): void {
30
+ this.settle('missing');
31
+ this.open = true;
32
+ }
33
+
34
+ /** The provider reported the usage of the current call. A count it leaves out counts as zero. */
35
+ public end(responseModelId: string, providerUsage: LanguageModelUsage): void {
36
+ this.open = false;
37
+ const inputTokens = providerUsage.inputTokens ?? 0;
38
+ const outputTokens = providerUsage.outputTokens ?? 0;
39
+ const usage: IAgentUsage = {
40
+ inputTokens,
41
+ outputTokens,
42
+ totalTokens: inputTokens + outputTokens,
43
+ cacheReadTokens: providerUsage.inputTokenDetails.cacheReadTokens ?? 0,
44
+ cacheWriteTokens: providerUsage.inputTokenDetails.cacheWriteTokens ?? 0,
45
+ };
46
+ this.report({
47
+ status: 'reported',
48
+ provider: this.model.provider,
49
+ requestedModelId: this.model.modelId,
50
+ responseModelId,
51
+ usage,
52
+ });
53
+ }
54
+
55
+ /** The current call ended without usage. Does nothing when no call is open. */
56
+ public settle(reason: TAgentModelCallUnreportedReason): void {
57
+ if (!this.open) return;
58
+ this.open = false;
59
+ this.report({
60
+ status: 'unreported',
61
+ provider: this.model.provider,
62
+ requestedModelId: this.model.modelId,
63
+ reason,
64
+ });
65
+ }
66
+ }