floe-guard 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -3
- package/dist/index.cjs +206 -21
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +166 -15
- package/dist/index.d.ts +166 -15
- package/dist/index.js +204 -21
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.cts
CHANGED
|
@@ -68,7 +68,8 @@ declare namespace pricing {
|
|
|
68
68
|
* One priced spend event in the guard's per-call ledger.
|
|
69
69
|
*
|
|
70
70
|
* Every {@link BudgetGuard.record} / {@link BudgetGuard.settle} /
|
|
71
|
-
* {@link BudgetGuard.recordTool}
|
|
71
|
+
* {@link BudgetGuard.recordTool} / {@link BudgetGuard.settleTool} that accrues
|
|
72
|
+
* spend appends exactly one event, so
|
|
72
73
|
* the ledger's costs sum to `spentUsd` (unless a `maxLogEvents` ring buffer has
|
|
73
74
|
* evicted old events). The schema is identical in the Python
|
|
74
75
|
* package (`SpendEvent` in `src/floe_guard/guard.py`) and
|
|
@@ -149,13 +150,26 @@ declare class BudgetGuard {
|
|
|
149
150
|
failClosed: boolean;
|
|
150
151
|
nearLimitBps: number;
|
|
151
152
|
private readonly onBlock;
|
|
152
|
-
/**
|
|
153
|
-
|
|
153
|
+
/**
|
|
154
|
+
* Costs of the most recent priced LLM call and tool call, tracked
|
|
155
|
+
* SEPARATELY: the default next-call prediction is the max of the two, so a
|
|
156
|
+
* cheap tool call can't shrink the estimate right before an expensive LLM
|
|
157
|
+
* call (or vice versa) — conservative beats one-call-too-late.
|
|
158
|
+
*/
|
|
159
|
+
private lastLlmCost;
|
|
160
|
+
private lastToolCost;
|
|
154
161
|
/** USD held for in-flight calls (reserved, not yet settled). Counts toward the ceiling. */
|
|
155
162
|
private reserved;
|
|
156
163
|
/** Per-call ledger, oldest first; a ring buffer when maxLogEvents is set. */
|
|
157
164
|
private readonly spendEvents;
|
|
158
165
|
private readonly maxLogEvents?;
|
|
166
|
+
/**
|
|
167
|
+
* Per-tool running totals (settleTool/recordTool) — the tool side of the one
|
|
168
|
+
* shared ceiling, exposed via the toolCosts getter. null-prototype: tool
|
|
169
|
+
* names are caller-supplied strings, so a "__proto__" name is stored as
|
|
170
|
+
* plain data instead of mutating the object's prototype.
|
|
171
|
+
*/
|
|
172
|
+
private readonly toolCostTotals;
|
|
159
173
|
/**
|
|
160
174
|
* @param limitUsd the spend ceiling, in USD. `0` blocks the very first call.
|
|
161
175
|
*/
|
|
@@ -164,7 +178,8 @@ declare class BudgetGuard {
|
|
|
164
178
|
* Throw {@link BudgetExceeded} if the next call would cross the ceiling.
|
|
165
179
|
*
|
|
166
180
|
* Call this immediately before each LLM request. The "next call" is estimated
|
|
167
|
-
*
|
|
181
|
+
* conservatively as the costlier of the last LLM call and the last tool call
|
|
182
|
+
* (override with `estimatedNextCost`); the
|
|
168
183
|
* first call is always allowed unless the ceiling is already met. In-flight
|
|
169
184
|
* reservations count toward the total, so this stays correct alongside
|
|
170
185
|
* {@link BudgetGuard.reserve}.
|
|
@@ -181,7 +196,8 @@ declare class BudgetGuard {
|
|
|
181
196
|
* the same stale total. Throws {@link BudgetExceeded} (without reserving) if
|
|
182
197
|
* the reservation would cross the ceiling. Returns the reservation handle to
|
|
183
198
|
* pass to {@link BudgetGuard.settle} (or {@link BudgetGuard.release} on error).
|
|
184
|
-
* `estimatedCost` defaults to the last call
|
|
199
|
+
* `estimatedCost` defaults to the costlier of the last LLM call and the last
|
|
200
|
+
* tool call.
|
|
185
201
|
*/
|
|
186
202
|
reserve(estimatedCost?: number): number;
|
|
187
203
|
/**
|
|
@@ -209,16 +225,49 @@ declare class BudgetGuard {
|
|
|
209
225
|
price?: ManualPrice;
|
|
210
226
|
label?: string;
|
|
211
227
|
}): number;
|
|
228
|
+
/**
|
|
229
|
+
* Atomically check the ceiling AND hold a tool call's cost in flight.
|
|
230
|
+
*
|
|
231
|
+
* The tool-spend counterpart of {@link BudgetGuard.reserve} — and STRONGER
|
|
232
|
+
* than the LLM path, because a paid tool's price is usually known exactly
|
|
233
|
+
* before the call, so the pre-call hard-stop is precise rather than
|
|
234
|
+
* estimated:
|
|
235
|
+
*
|
|
236
|
+
* const handle = guard.reserveTool(0.02); // throws BEFORE Apollo runs
|
|
237
|
+
* const result = await apollo.peopleLookup(...);
|
|
238
|
+
* guard.settleTool("apollo.people_lookup", 0.02, { reserved: handle });
|
|
239
|
+
*
|
|
240
|
+
* Throws {@link BudgetExceeded} (without reserving) if the call would cross
|
|
241
|
+
* the ceiling. The estimate is required — tools have no last-cost prediction
|
|
242
|
+
* worth falling back to. Pass the returned handle to
|
|
243
|
+
* {@link BudgetGuard.settleTool}, or {@link BudgetGuard.release} on failure.
|
|
244
|
+
*/
|
|
245
|
+
reserveTool(estimatedCost: number): number;
|
|
246
|
+
/**
|
|
247
|
+
* Release a reservation and record a tool call's actual cost.
|
|
248
|
+
*
|
|
249
|
+
* `recordTool` is `settleTool` with no reservation. The caller supplies the
|
|
250
|
+
* cost — tools have no token usage to price. Accrues into the same
|
|
251
|
+
* `spentUsd` ceiling as tokens, tallies the per-tool total
|
|
252
|
+
* ({@link BudgetGuard.toolCosts}), updates the tool side of the next-call
|
|
253
|
+
* estimate (tracked separately from the LLM side; the default prediction is
|
|
254
|
+
* the max of the two, so a tool-hammering loop's plain `check()` stops
|
|
255
|
+
* BEFORE the crossing call without a cheap tool shrinking the LLM
|
|
256
|
+
* prediction), and appends
|
|
257
|
+
* a `kind: "tool"` {@link SpendEvent} to {@link BudgetGuard.spendLog}.
|
|
258
|
+
* Returns `costUsd`.
|
|
259
|
+
*/
|
|
260
|
+
settleTool(tool: string, costUsd: number, options?: {
|
|
261
|
+
reserved?: number;
|
|
262
|
+
label?: string;
|
|
263
|
+
}): number;
|
|
212
264
|
/**
|
|
213
265
|
* Accrue a non-LLM cost (a paid tool/API call) against the same ceiling.
|
|
214
266
|
*
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
*
|
|
219
|
-
* cost: tools have no token usage to price. Deliberately does NOT update the
|
|
220
|
-
* next-call estimate — that predicts the next *LLM* call, and a tool's price
|
|
221
|
-
* would skew it. Returns `costUsd`.
|
|
267
|
+
* Post-hoc accrual for costs only known after the call (metered APIs); when
|
|
268
|
+
* the price is known up front, {@link BudgetGuard.reserveTool} /
|
|
269
|
+
* {@link BudgetGuard.settleTool} give the stronger pre-call hard-stop. See
|
|
270
|
+
* `settleTool` for the full contract. Returns `costUsd`.
|
|
222
271
|
*/
|
|
223
272
|
recordTool(tool: string, costUsd: number, options?: {
|
|
224
273
|
label?: string;
|
|
@@ -230,10 +279,17 @@ declare class BudgetGuard {
|
|
|
230
279
|
release(reserved: number): void;
|
|
231
280
|
/** USD left before the ceiling, net of in-flight reservations (never negative). */
|
|
232
281
|
get remainingUsd(): number;
|
|
282
|
+
/**
|
|
283
|
+
* Per-tool running USD totals, keyed by the name given to `settleTool()` /
|
|
284
|
+
* `recordTool()` — e.g. `{"apollo.people_lookup": 0.42, "exa.search": 0.11}`.
|
|
285
|
+
* Makes the token/tool split of the one shared ceiling inspectable
|
|
286
|
+
* (`spentUsd - sum of toolCosts` is the token side). Returns a snapshot copy.
|
|
287
|
+
*/
|
|
288
|
+
get toolCosts(): Record<string, number>;
|
|
233
289
|
/**
|
|
234
290
|
* The per-call spend ledger, oldest first — one {@link SpendEvent} per priced
|
|
235
|
-
* `record()` / `settle()` / `recordTool()`. Returns a
|
|
236
|
-
* it cannot corrupt the ledger.
|
|
291
|
+
* `record()` / `settle()` / `recordTool()` / `settleTool()`. Returns a
|
|
292
|
+
* snapshot copy: mutating it cannot corrupt the ledger.
|
|
237
293
|
*/
|
|
238
294
|
get spendLog(): SpendEvent[];
|
|
239
295
|
/**
|
|
@@ -247,6 +303,25 @@ declare class BudgetGuard {
|
|
|
247
303
|
* Python `2.5e-06`.) Empty ledger yields `""`.
|
|
248
304
|
*/
|
|
249
305
|
exportLog(): string;
|
|
306
|
+
/**
|
|
307
|
+
* The default next-call prediction when the caller supplies no estimate.
|
|
308
|
+
* Conservative: the costlier of the last LLM call and the last tool call — a
|
|
309
|
+
* mixed loop predicts the pricier kind, which at worst blocks one call early
|
|
310
|
+
* (fail-closed) rather than letting a crossing call through because the LAST
|
|
311
|
+
* event happened to be cheap.
|
|
312
|
+
*/
|
|
313
|
+
private defaultEstimate;
|
|
314
|
+
/**
|
|
315
|
+
* Subtract a settled/released hold from the in-flight tally. A handle larger
|
|
316
|
+
* than EVERYTHING currently held cannot have come from a matching
|
|
317
|
+
* `reserve()` — throwing beats silently clamping, which would free OTHER
|
|
318
|
+
* callers' holds and fail the ceiling open. The epsilon absorbs float dust
|
|
319
|
+
* from accumulating and draining many holds; per-caller over-release (a
|
|
320
|
+
* handle within the total but larger than the caller's own hold) is
|
|
321
|
+
* undetectable without per-handle tracking and remains the caller's
|
|
322
|
+
* responsibility.
|
|
323
|
+
*/
|
|
324
|
+
private consumeReservation;
|
|
250
325
|
private appendEvent;
|
|
251
326
|
/**
|
|
252
327
|
* Context-aware spend advisory for this budget — see {@link BudgetAdvisory}.
|
|
@@ -258,6 +333,77 @@ declare class BudgetGuard {
|
|
|
258
333
|
advisory(): BudgetAdvisory;
|
|
259
334
|
}
|
|
260
335
|
|
|
336
|
+
/**
|
|
337
|
+
* LatencyBudget — a cumulative tool-chain deadline, sibling to BudgetGuard.
|
|
338
|
+
*
|
|
339
|
+
* BudgetGuard stops an agent before its next call crosses a USD ceiling;
|
|
340
|
+
* LatencyBudget stops it before the next call would blow an end-user SLA:
|
|
341
|
+
*
|
|
342
|
+
* ```ts
|
|
343
|
+
* const deadline = new LatencyBudget(5000);
|
|
344
|
+
* ...
|
|
345
|
+
* deadline.check(800); // throws DeadlineExceeded when projected over
|
|
346
|
+
* if (deadline.advisory().nearDeadline) useFasterModel();
|
|
347
|
+
* router.pick({ maxLatencyMs: deadline.remainingMs });
|
|
348
|
+
* ```
|
|
349
|
+
*
|
|
350
|
+
* Design notes (mirroring `src/floe_guard/latency.py`):
|
|
351
|
+
* - **Monotonic clock** — `performance.now()`, never wall time.
|
|
352
|
+
* - **Cooperative, not preemptive** — the guard provides the deadline signal;
|
|
353
|
+
* aborting a stalled in-flight call is the framework's job (AbortSignal).
|
|
354
|
+
* `check()` prevents the NEXT call from starting.
|
|
355
|
+
* - **Advisory symmetry** — `nearDeadline` / `usedBps` / `remainingMs` are the
|
|
356
|
+
* latency twin of BudgetGuard's `nearLimit` / `usedBps` / `remainingUsd`.
|
|
357
|
+
* - **In-process scope** — one instance per request/run; distributed latency
|
|
358
|
+
* tracking is out of scope.
|
|
359
|
+
*/
|
|
360
|
+
/** A context-aware deadline signal — the latency twin of {@link BudgetAdvisory}.
|
|
361
|
+
* Soft by design; the hard-stop is {@link LatencyBudget.check}. */
|
|
362
|
+
interface LatencyAdvisory {
|
|
363
|
+
nearDeadline: boolean;
|
|
364
|
+
/** SLA consumed, basis points 0..10000 (8500 = 85%). */
|
|
365
|
+
usedBps: number;
|
|
366
|
+
remainingMs: number;
|
|
367
|
+
slaMs: number;
|
|
368
|
+
elapsedMs: number;
|
|
369
|
+
}
|
|
370
|
+
interface LatencyBudgetOptions {
|
|
371
|
+
/**
|
|
372
|
+
* Utilization (basis points, 0..10000) at which {@link LatencyBudget.advisory}
|
|
373
|
+
* flags `nearDeadline` so an agent can downshift to a faster path before the
|
|
374
|
+
* wall. Default 8000 (80%), matching BudgetGuard's `nearLimitBps`.
|
|
375
|
+
*/
|
|
376
|
+
nearDeadlineBps?: number;
|
|
377
|
+
/** Invoked with `(elapsedMs, slaMs)` right before {@link DeadlineExceeded} is thrown. */
|
|
378
|
+
onBlock?: (elapsedMs: number, slaMs: number) => void;
|
|
379
|
+
/** Milliseconds-returning monotonic clock, injectable for tests. Defaults to `performance.now`. */
|
|
380
|
+
clock?: () => number;
|
|
381
|
+
}
|
|
382
|
+
declare class LatencyBudget {
|
|
383
|
+
readonly slaMs: number;
|
|
384
|
+
readonly nearDeadlineBps: number;
|
|
385
|
+
private readonly onBlock?;
|
|
386
|
+
private readonly clock;
|
|
387
|
+
private readonly startedAt;
|
|
388
|
+
/** The budget starts counting at construction — build it when the request
|
|
389
|
+
* (and its SLA) starts. */
|
|
390
|
+
constructor(slaMs: number, options?: LatencyBudgetOptions);
|
|
391
|
+
/** Milliseconds since construction (monotonic). */
|
|
392
|
+
get elapsedMs(): number;
|
|
393
|
+
/** Milliseconds left before the SLA, floored at 0 — the readable signal a
|
|
394
|
+
* router uses to pick a faster fallback or truncate work mid-chain. */
|
|
395
|
+
get remainingMs(): number;
|
|
396
|
+
/**
|
|
397
|
+
* Throw {@link DeadlineExceeded} when the projected elapsed time (now +
|
|
398
|
+
* `expectedMs` for the upcoming call) would blow the SLA. Call it
|
|
399
|
+
* immediately before each tool/model call; pass 0 to only gate on time
|
|
400
|
+
* already spent.
|
|
401
|
+
*/
|
|
402
|
+
check(expectedMs?: number): void;
|
|
403
|
+
/** The soft near-deadline signal — symmetric to `BudgetGuard.advisory()`. */
|
|
404
|
+
advisory(): LatencyAdvisory;
|
|
405
|
+
}
|
|
406
|
+
|
|
261
407
|
/**
|
|
262
408
|
* Exceptions for floe-guard.
|
|
263
409
|
*
|
|
@@ -293,6 +439,11 @@ declare class UnpriceableModelError extends FloeGuardError {
|
|
|
293
439
|
readonly model: string;
|
|
294
440
|
constructor(model: string);
|
|
295
441
|
}
|
|
442
|
+
declare class DeadlineExceeded extends FloeGuardError {
|
|
443
|
+
readonly elapsedMs: number;
|
|
444
|
+
readonly slaMs: number;
|
|
445
|
+
constructor(elapsedMs: number, slaMs: number);
|
|
446
|
+
}
|
|
296
447
|
|
|
297
448
|
/**
|
|
298
449
|
* Vercel AI SDK middleware that enforces a {@link BudgetGuard} in the call path.
|
|
@@ -363,4 +514,4 @@ interface BudgetGuardMiddleware {
|
|
|
363
514
|
*/
|
|
364
515
|
declare function budgetGuardMiddleware(guard: BudgetGuard): BudgetGuardMiddleware;
|
|
365
516
|
|
|
366
|
-
export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, FloeGuardError, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
|
|
517
|
+
export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, DeadlineExceeded, FloeGuardError, type LatencyAdvisory, LatencyBudget, type LatencyBudgetOptions, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
|
package/dist/index.d.ts
CHANGED
|
@@ -68,7 +68,8 @@ declare namespace pricing {
|
|
|
68
68
|
* One priced spend event in the guard's per-call ledger.
|
|
69
69
|
*
|
|
70
70
|
* Every {@link BudgetGuard.record} / {@link BudgetGuard.settle} /
|
|
71
|
-
* {@link BudgetGuard.recordTool}
|
|
71
|
+
* {@link BudgetGuard.recordTool} / {@link BudgetGuard.settleTool} that accrues
|
|
72
|
+
* spend appends exactly one event, so
|
|
72
73
|
* the ledger's costs sum to `spentUsd` (unless a `maxLogEvents` ring buffer has
|
|
73
74
|
* evicted old events). The schema is identical in the Python
|
|
74
75
|
* package (`SpendEvent` in `src/floe_guard/guard.py`) and
|
|
@@ -149,13 +150,26 @@ declare class BudgetGuard {
|
|
|
149
150
|
failClosed: boolean;
|
|
150
151
|
nearLimitBps: number;
|
|
151
152
|
private readonly onBlock;
|
|
152
|
-
/**
|
|
153
|
-
|
|
153
|
+
/**
|
|
154
|
+
* Costs of the most recent priced LLM call and tool call, tracked
|
|
155
|
+
* SEPARATELY: the default next-call prediction is the max of the two, so a
|
|
156
|
+
* cheap tool call can't shrink the estimate right before an expensive LLM
|
|
157
|
+
* call (or vice versa) — conservative beats one-call-too-late.
|
|
158
|
+
*/
|
|
159
|
+
private lastLlmCost;
|
|
160
|
+
private lastToolCost;
|
|
154
161
|
/** USD held for in-flight calls (reserved, not yet settled). Counts toward the ceiling. */
|
|
155
162
|
private reserved;
|
|
156
163
|
/** Per-call ledger, oldest first; a ring buffer when maxLogEvents is set. */
|
|
157
164
|
private readonly spendEvents;
|
|
158
165
|
private readonly maxLogEvents?;
|
|
166
|
+
/**
|
|
167
|
+
* Per-tool running totals (settleTool/recordTool) — the tool side of the one
|
|
168
|
+
* shared ceiling, exposed via the toolCosts getter. null-prototype: tool
|
|
169
|
+
* names are caller-supplied strings, so a "__proto__" name is stored as
|
|
170
|
+
* plain data instead of mutating the object's prototype.
|
|
171
|
+
*/
|
|
172
|
+
private readonly toolCostTotals;
|
|
159
173
|
/**
|
|
160
174
|
* @param limitUsd the spend ceiling, in USD. `0` blocks the very first call.
|
|
161
175
|
*/
|
|
@@ -164,7 +178,8 @@ declare class BudgetGuard {
|
|
|
164
178
|
* Throw {@link BudgetExceeded} if the next call would cross the ceiling.
|
|
165
179
|
*
|
|
166
180
|
* Call this immediately before each LLM request. The "next call" is estimated
|
|
167
|
-
*
|
|
181
|
+
* conservatively as the costlier of the last LLM call and the last tool call
|
|
182
|
+
* (override with `estimatedNextCost`); the
|
|
168
183
|
* first call is always allowed unless the ceiling is already met. In-flight
|
|
169
184
|
* reservations count toward the total, so this stays correct alongside
|
|
170
185
|
* {@link BudgetGuard.reserve}.
|
|
@@ -181,7 +196,8 @@ declare class BudgetGuard {
|
|
|
181
196
|
* the same stale total. Throws {@link BudgetExceeded} (without reserving) if
|
|
182
197
|
* the reservation would cross the ceiling. Returns the reservation handle to
|
|
183
198
|
* pass to {@link BudgetGuard.settle} (or {@link BudgetGuard.release} on error).
|
|
184
|
-
* `estimatedCost` defaults to the last call
|
|
199
|
+
* `estimatedCost` defaults to the costlier of the last LLM call and the last
|
|
200
|
+
* tool call.
|
|
185
201
|
*/
|
|
186
202
|
reserve(estimatedCost?: number): number;
|
|
187
203
|
/**
|
|
@@ -209,16 +225,49 @@ declare class BudgetGuard {
|
|
|
209
225
|
price?: ManualPrice;
|
|
210
226
|
label?: string;
|
|
211
227
|
}): number;
|
|
228
|
+
/**
|
|
229
|
+
* Atomically check the ceiling AND hold a tool call's cost in flight.
|
|
230
|
+
*
|
|
231
|
+
* The tool-spend counterpart of {@link BudgetGuard.reserve} — and STRONGER
|
|
232
|
+
* than the LLM path, because a paid tool's price is usually known exactly
|
|
233
|
+
* before the call, so the pre-call hard-stop is precise rather than
|
|
234
|
+
* estimated:
|
|
235
|
+
*
|
|
236
|
+
* const handle = guard.reserveTool(0.02); // throws BEFORE Apollo runs
|
|
237
|
+
* const result = await apollo.peopleLookup(...);
|
|
238
|
+
* guard.settleTool("apollo.people_lookup", 0.02, { reserved: handle });
|
|
239
|
+
*
|
|
240
|
+
* Throws {@link BudgetExceeded} (without reserving) if the call would cross
|
|
241
|
+
* the ceiling. The estimate is required — tools have no last-cost prediction
|
|
242
|
+
* worth falling back to. Pass the returned handle to
|
|
243
|
+
* {@link BudgetGuard.settleTool}, or {@link BudgetGuard.release} on failure.
|
|
244
|
+
*/
|
|
245
|
+
reserveTool(estimatedCost: number): number;
|
|
246
|
+
/**
|
|
247
|
+
* Release a reservation and record a tool call's actual cost.
|
|
248
|
+
*
|
|
249
|
+
* `recordTool` is `settleTool` with no reservation. The caller supplies the
|
|
250
|
+
* cost — tools have no token usage to price. Accrues into the same
|
|
251
|
+
* `spentUsd` ceiling as tokens, tallies the per-tool total
|
|
252
|
+
* ({@link BudgetGuard.toolCosts}), updates the tool side of the next-call
|
|
253
|
+
* estimate (tracked separately from the LLM side; the default prediction is
|
|
254
|
+
* the max of the two, so a tool-hammering loop's plain `check()` stops
|
|
255
|
+
* BEFORE the crossing call without a cheap tool shrinking the LLM
|
|
256
|
+
* prediction), and appends
|
|
257
|
+
* a `kind: "tool"` {@link SpendEvent} to {@link BudgetGuard.spendLog}.
|
|
258
|
+
* Returns `costUsd`.
|
|
259
|
+
*/
|
|
260
|
+
settleTool(tool: string, costUsd: number, options?: {
|
|
261
|
+
reserved?: number;
|
|
262
|
+
label?: string;
|
|
263
|
+
}): number;
|
|
212
264
|
/**
|
|
213
265
|
* Accrue a non-LLM cost (a paid tool/API call) against the same ceiling.
|
|
214
266
|
*
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
*
|
|
219
|
-
* cost: tools have no token usage to price. Deliberately does NOT update the
|
|
220
|
-
* next-call estimate — that predicts the next *LLM* call, and a tool's price
|
|
221
|
-
* would skew it. Returns `costUsd`.
|
|
267
|
+
* Post-hoc accrual for costs only known after the call (metered APIs); when
|
|
268
|
+
* the price is known up front, {@link BudgetGuard.reserveTool} /
|
|
269
|
+
* {@link BudgetGuard.settleTool} give the stronger pre-call hard-stop. See
|
|
270
|
+
* `settleTool` for the full contract. Returns `costUsd`.
|
|
222
271
|
*/
|
|
223
272
|
recordTool(tool: string, costUsd: number, options?: {
|
|
224
273
|
label?: string;
|
|
@@ -230,10 +279,17 @@ declare class BudgetGuard {
|
|
|
230
279
|
release(reserved: number): void;
|
|
231
280
|
/** USD left before the ceiling, net of in-flight reservations (never negative). */
|
|
232
281
|
get remainingUsd(): number;
|
|
282
|
+
/**
|
|
283
|
+
* Per-tool running USD totals, keyed by the name given to `settleTool()` /
|
|
284
|
+
* `recordTool()` — e.g. `{"apollo.people_lookup": 0.42, "exa.search": 0.11}`.
|
|
285
|
+
* Makes the token/tool split of the one shared ceiling inspectable
|
|
286
|
+
* (`spentUsd - sum of toolCosts` is the token side). Returns a snapshot copy.
|
|
287
|
+
*/
|
|
288
|
+
get toolCosts(): Record<string, number>;
|
|
233
289
|
/**
|
|
234
290
|
* The per-call spend ledger, oldest first — one {@link SpendEvent} per priced
|
|
235
|
-
* `record()` / `settle()` / `recordTool()`. Returns a
|
|
236
|
-
* it cannot corrupt the ledger.
|
|
291
|
+
* `record()` / `settle()` / `recordTool()` / `settleTool()`. Returns a
|
|
292
|
+
* snapshot copy: mutating it cannot corrupt the ledger.
|
|
237
293
|
*/
|
|
238
294
|
get spendLog(): SpendEvent[];
|
|
239
295
|
/**
|
|
@@ -247,6 +303,25 @@ declare class BudgetGuard {
|
|
|
247
303
|
* Python `2.5e-06`.) Empty ledger yields `""`.
|
|
248
304
|
*/
|
|
249
305
|
exportLog(): string;
|
|
306
|
+
/**
|
|
307
|
+
* The default next-call prediction when the caller supplies no estimate.
|
|
308
|
+
* Conservative: the costlier of the last LLM call and the last tool call — a
|
|
309
|
+
* mixed loop predicts the pricier kind, which at worst blocks one call early
|
|
310
|
+
* (fail-closed) rather than letting a crossing call through because the LAST
|
|
311
|
+
* event happened to be cheap.
|
|
312
|
+
*/
|
|
313
|
+
private defaultEstimate;
|
|
314
|
+
/**
|
|
315
|
+
* Subtract a settled/released hold from the in-flight tally. A handle larger
|
|
316
|
+
* than EVERYTHING currently held cannot have come from a matching
|
|
317
|
+
* `reserve()` — throwing beats silently clamping, which would free OTHER
|
|
318
|
+
* callers' holds and fail the ceiling open. The epsilon absorbs float dust
|
|
319
|
+
* from accumulating and draining many holds; per-caller over-release (a
|
|
320
|
+
* handle within the total but larger than the caller's own hold) is
|
|
321
|
+
* undetectable without per-handle tracking and remains the caller's
|
|
322
|
+
* responsibility.
|
|
323
|
+
*/
|
|
324
|
+
private consumeReservation;
|
|
250
325
|
private appendEvent;
|
|
251
326
|
/**
|
|
252
327
|
* Context-aware spend advisory for this budget — see {@link BudgetAdvisory}.
|
|
@@ -258,6 +333,77 @@ declare class BudgetGuard {
|
|
|
258
333
|
advisory(): BudgetAdvisory;
|
|
259
334
|
}
|
|
260
335
|
|
|
336
|
+
/**
|
|
337
|
+
* LatencyBudget — a cumulative tool-chain deadline, sibling to BudgetGuard.
|
|
338
|
+
*
|
|
339
|
+
* BudgetGuard stops an agent before its next call crosses a USD ceiling;
|
|
340
|
+
* LatencyBudget stops it before the next call would blow an end-user SLA:
|
|
341
|
+
*
|
|
342
|
+
* ```ts
|
|
343
|
+
* const deadline = new LatencyBudget(5000);
|
|
344
|
+
* ...
|
|
345
|
+
* deadline.check(800); // throws DeadlineExceeded when projected over
|
|
346
|
+
* if (deadline.advisory().nearDeadline) useFasterModel();
|
|
347
|
+
* router.pick({ maxLatencyMs: deadline.remainingMs });
|
|
348
|
+
* ```
|
|
349
|
+
*
|
|
350
|
+
* Design notes (mirroring `src/floe_guard/latency.py`):
|
|
351
|
+
* - **Monotonic clock** — `performance.now()`, never wall time.
|
|
352
|
+
* - **Cooperative, not preemptive** — the guard provides the deadline signal;
|
|
353
|
+
* aborting a stalled in-flight call is the framework's job (AbortSignal).
|
|
354
|
+
* `check()` prevents the NEXT call from starting.
|
|
355
|
+
* - **Advisory symmetry** — `nearDeadline` / `usedBps` / `remainingMs` are the
|
|
356
|
+
* latency twin of BudgetGuard's `nearLimit` / `usedBps` / `remainingUsd`.
|
|
357
|
+
* - **In-process scope** — one instance per request/run; distributed latency
|
|
358
|
+
* tracking is out of scope.
|
|
359
|
+
*/
|
|
360
|
+
/** A context-aware deadline signal — the latency twin of {@link BudgetAdvisory}.
|
|
361
|
+
* Soft by design; the hard-stop is {@link LatencyBudget.check}. */
|
|
362
|
+
interface LatencyAdvisory {
|
|
363
|
+
nearDeadline: boolean;
|
|
364
|
+
/** SLA consumed, basis points 0..10000 (8500 = 85%). */
|
|
365
|
+
usedBps: number;
|
|
366
|
+
remainingMs: number;
|
|
367
|
+
slaMs: number;
|
|
368
|
+
elapsedMs: number;
|
|
369
|
+
}
|
|
370
|
+
interface LatencyBudgetOptions {
|
|
371
|
+
/**
|
|
372
|
+
* Utilization (basis points, 0..10000) at which {@link LatencyBudget.advisory}
|
|
373
|
+
* flags `nearDeadline` so an agent can downshift to a faster path before the
|
|
374
|
+
* wall. Default 8000 (80%), matching BudgetGuard's `nearLimitBps`.
|
|
375
|
+
*/
|
|
376
|
+
nearDeadlineBps?: number;
|
|
377
|
+
/** Invoked with `(elapsedMs, slaMs)` right before {@link DeadlineExceeded} is thrown. */
|
|
378
|
+
onBlock?: (elapsedMs: number, slaMs: number) => void;
|
|
379
|
+
/** Milliseconds-returning monotonic clock, injectable for tests. Defaults to `performance.now`. */
|
|
380
|
+
clock?: () => number;
|
|
381
|
+
}
|
|
382
|
+
declare class LatencyBudget {
|
|
383
|
+
readonly slaMs: number;
|
|
384
|
+
readonly nearDeadlineBps: number;
|
|
385
|
+
private readonly onBlock?;
|
|
386
|
+
private readonly clock;
|
|
387
|
+
private readonly startedAt;
|
|
388
|
+
/** The budget starts counting at construction — build it when the request
|
|
389
|
+
* (and its SLA) starts. */
|
|
390
|
+
constructor(slaMs: number, options?: LatencyBudgetOptions);
|
|
391
|
+
/** Milliseconds since construction (monotonic). */
|
|
392
|
+
get elapsedMs(): number;
|
|
393
|
+
/** Milliseconds left before the SLA, floored at 0 — the readable signal a
|
|
394
|
+
* router uses to pick a faster fallback or truncate work mid-chain. */
|
|
395
|
+
get remainingMs(): number;
|
|
396
|
+
/**
|
|
397
|
+
* Throw {@link DeadlineExceeded} when the projected elapsed time (now +
|
|
398
|
+
* `expectedMs` for the upcoming call) would blow the SLA. Call it
|
|
399
|
+
* immediately before each tool/model call; pass 0 to only gate on time
|
|
400
|
+
* already spent.
|
|
401
|
+
*/
|
|
402
|
+
check(expectedMs?: number): void;
|
|
403
|
+
/** The soft near-deadline signal — symmetric to `BudgetGuard.advisory()`. */
|
|
404
|
+
advisory(): LatencyAdvisory;
|
|
405
|
+
}
|
|
406
|
+
|
|
261
407
|
/**
|
|
262
408
|
* Exceptions for floe-guard.
|
|
263
409
|
*
|
|
@@ -293,6 +439,11 @@ declare class UnpriceableModelError extends FloeGuardError {
|
|
|
293
439
|
readonly model: string;
|
|
294
440
|
constructor(model: string);
|
|
295
441
|
}
|
|
442
|
+
declare class DeadlineExceeded extends FloeGuardError {
|
|
443
|
+
readonly elapsedMs: number;
|
|
444
|
+
readonly slaMs: number;
|
|
445
|
+
constructor(elapsedMs: number, slaMs: number);
|
|
446
|
+
}
|
|
296
447
|
|
|
297
448
|
/**
|
|
298
449
|
* Vercel AI SDK middleware that enforces a {@link BudgetGuard} in the call path.
|
|
@@ -363,4 +514,4 @@ interface BudgetGuardMiddleware {
|
|
|
363
514
|
*/
|
|
364
515
|
declare function budgetGuardMiddleware(guard: BudgetGuard): BudgetGuardMiddleware;
|
|
365
516
|
|
|
366
|
-
export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, FloeGuardError, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
|
|
517
|
+
export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, DeadlineExceeded, FloeGuardError, type LatencyAdvisory, LatencyBudget, type LatencyBudgetOptions, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
|