floe-guard 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -68,7 +68,8 @@ declare namespace pricing {
68
68
  * One priced spend event in the guard's per-call ledger.
69
69
  *
70
70
  * Every {@link BudgetGuard.record} / {@link BudgetGuard.settle} /
71
- * {@link BudgetGuard.recordTool} that accrues spend appends exactly one event, so
71
+ * {@link BudgetGuard.recordTool} / {@link BudgetGuard.settleTool} that accrues
72
+ * spend appends exactly one event, so
72
73
  * the ledger's costs sum to `spentUsd` (unless a `maxLogEvents` ring buffer has
73
74
  * evicted old events). The schema is identical in the Python
74
75
  * package (`SpendEvent` in `src/floe_guard/guard.py`) and
@@ -149,13 +150,26 @@ declare class BudgetGuard {
149
150
  failClosed: boolean;
150
151
  nearLimitBps: number;
151
152
  private readonly onBlock;
152
- /** Cost of the most recent priced call, used to predict the next one. */
153
- private lastCost;
153
+ /**
154
+ * Costs of the most recent priced LLM call and tool call, tracked
155
+ * SEPARATELY: the default next-call prediction is the max of the two, so a
156
+ * cheap tool call can't shrink the estimate right before an expensive LLM
157
+ * call (or vice versa) — conservative beats one-call-too-late.
158
+ */
159
+ private lastLlmCost;
160
+ private lastToolCost;
154
161
  /** USD held for in-flight calls (reserved, not yet settled). Counts toward the ceiling. */
155
162
  private reserved;
156
163
  /** Per-call ledger, oldest first; a ring buffer when maxLogEvents is set. */
157
164
  private readonly spendEvents;
158
165
  private readonly maxLogEvents?;
166
+ /**
167
+ * Per-tool running totals (settleTool/recordTool) — the tool side of the one
168
+ * shared ceiling, exposed via the toolCosts getter. null-prototype: tool
169
+ * names are caller-supplied strings, so a "__proto__" name is stored as
170
+ * plain data instead of mutating the object's prototype.
171
+ */
172
+ private readonly toolCostTotals;
159
173
  /**
160
174
  * @param limitUsd the spend ceiling, in USD. `0` blocks the very first call.
161
175
  */
@@ -164,7 +178,8 @@ declare class BudgetGuard {
164
178
  * Throw {@link BudgetExceeded} if the next call would cross the ceiling.
165
179
  *
166
180
  * Call this immediately before each LLM request. The "next call" is estimated
167
- * from the last recorded call's cost (override with `estimatedNextCost`); the
181
+ * conservatively as the costlier of the last LLM call and the last tool call
182
+ * (override with `estimatedNextCost`); the
168
183
  * first call is always allowed unless the ceiling is already met. In-flight
169
184
  * reservations count toward the total, so this stays correct alongside
170
185
  * {@link BudgetGuard.reserve}.
@@ -181,7 +196,8 @@ declare class BudgetGuard {
181
196
  * the same stale total. Throws {@link BudgetExceeded} (without reserving) if
182
197
  * the reservation would cross the ceiling. Returns the reservation handle to
183
198
  * pass to {@link BudgetGuard.settle} (or {@link BudgetGuard.release} on error).
184
- * `estimatedCost` defaults to the last call's cost.
199
+ * `estimatedCost` defaults to the costlier of the last LLM call and the last
200
+ * tool call.
185
201
  */
186
202
  reserve(estimatedCost?: number): number;
187
203
  /**
@@ -209,16 +225,49 @@ declare class BudgetGuard {
209
225
  price?: ManualPrice;
210
226
  label?: string;
211
227
  }): number;
228
+ /**
229
+ * Atomically check the ceiling AND hold a tool call's cost in flight.
230
+ *
231
+ * The tool-spend counterpart of {@link BudgetGuard.reserve} — and STRONGER
232
+ * than the LLM path, because a paid tool's price is usually known exactly
233
+ * before the call, so the pre-call hard-stop is precise rather than
234
+ * estimated:
235
+ *
236
+ * const handle = guard.reserveTool(0.02); // throws BEFORE Apollo runs
237
+ * const result = await apollo.peopleLookup(...);
238
+ * guard.settleTool("apollo.people_lookup", 0.02, { reserved: handle });
239
+ *
240
+ * Throws {@link BudgetExceeded} (without reserving) if the call would cross
241
+ * the ceiling. The estimate is required — tools have no last-cost prediction
242
+ * worth falling back to. Pass the returned handle to
243
+ * {@link BudgetGuard.settleTool}, or {@link BudgetGuard.release} on failure.
244
+ */
245
+ reserveTool(estimatedCost: number): number;
246
+ /**
247
+ * Release a reservation and record a tool call's actual cost.
248
+ *
249
+ * `recordTool` is `settleTool` with no reservation. The caller supplies the
250
+ * cost — tools have no token usage to price. Accrues into the same
251
+ * `spentUsd` ceiling as tokens, tallies the per-tool total
252
+ * ({@link BudgetGuard.toolCosts}), updates the tool side of the next-call
253
+ * estimate (tracked separately from the LLM side; the default prediction is
254
+ * the max of the two, so a tool-hammering loop's plain `check()` stops
255
+ * BEFORE the crossing call without a cheap tool shrinking the LLM
256
+ * prediction), and appends
257
+ * a `kind: "tool"` {@link SpendEvent} to {@link BudgetGuard.spendLog}.
258
+ * Returns `costUsd`.
259
+ */
260
+ settleTool(tool: string, costUsd: number, options?: {
261
+ reserved?: number;
262
+ label?: string;
263
+ }): number;
212
264
  /**
213
265
  * Accrue a non-LLM cost (a paid tool/API call) against the same ceiling.
214
266
  *
215
- * Tools with direct dollar costs — search APIs, scrapers, sandboxes — spend the
216
- * same budget the LLM calls do; `recordTool` folds them into `spentUsd` (so
217
- * `check()` / `reserve()` see them) and appends a `kind: "tool"`
218
- * {@link SpendEvent} to {@link BudgetGuard.spendLog}. The caller supplies the
219
- * cost: tools have no token usage to price. Deliberately does NOT update the
220
- * next-call estimate — that predicts the next *LLM* call, and a tool's price
221
- * would skew it. Returns `costUsd`.
267
+ * Post-hoc accrual for costs only known after the call (metered APIs); when
268
+ * the price is known up front, {@link BudgetGuard.reserveTool} /
269
+ * {@link BudgetGuard.settleTool} give the stronger pre-call hard-stop. See
270
+ * `settleTool` for the full contract. Returns `costUsd`.
222
271
  */
223
272
  recordTool(tool: string, costUsd: number, options?: {
224
273
  label?: string;
@@ -230,10 +279,17 @@ declare class BudgetGuard {
230
279
  release(reserved: number): void;
231
280
  /** USD left before the ceiling, net of in-flight reservations (never negative). */
232
281
  get remainingUsd(): number;
282
+ /**
283
+ * Per-tool running USD totals, keyed by the name given to `settleTool()` /
284
+ * `recordTool()` — e.g. `{"apollo.people_lookup": 0.42, "exa.search": 0.11}`.
285
+ * Makes the token/tool split of the one shared ceiling inspectable
286
+ * (`spentUsd - sum of toolCosts` is the token side). Returns a snapshot copy.
287
+ */
288
+ get toolCosts(): Record<string, number>;
233
289
  /**
234
290
  * The per-call spend ledger, oldest first — one {@link SpendEvent} per priced
235
- * `record()` / `settle()` / `recordTool()`. Returns a snapshot copy: mutating
236
- * it cannot corrupt the ledger.
291
+ * `record()` / `settle()` / `recordTool()` / `settleTool()`. Returns a
292
+ * snapshot copy: mutating it cannot corrupt the ledger.
237
293
  */
238
294
  get spendLog(): SpendEvent[];
239
295
  /**
@@ -247,6 +303,25 @@ declare class BudgetGuard {
247
303
  * Python `2.5e-06`.) Empty ledger yields `""`.
248
304
  */
249
305
  exportLog(): string;
306
+ /**
307
+ * The default next-call prediction when the caller supplies no estimate.
308
+ * Conservative: the costlier of the last LLM call and the last tool call — a
309
+ * mixed loop predicts the pricier kind, which at worst blocks one call early
310
+ * (fail-closed) rather than letting a crossing call through because the LAST
311
+ * event happened to be cheap.
312
+ */
313
+ private defaultEstimate;
314
+ /**
315
+ * Subtract a settled/released hold from the in-flight tally. A handle larger
316
+ * than EVERYTHING currently held cannot have come from a matching
317
+ * `reserve()` — throwing beats silently clamping, which would free OTHER
318
+ * callers' holds and fail the ceiling open. The epsilon absorbs float dust
319
+ * from accumulating and draining many holds; per-caller over-release (a
320
+ * handle within the total but larger than the caller's own hold) is
321
+ * undetectable without per-handle tracking and remains the caller's
322
+ * responsibility.
323
+ */
324
+ private consumeReservation;
250
325
  private appendEvent;
251
326
  /**
252
327
  * Context-aware spend advisory for this budget — see {@link BudgetAdvisory}.
@@ -258,6 +333,77 @@ declare class BudgetGuard {
258
333
  advisory(): BudgetAdvisory;
259
334
  }
260
335
 
336
+ /**
337
+ * LatencyBudget — a cumulative tool-chain deadline, sibling to BudgetGuard.
338
+ *
339
+ * BudgetGuard stops an agent before its next call crosses a USD ceiling;
340
+ * LatencyBudget stops it before the next call would blow an end-user SLA:
341
+ *
342
+ * ```ts
343
+ * const deadline = new LatencyBudget(5000);
344
+ * ...
345
+ * deadline.check(800); // throws DeadlineExceeded when projected over
346
+ * if (deadline.advisory().nearDeadline) useFasterModel();
347
+ * router.pick({ maxLatencyMs: deadline.remainingMs });
348
+ * ```
349
+ *
350
+ * Design notes (mirroring `src/floe_guard/latency.py`):
351
+ * - **Monotonic clock** — `performance.now()`, never wall time.
352
+ * - **Cooperative, not preemptive** — the guard provides the deadline signal;
353
+ * aborting a stalled in-flight call is the framework's job (AbortSignal).
354
+ * `check()` prevents the NEXT call from starting.
355
+ * - **Advisory symmetry** — `nearDeadline` / `usedBps` / `remainingMs` are the
356
+ * latency twin of BudgetGuard's `nearLimit` / `usedBps` / `remainingUsd`.
357
+ * - **In-process scope** — one instance per request/run; distributed latency
358
+ * tracking is out of scope.
359
+ */
360
+ /** A context-aware deadline signal — the latency twin of {@link BudgetAdvisory}.
361
+ * Soft by design; the hard-stop is {@link LatencyBudget.check}. */
362
+ interface LatencyAdvisory {
363
+ nearDeadline: boolean;
364
+ /** SLA consumed, basis points 0..10000 (8500 = 85%). */
365
+ usedBps: number;
366
+ remainingMs: number;
367
+ slaMs: number;
368
+ elapsedMs: number;
369
+ }
370
+ interface LatencyBudgetOptions {
371
+ /**
372
+ * Utilization (basis points, 0..10000) at which {@link LatencyBudget.advisory}
373
+ * flags `nearDeadline` so an agent can downshift to a faster path before the
374
+ * wall. Default 8000 (80%), matching BudgetGuard's `nearLimitBps`.
375
+ */
376
+ nearDeadlineBps?: number;
377
+ /** Invoked with `(elapsedMs, slaMs)` right before {@link DeadlineExceeded} is thrown. */
378
+ onBlock?: (elapsedMs: number, slaMs: number) => void;
379
+ /** Milliseconds-returning monotonic clock, injectable for tests. Defaults to `performance.now`. */
380
+ clock?: () => number;
381
+ }
382
+ declare class LatencyBudget {
383
+ readonly slaMs: number;
384
+ readonly nearDeadlineBps: number;
385
+ private readonly onBlock?;
386
+ private readonly clock;
387
+ private readonly startedAt;
388
+ /** The budget starts counting at construction — build it when the request
389
+ * (and its SLA) starts. */
390
+ constructor(slaMs: number, options?: LatencyBudgetOptions);
391
+ /** Milliseconds since construction (monotonic). */
392
+ get elapsedMs(): number;
393
+ /** Milliseconds left before the SLA, floored at 0 — the readable signal a
394
+ * router uses to pick a faster fallback or truncate work mid-chain. */
395
+ get remainingMs(): number;
396
+ /**
397
+ * Throw {@link DeadlineExceeded} when the projected elapsed time (now +
398
+ * `expectedMs` for the upcoming call) would blow the SLA. Call it
399
+ * immediately before each tool/model call; pass 0 to only gate on time
400
+ * already spent.
401
+ */
402
+ check(expectedMs?: number): void;
403
+ /** The soft near-deadline signal — symmetric to `BudgetGuard.advisory()`. */
404
+ advisory(): LatencyAdvisory;
405
+ }
406
+
261
407
  /**
262
408
  * Exceptions for floe-guard.
263
409
  *
@@ -293,6 +439,11 @@ declare class UnpriceableModelError extends FloeGuardError {
293
439
  readonly model: string;
294
440
  constructor(model: string);
295
441
  }
442
+ declare class DeadlineExceeded extends FloeGuardError {
443
+ readonly elapsedMs: number;
444
+ readonly slaMs: number;
445
+ constructor(elapsedMs: number, slaMs: number);
446
+ }
296
447
 
297
448
  /**
298
449
  * Vercel AI SDK middleware that enforces a {@link BudgetGuard} in the call path.
@@ -363,4 +514,4 @@ interface BudgetGuardMiddleware {
363
514
  */
364
515
  declare function budgetGuardMiddleware(guard: BudgetGuard): BudgetGuardMiddleware;
365
516
 
366
- export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, FloeGuardError, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
517
+ export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, DeadlineExceeded, FloeGuardError, type LatencyAdvisory, LatencyBudget, type LatencyBudgetOptions, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
package/dist/index.d.ts CHANGED
@@ -68,7 +68,8 @@ declare namespace pricing {
68
68
  * One priced spend event in the guard's per-call ledger.
69
69
  *
70
70
  * Every {@link BudgetGuard.record} / {@link BudgetGuard.settle} /
71
- * {@link BudgetGuard.recordTool} that accrues spend appends exactly one event, so
71
+ * {@link BudgetGuard.recordTool} / {@link BudgetGuard.settleTool} that accrues
72
+ * spend appends exactly one event, so
72
73
  * the ledger's costs sum to `spentUsd` (unless a `maxLogEvents` ring buffer has
73
74
  * evicted old events). The schema is identical in the Python
74
75
  * package (`SpendEvent` in `src/floe_guard/guard.py`) and
@@ -149,13 +150,26 @@ declare class BudgetGuard {
149
150
  failClosed: boolean;
150
151
  nearLimitBps: number;
151
152
  private readonly onBlock;
152
- /** Cost of the most recent priced call, used to predict the next one. */
153
- private lastCost;
153
+ /**
154
+ * Costs of the most recent priced LLM call and tool call, tracked
155
+ * SEPARATELY: the default next-call prediction is the max of the two, so a
156
+ * cheap tool call can't shrink the estimate right before an expensive LLM
157
+ * call (or vice versa) — conservative beats one-call-too-late.
158
+ */
159
+ private lastLlmCost;
160
+ private lastToolCost;
154
161
  /** USD held for in-flight calls (reserved, not yet settled). Counts toward the ceiling. */
155
162
  private reserved;
156
163
  /** Per-call ledger, oldest first; a ring buffer when maxLogEvents is set. */
157
164
  private readonly spendEvents;
158
165
  private readonly maxLogEvents?;
166
+ /**
167
+ * Per-tool running totals (settleTool/recordTool) — the tool side of the one
168
+ * shared ceiling, exposed via the toolCosts getter. null-prototype: tool
169
+ * names are caller-supplied strings, so a "__proto__" name is stored as
170
+ * plain data instead of mutating the object's prototype.
171
+ */
172
+ private readonly toolCostTotals;
159
173
  /**
160
174
  * @param limitUsd the spend ceiling, in USD. `0` blocks the very first call.
161
175
  */
@@ -164,7 +178,8 @@ declare class BudgetGuard {
164
178
  * Throw {@link BudgetExceeded} if the next call would cross the ceiling.
165
179
  *
166
180
  * Call this immediately before each LLM request. The "next call" is estimated
167
- * from the last recorded call's cost (override with `estimatedNextCost`); the
181
+ * conservatively as the costlier of the last LLM call and the last tool call
182
+ * (override with `estimatedNextCost`); the
168
183
  * first call is always allowed unless the ceiling is already met. In-flight
169
184
  * reservations count toward the total, so this stays correct alongside
170
185
  * {@link BudgetGuard.reserve}.
@@ -181,7 +196,8 @@ declare class BudgetGuard {
181
196
  * the same stale total. Throws {@link BudgetExceeded} (without reserving) if
182
197
  * the reservation would cross the ceiling. Returns the reservation handle to
183
198
  * pass to {@link BudgetGuard.settle} (or {@link BudgetGuard.release} on error).
184
- * `estimatedCost` defaults to the last call's cost.
199
+ * `estimatedCost` defaults to the costlier of the last LLM call and the last
200
+ * tool call.
185
201
  */
186
202
  reserve(estimatedCost?: number): number;
187
203
  /**
@@ -209,16 +225,49 @@ declare class BudgetGuard {
209
225
  price?: ManualPrice;
210
226
  label?: string;
211
227
  }): number;
228
+ /**
229
+ * Atomically check the ceiling AND hold a tool call's cost in flight.
230
+ *
231
+ * The tool-spend counterpart of {@link BudgetGuard.reserve} — and STRONGER
232
+ * than the LLM path, because a paid tool's price is usually known exactly
233
+ * before the call, so the pre-call hard-stop is precise rather than
234
+ * estimated:
235
+ *
236
+ * const handle = guard.reserveTool(0.02); // throws BEFORE Apollo runs
237
+ * const result = await apollo.peopleLookup(...);
238
+ * guard.settleTool("apollo.people_lookup", 0.02, { reserved: handle });
239
+ *
240
+ * Throws {@link BudgetExceeded} (without reserving) if the call would cross
241
+ * the ceiling. The estimate is required — tools have no last-cost prediction
242
+ * worth falling back to. Pass the returned handle to
243
+ * {@link BudgetGuard.settleTool}, or {@link BudgetGuard.release} on failure.
244
+ */
245
+ reserveTool(estimatedCost: number): number;
246
+ /**
247
+ * Release a reservation and record a tool call's actual cost.
248
+ *
249
+ * `recordTool` is `settleTool` with no reservation. The caller supplies the
250
+ * cost — tools have no token usage to price. Accrues into the same
251
+ * `spentUsd` ceiling as tokens, tallies the per-tool total
252
+ * ({@link BudgetGuard.toolCosts}), updates the tool side of the next-call
253
+ * estimate (tracked separately from the LLM side; the default prediction is
254
+ * the max of the two, so a tool-hammering loop's plain `check()` stops
255
+ * BEFORE the crossing call without a cheap tool shrinking the LLM
256
+ * prediction), and appends
257
+ * a `kind: "tool"` {@link SpendEvent} to {@link BudgetGuard.spendLog}.
258
+ * Returns `costUsd`.
259
+ */
260
+ settleTool(tool: string, costUsd: number, options?: {
261
+ reserved?: number;
262
+ label?: string;
263
+ }): number;
212
264
  /**
213
265
  * Accrue a non-LLM cost (a paid tool/API call) against the same ceiling.
214
266
  *
215
- * Tools with direct dollar costs — search APIs, scrapers, sandboxes — spend the
216
- * same budget the LLM calls do; `recordTool` folds them into `spentUsd` (so
217
- * `check()` / `reserve()` see them) and appends a `kind: "tool"`
218
- * {@link SpendEvent} to {@link BudgetGuard.spendLog}. The caller supplies the
219
- * cost: tools have no token usage to price. Deliberately does NOT update the
220
- * next-call estimate — that predicts the next *LLM* call, and a tool's price
221
- * would skew it. Returns `costUsd`.
267
+ * Post-hoc accrual for costs only known after the call (metered APIs); when
268
+ * the price is known up front, {@link BudgetGuard.reserveTool} /
269
+ * {@link BudgetGuard.settleTool} give the stronger pre-call hard-stop. See
270
+ * `settleTool` for the full contract. Returns `costUsd`.
222
271
  */
223
272
  recordTool(tool: string, costUsd: number, options?: {
224
273
  label?: string;
@@ -230,10 +279,17 @@ declare class BudgetGuard {
230
279
  release(reserved: number): void;
231
280
  /** USD left before the ceiling, net of in-flight reservations (never negative). */
232
281
  get remainingUsd(): number;
282
+ /**
283
+ * Per-tool running USD totals, keyed by the name given to `settleTool()` /
284
+ * `recordTool()` — e.g. `{"apollo.people_lookup": 0.42, "exa.search": 0.11}`.
285
+ * Makes the token/tool split of the one shared ceiling inspectable
286
+ * (`spentUsd - sum of toolCosts` is the token side). Returns a snapshot copy.
287
+ */
288
+ get toolCosts(): Record<string, number>;
233
289
  /**
234
290
  * The per-call spend ledger, oldest first — one {@link SpendEvent} per priced
235
- * `record()` / `settle()` / `recordTool()`. Returns a snapshot copy: mutating
236
- * it cannot corrupt the ledger.
291
+ * `record()` / `settle()` / `recordTool()` / `settleTool()`. Returns a
292
+ * snapshot copy: mutating it cannot corrupt the ledger.
237
293
  */
238
294
  get spendLog(): SpendEvent[];
239
295
  /**
@@ -247,6 +303,25 @@ declare class BudgetGuard {
247
303
  * Python `2.5e-06`.) Empty ledger yields `""`.
248
304
  */
249
305
  exportLog(): string;
306
+ /**
307
+ * The default next-call prediction when the caller supplies no estimate.
308
+ * Conservative: the costlier of the last LLM call and the last tool call — a
309
+ * mixed loop predicts the pricier kind, which at worst blocks one call early
310
+ * (fail-closed) rather than letting a crossing call through because the LAST
311
+ * event happened to be cheap.
312
+ */
313
+ private defaultEstimate;
314
+ /**
315
+ * Subtract a settled/released hold from the in-flight tally. A handle larger
316
+ * than EVERYTHING currently held cannot have come from a matching
317
+ * `reserve()` — throwing beats silently clamping, which would free OTHER
318
+ * callers' holds and fail the ceiling open. The epsilon absorbs float dust
319
+ * from accumulating and draining many holds; per-caller over-release (a
320
+ * handle within the total but larger than the caller's own hold) is
321
+ * undetectable without per-handle tracking and remains the caller's
322
+ * responsibility.
323
+ */
324
+ private consumeReservation;
250
325
  private appendEvent;
251
326
  /**
252
327
  * Context-aware spend advisory for this budget — see {@link BudgetAdvisory}.
@@ -258,6 +333,77 @@ declare class BudgetGuard {
258
333
  advisory(): BudgetAdvisory;
259
334
  }
260
335
 
336
+ /**
337
+ * LatencyBudget — a cumulative tool-chain deadline, sibling to BudgetGuard.
338
+ *
339
+ * BudgetGuard stops an agent before its next call crosses a USD ceiling;
340
+ * LatencyBudget stops it before the next call would blow an end-user SLA:
341
+ *
342
+ * ```ts
343
+ * const deadline = new LatencyBudget(5000);
344
+ * ...
345
+ * deadline.check(800); // throws DeadlineExceeded when projected over
346
+ * if (deadline.advisory().nearDeadline) useFasterModel();
347
+ * router.pick({ maxLatencyMs: deadline.remainingMs });
348
+ * ```
349
+ *
350
+ * Design notes (mirroring `src/floe_guard/latency.py`):
351
+ * - **Monotonic clock** — `performance.now()`, never wall time.
352
+ * - **Cooperative, not preemptive** — the guard provides the deadline signal;
353
+ * aborting a stalled in-flight call is the framework's job (AbortSignal).
354
+ * `check()` prevents the NEXT call from starting.
355
+ * - **Advisory symmetry** — `nearDeadline` / `usedBps` / `remainingMs` are the
356
+ * latency twin of BudgetGuard's `nearLimit` / `usedBps` / `remainingUsd`.
357
+ * - **In-process scope** — one instance per request/run; distributed latency
358
+ * tracking is out of scope.
359
+ */
360
+ /** A context-aware deadline signal — the latency twin of {@link BudgetAdvisory}.
361
+ * Soft by design; the hard-stop is {@link LatencyBudget.check}. */
362
+ interface LatencyAdvisory {
363
+ nearDeadline: boolean;
364
+ /** SLA consumed, basis points 0..10000 (8500 = 85%). */
365
+ usedBps: number;
366
+ remainingMs: number;
367
+ slaMs: number;
368
+ elapsedMs: number;
369
+ }
370
+ interface LatencyBudgetOptions {
371
+ /**
372
+ * Utilization (basis points, 0..10000) at which {@link LatencyBudget.advisory}
373
+ * flags `nearDeadline` so an agent can downshift to a faster path before the
374
+ * wall. Default 8000 (80%), matching BudgetGuard's `nearLimitBps`.
375
+ */
376
+ nearDeadlineBps?: number;
377
+ /** Invoked with `(elapsedMs, slaMs)` right before {@link DeadlineExceeded} is thrown. */
378
+ onBlock?: (elapsedMs: number, slaMs: number) => void;
379
+ /** Milliseconds-returning monotonic clock, injectable for tests. Defaults to `performance.now`. */
380
+ clock?: () => number;
381
+ }
382
+ declare class LatencyBudget {
383
+ readonly slaMs: number;
384
+ readonly nearDeadlineBps: number;
385
+ private readonly onBlock?;
386
+ private readonly clock;
387
+ private readonly startedAt;
388
+ /** The budget starts counting at construction — build it when the request
389
+ * (and its SLA) starts. */
390
+ constructor(slaMs: number, options?: LatencyBudgetOptions);
391
+ /** Milliseconds since construction (monotonic). */
392
+ get elapsedMs(): number;
393
+ /** Milliseconds left before the SLA, floored at 0 — the readable signal a
394
+ * router uses to pick a faster fallback or truncate work mid-chain. */
395
+ get remainingMs(): number;
396
+ /**
397
+ * Throw {@link DeadlineExceeded} when the projected elapsed time (now +
398
+ * `expectedMs` for the upcoming call) would blow the SLA. Call it
399
+ * immediately before each tool/model call; pass 0 to only gate on time
400
+ * already spent.
401
+ */
402
+ check(expectedMs?: number): void;
403
+ /** The soft near-deadline signal — symmetric to `BudgetGuard.advisory()`. */
404
+ advisory(): LatencyAdvisory;
405
+ }
406
+
261
407
  /**
262
408
  * Exceptions for floe-guard.
263
409
  *
@@ -293,6 +439,11 @@ declare class UnpriceableModelError extends FloeGuardError {
293
439
  readonly model: string;
294
440
  constructor(model: string);
295
441
  }
442
+ declare class DeadlineExceeded extends FloeGuardError {
443
+ readonly elapsedMs: number;
444
+ readonly slaMs: number;
445
+ constructor(elapsedMs: number, slaMs: number);
446
+ }
296
447
 
297
448
  /**
298
449
  * Vercel AI SDK middleware that enforces a {@link BudgetGuard} in the call path.
@@ -363,4 +514,4 @@ interface BudgetGuardMiddleware {
363
514
  */
364
515
  declare function budgetGuardMiddleware(guard: BudgetGuard): BudgetGuardMiddleware;
365
516
 
366
- export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, FloeGuardError, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };
517
+ export { type BudgetAdvisory, BudgetExceeded, BudgetGuard, type BudgetGuardMiddleware, type BudgetGuardOptions, DeadlineExceeded, FloeGuardError, type LatencyAdvisory, LatencyBudget, type LatencyBudgetOptions, type ManualPrice, type PricedModel, type SpendEvent, UnpriceableModelError, budgetGuardMiddleware, priceTokens, pricing, resolvePrice };