@adhd/agent-plugin-budget 0.1.0 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.js CHANGED
@@ -1,14 +1,15 @@
1
- import { z as p } from "zod";
2
- function S(f) {
3
- const e = /^P(?:(\d+)Y)?(?:(\d+)M)?(?:(\d+)D)?(?:T(?:(\d+)H)?(?:(\d+)M)?(?:(\d+(?:\.\d+)?)S)?)?$/, t = f.match(e);
4
- if (!t)
5
- throw new Error(`invalid ISO 8601 duration: ${f}`);
6
- const [, o, s, a, n, i, c] = t;
7
- let r = 0;
8
- return o && (r += parseInt(o) * 365.25 * 864e5), s && (r += parseInt(s) * 30.44 * 864e5), a && (r += parseInt(a) * 864e5), n && (r += parseInt(n) * 36e5), i && (r += parseInt(i) * 6e4), c && (r += parseFloat(c) * 1e3), Math.round(r);
1
+ import { z as u } from "zod";
2
+ import { contextWindowFor as S } from "@adhd/agent-base-types";
3
+ function E(c) {
4
+ const t = /^P(?:(\d+)Y)?(?:(\d+)M)?(?:(\d+)D)?(?:T(?:(\d+)H)?(?:(\d+)M)?(?:(\d+(?:\.\d+)?)S)?)?$/, e = c.match(t);
5
+ if (!e)
6
+ throw new Error(`invalid ISO 8601 duration: ${c}`);
7
+ const [, o, s, n, i, r, a] = e;
8
+ let l = 0;
9
+ return o && (l += parseInt(o) * 365.25 * 864e5), s && (l += parseInt(s) * 30.44 * 864e5), n && (l += parseInt(n) * 864e5), i && (l += parseInt(i) * 36e5), r && (l += parseInt(r) * 6e4), a && (l += parseFloat(a) * 1e3), Math.round(l);
9
10
  }
10
- const E = [
11
- "tokens",
11
+ const M = [
12
+ "context",
12
13
  "inputTokens",
13
14
  "outputTokens",
14
15
  "calls",
@@ -16,268 +17,417 @@ const E = [
16
17
  "modelMs",
17
18
  "cost",
18
19
  "toolCalls",
20
+ "errors",
21
+ "consecutiveErrors",
19
22
  "responseSize"
20
- ], M = p.object({
21
- field: p.enum(E),
22
- maximum: p.number().min(0),
23
- window: p.string().optional(),
24
- scope: p.enum(["task", "session", "agent", "global"]).optional(),
25
- mode: p.enum(["warning", "block"]).optional(),
26
- message: p.string().optional()
27
- }), g = p.object({
28
- caps: p.array(M).optional(),
29
- mode: p.enum(["warning", "block"]).optional(),
30
- costPerInputToken: p.number().min(0).optional(),
31
- costPerOutputToken: p.number().min(0).optional(),
32
- scope: p.enum(["task", "session", "agent", "global"]).optional()
33
- }), w = p.object({
23
+ ], b = /* @__PURE__ */ new Set(["toolCalls", "errors", "consecutiveErrors"]), y = u.object({
24
+ field: u.enum(M),
25
+ maximum: u.number().min(0).optional(),
26
+ contextWindowFraction: u.number().min(0).max(1).optional(),
27
+ window: u.string().optional(),
28
+ scope: u.enum(["task", "session", "agent", "global"]).optional(),
29
+ mode: u.enum(["warning", "block"]).optional(),
30
+ message: u.string().optional()
31
+ }).superRefine((c, t) => {
32
+ if (c.field === "context") {
33
+ const e = c.maximum !== void 0, o = c.contextWindowFraction !== void 0;
34
+ e === o && t.addIssue({
35
+ code: u.ZodIssueCode.custom,
36
+ path: ["contextWindowFraction"],
37
+ message: "cap field 'context' requires exactly one of 'maximum' or 'contextWindowFraction'"
38
+ }), c.window !== void 0 && t.addIssue({
39
+ code: u.ZodIssueCode.custom,
40
+ path: ["window"],
41
+ message: "cap field 'context' rejects 'window' — a windowed peak is meaningless; windowed cumulative volume is expressible via 'inputTokens'/'outputTokens' + 'window'"
42
+ });
43
+ } else
44
+ (c.field === "errors" || c.field === "consecutiveErrors") && (c.window !== void 0 && t.addIssue({
45
+ code: u.ZodIssueCode.custom,
46
+ path: ["window"],
47
+ message: `cap field '${c.field}' rejects 'window' — windowed error budgets are not expressible this wave`
48
+ }), c.scope !== void 0 && c.scope !== "task" && t.addIssue({
49
+ code: u.ZodIssueCode.custom,
50
+ path: ["scope"],
51
+ message: `cap field '${c.field}' is counted per-task in memory (no task_usage column) — only 'task' scope is expressible this wave`
52
+ })), c.window !== void 0 && (c.scope === void 0 || c.scope === "task") && t.addIssue({
53
+ code: u.ZodIssueCode.custom,
54
+ path: ["scope"],
55
+ message: `cap field '${c.field}' carries 'window' but resolves to 'task' scope, where windowed caps are a silent no-op (no task-level window query); set an explicit 'scope': 'session' | 'agent' | 'global'`
56
+ }), c.contextWindowFraction !== void 0 && t.addIssue({
57
+ code: u.ZodIssueCode.custom,
58
+ path: ["contextWindowFraction"],
59
+ message: `'contextWindowFraction' is only valid on cap field 'context' (got '${c.field}')`
60
+ }), c.maximum === void 0 && t.addIssue({
61
+ code: u.ZodIssueCode.custom,
62
+ path: ["maximum"],
63
+ message: `cap field '${c.field}' requires a 'maximum'`
64
+ });
65
+ }), g = u.object({
66
+ caps: u.array(y).optional(),
67
+ mode: u.enum(["warning", "block"]).optional(),
68
+ costPerInputToken: u.number().min(0).optional(),
69
+ costPerOutputToken: u.number().min(0).optional(),
70
+ // Cache-weighted cost rates (BUG-AGENTMCP-008). Default = costPerInputToken when
71
+ // unset, so a config without them bills every input token at the flat input rate —
72
+ // byte-for-byte identical to the pre-fix behavior.
73
+ costPerCacheReadToken: u.number().min(0).optional(),
74
+ costPerCacheWriteToken: u.number().min(0).optional(),
75
+ scope: u.enum(["task", "session", "agent", "global"]).optional()
76
+ }), x = g.partial().superRefine((c, t) => {
77
+ const e = c.caps ?? [];
78
+ for (const [o, s] of e.entries())
79
+ s.field === "context" && t.addIssue({
80
+ code: u.ZodIssueCode.custom,
81
+ path: ["caps", o, "field"],
82
+ message: "cap field 'context' is only valid at model scope — a tool-scoped 'context' cap is a silent no-op (enforced on the model path, invisible to tool overrides); place it in 'defaults'/'agent'/'provider' instead"
83
+ });
84
+ }), _ = u.object({
34
85
  defaults: g.optional(),
35
- agent: p.object({
86
+ agent: u.object({
36
87
  default: g.optional(),
37
- overrides: p.record(p.string(), g.partial()).optional().default({})
88
+ overrides: u.record(u.string(), g.partial()).optional().default({})
38
89
  }).optional(),
39
- provider: p.object({
90
+ provider: u.object({
40
91
  default: g.optional(),
41
- overrides: p.record(p.string(), g.partial()).optional().default({})
92
+ overrides: u.record(u.string(), g.partial()).optional().default({})
42
93
  }).optional(),
43
- tool: p.object({
44
- default: g.optional(),
45
- overrides: p.record(p.string(), g.partial()).optional().default({})
94
+ tool: u.object({
95
+ default: x.optional(),
96
+ overrides: u.record(u.string(), x).optional().default({})
46
97
  }).optional()
47
- }), y = p.object({}).passthrough(), x = {
98
+ }), N = u.object({}).passthrough(), A = {
48
99
  maxInputTokens: { field: "inputTokens" },
49
100
  maxOutputTokens: { field: "outputTokens" },
50
- maxTotalTokens: { field: "tokens" },
51
101
  maxModelCalls: { field: "calls" },
52
102
  maxWallClockMs: { field: "wallClock" },
53
103
  maxModelMs: { field: "modelMs" },
54
104
  maxCostUSD: { field: "cost" },
55
- maxTokensPer24h: { field: "tokens", window: "PT24H" },
105
+ maxTokensPer24h: { field: "inputTokens", window: "PT24H" },
56
106
  maxCalls: { field: "toolCalls" }
57
107
  };
58
- function b(f) {
59
- const e = [], t = {};
60
- for (const [o, s] of Object.entries(f)) {
61
- const a = x[o];
62
- if (a && typeof s == "number") {
63
- const n = { field: a.field, maximum: s };
64
- a.window && (n.window = a.window), e.push(n);
108
+ function O(c) {
109
+ const t = [], e = {};
110
+ for (const [n, i] of Object.entries(c)) {
111
+ const r = A[n];
112
+ if (r && typeof i == "number") {
113
+ const a = { field: r.field, maximum: i };
114
+ r.window && (a.window = r.window), t.push(a);
65
115
  } else
66
- (o === "scope" || o === "mode" || o === "costPerInputToken" || o === "costPerOutputToken" || o === "message") && (t[o] = s);
116
+ (n === "scope" || n === "mode" || n === "costPerInputToken" || n === "costPerOutputToken" || n === "costPerCacheReadToken" || n === "costPerCacheWriteToken" || n === "message") && (e[n] = i);
67
117
  }
68
- return e.length > 0 && (t.caps = e), g.parse(t);
118
+ t.length > 0 && (e.caps = t);
119
+ const o = e.scope, s = o === "task" || o === "session" || o === "agent" || o === "global" ? o : void 0;
120
+ for (const n of t)
121
+ n.window !== void 0 && n.scope === void 0 && (n.scope = s ?? "global");
122
+ return g.parse(e);
69
123
  }
70
- function _(f) {
71
- const e = f;
72
- if (e.defaults !== void 0 || e.agent !== void 0 || e.provider !== void 0 || e.tool !== void 0) {
73
- const t = w.parse(f);
124
+ const w = (c) => `legacy 'tokens' budget cap detected (${c}): cap field 'tokens' is removed; use 'context' (peak request input) or 'inputTokens'/'outputTokens' (windowed volume)`;
125
+ function P(c) {
126
+ const t = c ?? {};
127
+ if (typeof t.maxTotalTokens == "number")
128
+ throw new Error(w("flat 'maxTotalTokens'"));
129
+ const e = (s) => {
130
+ if (typeof s != "object" || s === null)
131
+ return;
132
+ const n = s;
133
+ if (typeof n.maxTotalTokens == "number")
134
+ throw new Error(w("structured 'maxTotalTokens'"));
135
+ const i = n.caps;
136
+ if (Array.isArray(i)) {
137
+ for (const r of i)
138
+ if (typeof r == "object" && r !== null && r.field === "tokens")
139
+ throw new Error(w("caps[].field === 'tokens'"));
140
+ }
141
+ }, o = (s) => {
142
+ if (typeof s != "object" || s === null)
143
+ return;
144
+ const n = s;
145
+ e(n), "default" in n && e(n.default);
146
+ const i = n.overrides;
147
+ if (typeof i == "object" && i !== null)
148
+ for (const r of Object.values(i))
149
+ e(r);
150
+ };
151
+ e(t.defaults), o(t.agent), o(t.provider), o(t.tool);
152
+ }
153
+ function I(c) {
154
+ P(c);
155
+ const t = c;
156
+ if (t.defaults !== void 0 || t.agent !== void 0 || t.provider !== void 0 || t.tool !== void 0) {
157
+ const e = _.parse(c);
74
158
  return {
75
- defaults: t.defaults ?? g.parse({}),
76
- agent: t.agent ?? { overrides: {} },
77
- provider: t.provider ?? { overrides: {} },
78
- tool: t.tool ?? { overrides: {} }
159
+ defaults: e.defaults ?? g.parse({}),
160
+ agent: e.agent ?? { overrides: {} },
161
+ provider: e.provider ?? { overrides: {} },
162
+ tool: e.tool ?? { overrides: {} }
79
163
  };
80
164
  }
81
165
  return {
82
- defaults: b(e),
166
+ defaults: O(t),
83
167
  agent: { overrides: {} },
84
168
  provider: { overrides: {} },
85
169
  tool: { overrides: {} }
86
170
  };
87
171
  }
88
- function T(f, e, t, o) {
172
+ function v(c, t, e, o) {
89
173
  return {
90
174
  isEnforcementError: !0,
91
175
  code: "BUDGET_EXCEEDED",
92
- message: o ?? `${f} limit is ${e}, current value is ${Math.round(t)}`
176
+ message: o ?? `${c} limit is ${t}, current value is ${Math.round(e)}`
93
177
  };
94
178
  }
95
- function v(f, e, t) {
96
- return { isToolWarning: !0, toolName: f, callId: e, message: t };
179
+ function $(c, t, e) {
180
+ return { isToolWarning: !0, toolName: c, callId: t, message: e };
181
+ }
182
+ function R(c, t) {
183
+ var o;
184
+ let e = 0;
185
+ for (const s of c) {
186
+ e += ((o = s.content) == null ? void 0 : o.length) ?? 0;
187
+ for (const n of s.toolCalls ?? [])
188
+ e += JSON.stringify(n.arguments ?? {}).length;
189
+ for (const n of s.toolResults ?? [])
190
+ e += JSON.stringify(n.result ?? null).length;
191
+ }
192
+ for (const s of t)
193
+ e += JSON.stringify({
194
+ name: s.name,
195
+ description: s.description,
196
+ inputSchema: s.inputSchema
197
+ }).length;
198
+ return Math.ceil(e / 4);
97
199
  }
98
- class O {
99
- constructor(e, t, o = 0, s = 0) {
100
- this.db = e, this.cfg = t, this.costPerInput = o, this.costPerOutput = s, this.name = "agent-mcp-budget", this.accumulators = /* @__PURE__ */ new Map();
200
+ class L {
201
+ constructor(t, e, o = 0, s = 0, n = 0, i = 0) {
202
+ this.db = t, this.cfg = e, this.costPerInput = o, this.costPerOutput = s, this.costPerCacheRead = n, this.costPerCacheWrite = i, this.name = "agent-mcp-budget", this.accumulators = /* @__PURE__ */ new Map();
101
203
  }
102
- install(e) {
103
- e.register("task:start", (t) => {
204
+ install(t) {
205
+ this.hooks = t, t.register("task:start", (e) => {
206
+ try {
207
+ this.onTaskStart(e);
208
+ } catch {
209
+ }
210
+ }), t.register("pre:model_request", (e) => {
104
211
  try {
105
- this.onTaskStart(t);
212
+ this.onPreModelRequest(e);
106
213
  } catch {
107
214
  }
108
- }), e.register("pre:model_request", (t) => {
215
+ }), t.register("post:model_response", (e) => {
109
216
  try {
110
- this.onPreModelRequest(t);
217
+ this.onPostModelResponse(e);
111
218
  } catch {
112
219
  }
113
- }), e.register("post:model_response", (t) => {
220
+ }), t.register("post:tool_call", (e) => {
114
221
  try {
115
- this.onPostModelResponse(t);
222
+ this.onPostToolCall(e);
116
223
  } catch {
117
224
  }
118
- }), e.register("task:completed", (t) => {
225
+ }), t.register("task:completed", (e) => {
119
226
  try {
120
- this.onTerminal(t.executionContext.taskId);
227
+ this.onTerminal(e.executionContext.taskId);
121
228
  } catch {
122
229
  }
123
- }), e.register("task:failed", (t) => {
230
+ }), t.register("task:failed", (e) => {
124
231
  try {
125
- this.onTerminal(t.executionContext.taskId);
232
+ this.onTerminal(e.executionContext.taskId);
126
233
  } catch {
127
234
  }
128
- }), e.register("task:cancelled", (t) => {
235
+ }), t.register("task:cancelled", (e) => {
129
236
  try {
130
- this.onTerminal(t.executionContext.taskId);
237
+ this.onTerminal(e.executionContext.taskId);
131
238
  } catch {
132
239
  }
133
- }), e.registerEnforcement(
240
+ }), t.registerEnforcement(
134
241
  "pre:model_request",
135
- (t) => this.enforcePreModel(t)
136
- ), e.registerEnforcement("pre:tool_call", (t) => this.enforcePreTool(t)), e.register("transform:tool_result", (t) => {
242
+ (e) => this.enforcePreModel(e)
243
+ ), t.registerEnforcement("pre:tool_call", (e) => this.enforcePreTool(e)), t.register("transform:tool_result", (e) => {
137
244
  try {
138
- this.enforceResponseSize(t);
245
+ this.enforceResponseSize(e);
139
246
  } catch {
140
247
  }
141
248
  });
142
249
  }
143
250
  // ── Observational handlers ────────────────────────────────────────────────
144
- onTaskStart(e) {
145
- var n, i;
146
- const { taskId: t, sessionId: o, agentName: s } = e.executionContext, a = ((i = (n = e.executionContext.agentDefinition) == null ? void 0 : n.provider) == null ? void 0 : i.type) ?? "unknown";
147
- this.accumulators.set(t, {
148
- taskId: t,
251
+ onTaskStart(t) {
252
+ var i, r;
253
+ const { taskId: e, sessionId: o, agentName: s } = t.executionContext, n = ((r = (i = t.executionContext.agentDefinition) == null ? void 0 : i.provider) == null ? void 0 : r.type) ?? "unknown";
254
+ this.accumulators.set(e, {
255
+ taskId: e,
149
256
  sessionId: o ?? void 0,
150
257
  agentName: s,
151
- providerType: a,
258
+ providerType: n,
152
259
  startedAtMs: Date.now(),
260
+ uncachedInputTokens: 0,
261
+ cacheReadTokens: 0,
262
+ cacheCreationTokens: 0,
153
263
  inputTokens: 0,
154
264
  outputTokens: 0,
265
+ peakContextTokens: 0,
155
266
  modelCalls: 0,
156
267
  totalModelMs: 0,
157
- toolCalls: /* @__PURE__ */ new Map()
268
+ toolCalls: /* @__PURE__ */ new Map(),
269
+ errors: 0,
270
+ consecutiveErrors: 0
158
271
  });
159
272
  }
160
- onPreModelRequest(e) {
161
- const t = this.accumulators.get(e.executionContext.taskId);
162
- t && (t.modelCallStartMs = Date.now());
273
+ onPreModelRequest(t) {
274
+ const e = this.accumulators.get(t.executionContext.taskId);
275
+ e && (e.modelCallStartMs = Date.now());
163
276
  }
164
- onPostModelResponse(e) {
165
- const t = this.accumulators.get(e.executionContext.taskId);
166
- if (!t)
277
+ onPostModelResponse(t) {
278
+ const e = this.accumulators.get(t.executionContext.taskId);
279
+ if (!e)
167
280
  return;
168
- const o = e.tokenUsage;
169
- o && (t.inputTokens += o.inputTokens ?? 0, t.outputTokens += o.outputTokens ?? 0), t.modelCalls += 1, t.modelCallStartMs !== void 0 && (t.totalModelMs += Date.now() - t.modelCallStartMs, t.modelCallStartMs = void 0);
281
+ const o = t.tokenUsage;
282
+ if (o) {
283
+ const s = o.inputTokens ?? 0, n = o.cacheReadTokens ?? 0, i = o.cacheCreationTokens ?? 0, a = o.uncachedInputTokens !== void 0 || o.cacheReadTokens !== void 0 || o.cacheCreationTokens !== void 0 ? o.uncachedInputTokens ?? Math.max(0, s - n - i) : s;
284
+ e.uncachedInputTokens += a, e.cacheReadTokens += n, e.cacheCreationTokens += i, e.inputTokens += a + n + i, e.outputTokens += o.outputTokens ?? 0, e.peakContextTokens = Math.max(e.peakContextTokens, s);
285
+ }
286
+ e.modelCalls += 1, e.modelCallStartMs !== void 0 && (e.totalModelMs += Date.now() - e.modelCallStartMs, e.modelCallStartMs = void 0);
287
+ }
288
+ onTerminal(t) {
289
+ this.accumulators.delete(t);
170
290
  }
171
- onTerminal(e) {
172
- this.accumulators.delete(e);
291
+ /**
292
+ * Packet C — error counters (plan §3): the ONLY place the error budget
293
+ * changes. Fires at post:tool_call, which the orchestrator emits exclusively
294
+ * for EXECUTED tools (orchestrator.ts Phase 2 map, post:tool_call emit). A
295
+ * call soft-blocked by an IToolWarning at pre:tool_call is injected as a
296
+ * warningResult and SKIPPED (orchestrator.ts:619-626, filter at 645-647) —
297
+ * it never reaches Phase 2 and never emits post:tool_call. That is the
298
+ * non-self-amplifying guarantee: a warning-mode errors cap's own warnings are
299
+ * never counted, so the counter cannot feed itself.
300
+ *
301
+ * Thrown-errors-only (owner lean, plan §3): `isError` is true at this
302
+ * boundary only when the tool's callTool THREW (orchestrator.ts:718);
303
+ * error-shaped non-throwing results are indistinguishable from success here.
304
+ */
305
+ onPostToolCall(t) {
306
+ const e = this.accumulators.get(t.executionContext.taskId);
307
+ e && (t.isError ? (e.errors += 1, e.consecutiveErrors += 1) : e.consecutiveErrors = 0);
173
308
  }
174
309
  // ── Config resolution ─────────────────────────────────────────────────────
175
- mergeDim(e) {
176
- let t = { caps: [] };
177
- for (const o of e)
178
- o && (t = {
179
- caps: [...t.caps ?? [], ...o.caps ?? []],
180
- mode: o.mode ?? t.mode,
181
- costPerInputToken: o.costPerInputToken ?? t.costPerInputToken,
182
- costPerOutputToken: o.costPerOutputToken ?? t.costPerOutputToken,
183
- scope: o.scope ?? t.scope
310
+ mergeDim(t) {
311
+ let e = { caps: [] };
312
+ for (const o of t)
313
+ o && (e = {
314
+ caps: [...e.caps ?? [], ...o.caps ?? []],
315
+ mode: o.mode ?? e.mode,
316
+ costPerInputToken: o.costPerInputToken ?? e.costPerInputToken,
317
+ costPerOutputToken: o.costPerOutputToken ?? e.costPerOutputToken,
318
+ costPerCacheReadToken: o.costPerCacheReadToken ?? e.costPerCacheReadToken,
319
+ costPerCacheWriteToken: o.costPerCacheWriteToken ?? e.costPerCacheWriteToken,
320
+ scope: o.scope ?? e.scope
184
321
  });
185
- return t;
322
+ return e;
186
323
  }
187
- resolveCaps(e, t, o) {
188
- var u, m, d;
189
- const s = this.cfg.defaults, a = this.cfg.agent, n = this.cfg.provider, i = this.cfg.tool;
324
+ resolveCaps(t, e, o) {
325
+ var f, d, k;
326
+ const s = this.cfg.defaults, n = this.cfg.agent, i = this.cfg.provider, r = this.cfg.tool;
190
327
  if (o) {
191
- const k = (u = i == null ? void 0 : i.overrides) == null ? void 0 : u[o], h = this.mergeDim([s, i == null ? void 0 : i.default, k]);
328
+ const m = (f = r == null ? void 0 : r.overrides) == null ? void 0 : f[o], h = this.mergeDim([s, r == null ? void 0 : r.default, m]);
192
329
  return {
193
330
  caps: h.caps ?? [],
194
331
  mode: h.mode,
195
332
  scope: h.scope
196
333
  };
197
334
  }
198
- const c = (m = a == null ? void 0 : a.overrides) == null ? void 0 : m[e], r = (d = n == null ? void 0 : n.overrides) == null ? void 0 : d[t], l = this.mergeDim([
335
+ const a = (d = n == null ? void 0 : n.overrides) == null ? void 0 : d[t], l = (k = i == null ? void 0 : i.overrides) == null ? void 0 : k[e], p = this.mergeDim([
199
336
  s,
200
- a == null ? void 0 : a.default,
201
- c,
202
337
  n == null ? void 0 : n.default,
203
- r
338
+ a,
339
+ i == null ? void 0 : i.default,
340
+ l
204
341
  ]);
205
- return { caps: l.caps ?? [], mode: l.mode, scope: l.scope };
342
+ return { caps: p.caps ?? [], mode: p.mode, scope: p.scope };
206
343
  }
207
344
  // ── Scope-aware DB queries ────────────────────────────────────────────────
208
- queryScopeTotals(e, t, o, s) {
209
- const a = this.accumulators.get(e), n = a ? {
210
- inputTokens: a.inputTokens,
211
- outputTokens: a.outputTokens,
212
- modelCalls: a.modelCalls
213
- } : { inputTokens: 0, outputTokens: 0, modelCalls: 0 };
345
+ queryScopeTotals(t, e, o, s) {
346
+ const n = this.accumulators.get(t), i = n ? {
347
+ inputTokens: n.inputTokens,
348
+ outputTokens: n.outputTokens,
349
+ modelCalls: n.modelCalls,
350
+ peakContextTokens: n.peakContextTokens
351
+ } : {
352
+ inputTokens: 0,
353
+ outputTokens: 0,
354
+ modelCalls: 0,
355
+ peakContextTokens: 0
356
+ };
214
357
  if (s === "task" || !this.db)
215
- return n;
358
+ return i;
216
359
  try {
217
- const i = this.db;
218
- let c;
219
- if (s === "session" && t ? c = i.prepare(
360
+ const r = this.db;
361
+ let a;
362
+ if (s === "session" && e ? a = r.prepare(
220
363
  `SELECT
221
364
  COALESCE(SUM(tu.input_tokens), 0) AS input,
222
365
  COALESCE(SUM(tu.output_tokens), 0) AS output,
223
- COALESCE(SUM(tu.model_calls), 0) AS calls
366
+ COALESCE(SUM(tu.model_calls), 0) AS calls,
367
+ COALESCE(MAX(tu.peak_context_tokens), 0) AS peak
224
368
  FROM task_usage tu
225
369
  JOIN tasks t ON tu.task_id = t.id
226
370
  WHERE t.session_id = ? AND tu.task_id != ?`
227
- ).get(t, e) : s === "agent" ? c = i.prepare(
371
+ ).get(e, t) : s === "agent" ? a = r.prepare(
228
372
  `SELECT
229
373
  COALESCE(SUM(input_tokens), 0) AS input,
230
374
  COALESCE(SUM(output_tokens), 0) AS output,
231
- COALESCE(SUM(model_calls), 0) AS calls
375
+ COALESCE(SUM(model_calls), 0) AS calls,
376
+ COALESCE(MAX(peak_context_tokens), 0) AS peak
232
377
  FROM task_usage
233
378
  WHERE agent_name = ? AND task_id != ?`
234
- ).get(o, e) : s === "global" && (c = i.prepare(
379
+ ).get(o, t) : s === "global" && (a = r.prepare(
235
380
  `SELECT
236
381
  COALESCE(SUM(input_tokens), 0) AS input,
237
382
  COALESCE(SUM(output_tokens), 0) AS output,
238
- COALESCE(SUM(model_calls), 0) AS calls
383
+ COALESCE(SUM(model_calls), 0) AS calls,
384
+ COALESCE(MAX(peak_context_tokens), 0) AS peak
239
385
  FROM task_usage
240
386
  WHERE task_id != ?`
241
- ).get(e)), c)
387
+ ).get(t)), a)
242
388
  return {
243
- inputTokens: (c.input ?? 0) + n.inputTokens,
244
- outputTokens: (c.output ?? 0) + n.outputTokens,
245
- modelCalls: (c.calls ?? 0) + n.modelCalls
389
+ inputTokens: (a.input ?? 0) + i.inputTokens,
390
+ outputTokens: (a.output ?? 0) + i.outputTokens,
391
+ modelCalls: (a.calls ?? 0) + i.modelCalls,
392
+ // Scoped context = MAX over the scope's task set of peak_context_tokens,
393
+ // with the current task's in-memory peak folded in (it is a member of the
394
+ // set; the queries above exclude it by `task_id != ?`).
395
+ peakContextTokens: Math.max(a.peak ?? 0, i.peakContextTokens)
246
396
  };
247
397
  } catch {
248
398
  }
249
- return n;
399
+ return i;
250
400
  }
251
- queryWindowTokens(e, t, o, s) {
401
+ queryWindowTokens(t, e, o, s) {
252
402
  if (!this.db)
253
403
  return 0;
254
404
  try {
255
- const a = this.db, n = new Date(Date.now() - o).toISOString();
256
- let i;
257
- if (e === "session") {
258
- const c = s ? " AND tu.task_id != ?" : "", r = [t, n];
259
- s && r.push(s), i = a.prepare(
405
+ const n = this.db, i = new Date(Date.now() - o).toISOString();
406
+ let r;
407
+ if (t === "session") {
408
+ const a = s ? " AND tu.task_id != ?" : "", l = [e, i];
409
+ s && l.push(s), r = n.prepare(
260
410
  `SELECT COALESCE(SUM(tu.input_tokens + tu.output_tokens), 0) AS total
261
411
  FROM task_usage tu
262
412
  JOIN tasks t ON tu.task_id = t.id
263
- WHERE t.session_id = ? AND tu.created_at >= ?${c}`
264
- ).get(...r);
265
- } else if (e === "agent") {
266
- const c = s ? " AND task_id != ?" : "", r = [t, n];
267
- s && r.push(s), i = a.prepare(
413
+ WHERE t.session_id = ? AND tu.created_at >= ?${a}`
414
+ ).get(...l);
415
+ } else if (t === "agent") {
416
+ const a = s ? " AND task_id != ?" : "", l = [e, i];
417
+ s && l.push(s), r = n.prepare(
268
418
  `SELECT COALESCE(SUM(input_tokens + output_tokens), 0) AS total
269
419
  FROM task_usage
270
- WHERE agent_name = ? AND created_at >= ?${c}`
271
- ).get(...r);
272
- } else if (e === "global") {
273
- const c = s ? " AND task_id != ?" : "", r = [n];
274
- s && r.push(s), i = a.prepare(
420
+ WHERE agent_name = ? AND created_at >= ?${a}`
421
+ ).get(...l);
422
+ } else if (t === "global") {
423
+ const a = s ? " AND task_id != ?" : "", l = [i];
424
+ s && l.push(s), r = n.prepare(
275
425
  `SELECT COALESCE(SUM(input_tokens + output_tokens), 0) AS total
276
426
  FROM task_usage
277
- WHERE created_at >= ?${c}`
278
- ).get(...r);
427
+ WHERE created_at >= ?${a}`
428
+ ).get(...l);
279
429
  }
280
- return (i == null ? void 0 : i.total) ?? 0;
430
+ return (r == null ? void 0 : r.total) ?? 0;
281
431
  } catch {
282
432
  return 0;
283
433
  }
@@ -292,189 +442,275 @@ class O {
292
442
  * W = number of unique (scope, window) pairs across all caps
293
443
  *
294
444
  * Independent of cap count, agent count, session count, or history depth.
445
+ *
446
+ * `requestEstimate` (tokens) is the tools-aware estimate of the PENDING
447
+ * request, computed on the model path; it feeds the 'context' enforcement
448
+ * value = max(provider-reported peak, estimate) per owner ruling 4.
295
449
  */
296
- buildSnapshot(e, t, o, s, a, n) {
297
- const i = {};
298
- i.inputTokens = t.inputTokens, i.outputTokens = t.outputTokens, i.calls = t.modelCalls, i.wallClock = Date.now() - t.startedAtMs, i.modelMs = t.totalModelMs, i.cost = t.inputTokens * this.costPerInput + t.outputTokens * this.costPerOutput;
299
- const c = /* @__PURE__ */ new Set(), r = /* @__PURE__ */ new Map();
300
- for (const l of e) {
301
- const u = l.scope ?? n ?? "task";
302
- if (u !== "task" && c.add(u), l.window) {
303
- const m = `${u}:${l.window}`;
304
- r.has(m) || r.set(m, {
305
- scope: u,
306
- windowMs: S(l.window)
450
+ buildSnapshot(t, e, o, s, n, i, r = 0) {
451
+ const a = {};
452
+ a.inputTokens = e.inputTokens, a.outputTokens = e.outputTokens, a.calls = e.modelCalls, a.wallClock = Date.now() - e.startedAtMs, a.modelMs = e.totalModelMs, a.context = Math.max(e.peakContextTokens, r), a.errors = e.errors, a.consecutiveErrors = e.consecutiveErrors, a.cost = e.uncachedInputTokens * this.costPerInput + e.cacheReadTokens * this.costPerCacheRead + e.cacheCreationTokens * this.costPerCacheWrite + e.outputTokens * this.costPerOutput;
453
+ const l = /* @__PURE__ */ new Set(), p = /* @__PURE__ */ new Map();
454
+ for (const f of t) {
455
+ const d = f.scope ?? i ?? "task";
456
+ if (d !== "task" && l.add(d), f.window) {
457
+ const k = `${d}:${f.window}`;
458
+ p.has(k) || p.set(k, {
459
+ scope: d,
460
+ windowMs: E(f.window)
307
461
  });
308
462
  }
309
463
  }
310
- for (const l of c) {
311
- const u = this.queryScopeTotals(o, s, a, l);
312
- i[`${l}:inputTokens`] = u.inputTokens, i[`${l}:outputTokens`] = u.outputTokens, i[`${l}:calls`] = u.modelCalls;
464
+ for (const f of l) {
465
+ const d = this.queryScopeTotals(o, s, n, f);
466
+ a[`${f}:inputTokens`] = d.inputTokens, a[`${f}:outputTokens`] = d.outputTokens, a[`${f}:calls`] = d.modelCalls, a[`${f}:context`] = Math.max(
467
+ d.peakContextTokens,
468
+ r
469
+ );
313
470
  }
314
- for (const [l, { scope: u, windowMs: m }] of r) {
315
- let d = "";
316
- u === "session" ? d = s ?? "" : u === "agent" && (d = a ?? ""), i[l] = this.queryWindowTokens(u, d, m, o);
471
+ for (const [f, { scope: d, windowMs: k }] of p) {
472
+ let m = "";
473
+ d === "session" ? m = s ?? "" : d === "agent" && (m = n ?? ""), a[f] = this.queryWindowTokens(d, m, k, o);
317
474
  }
318
- return i;
475
+ return a;
319
476
  }
320
- getSnapshotValue(e, t, o) {
321
- const s = t.scope ?? o ?? "task";
322
- let a;
323
- t.window ? a = "" : a = s !== "task" ? `${s}:` : "";
477
+ getSnapshotValue(t, e, o) {
478
+ const s = e.scope ?? o ?? "task";
324
479
  let n;
325
- switch (t.field) {
480
+ e.window ? n = "" : n = s !== "task" ? `${s}:` : "";
481
+ let i;
482
+ switch (e.field) {
326
483
  case "inputTokens":
327
- n = e[`${a}inputTokens`] ?? e.inputTokens;
484
+ i = t[`${n}inputTokens`] ?? t.inputTokens;
328
485
  break;
329
486
  case "outputTokens":
330
- n = e[`${a}outputTokens`] ?? e.outputTokens;
487
+ i = t[`${n}outputTokens`] ?? t.outputTokens;
331
488
  break;
332
- case "tokens":
333
- n = (e[`${a}inputTokens`] ?? e.inputTokens) + (e[`${a}outputTokens`] ?? e.outputTokens);
489
+ case "context":
490
+ i = t[`${n}context`] ?? t.context;
334
491
  break;
335
492
  case "calls":
336
- n = e[`${a}calls`] ?? e.calls;
493
+ i = t[`${n}calls`] ?? t.calls;
337
494
  break;
338
495
  case "wallClock":
339
- n = e.wallClock;
496
+ i = t.wallClock;
340
497
  break;
341
498
  case "modelMs":
342
- n = e.modelMs;
499
+ i = t.modelMs;
343
500
  break;
344
501
  case "cost":
345
- n = e.cost;
502
+ i = t.cost;
346
503
  break;
347
504
  case "toolCalls":
348
- n = 0;
505
+ i = 0;
506
+ break;
507
+ case "errors":
508
+ i = t.errors;
509
+ break;
510
+ case "consecutiveErrors":
511
+ i = t.consecutiveErrors;
349
512
  break;
350
513
  default:
351
- n = 0;
514
+ i = 0;
352
515
  }
353
- return t.window && (n += e[`${s}:${t.window}`] ?? 0), n;
516
+ return e.window && (i += t[`${s}:${e.window}`] ?? 0), i;
354
517
  }
355
- evaluateCap(e, t, o) {
356
- const s = this.getSnapshotValue(t, e, o);
357
- if (s >= e.maximum)
358
- throw T(e.field, e.maximum, s, e.message);
518
+ /**
519
+ * Emit a `budget:*` notification event for an exceeded cap. Pure notification
520
+ * layer (owner ruling 7): HookRegistry.emit swallows handler errors and is a
521
+ * no-op when no handler is registered, so emission can never affect
522
+ * enforcement. `message` defaults to cap.message, then the standard
523
+ * `"<field> limit is <limit>, current value is <current>"` string — the
524
+ * same resolution `makeEnforcementError` uses, so the block event's message
525
+ * always matches the error the orchestrator sees.
526
+ *
527
+ * `limit` is the RESOLVED cap limit (`maximum`, or the context-window-derived
528
+ * limit for `contextWindowFraction` caps) — the payload must report the real
529
+ * number the cap trips at, not an undefined `maximum`.
530
+ */
531
+ async emitBudgetEvent(t, e, o, s, n, i) {
532
+ this.hooks && await this.hooks.emit(t, {
533
+ executionContext: e,
534
+ field: o.field,
535
+ maximum: n,
536
+ current: s,
537
+ message: i ?? o.message ?? `${o.field} limit is ${n}, current value is ${Math.round(
538
+ s
539
+ )}`
540
+ });
541
+ }
542
+ /**
543
+ * Resolve a cap's numeric limit. A `context` cap configured via
544
+ * `contextWindowFraction` has no `maximum`: the limit is
545
+ * `contextWindowFor(modelId) * fraction` (128K fallback for unknown models).
546
+ * Every other cap carries an explicit `maximum` (schema-enforced).
547
+ */
548
+ resolveCapLimit(t, e) {
549
+ if (t.maximum !== void 0)
550
+ return t.maximum;
551
+ const o = e.agentDefinition.provider;
552
+ return Math.floor(
553
+ S(o.model) * (t.contextWindowFraction ?? 0)
554
+ );
555
+ }
556
+ /**
557
+ * Evaluate a single cap against the snapshot. Mode resolution matches the
558
+ * tool path (`cap.mode ?? dimMode ?? 'warning'`, owner ruling 5): warning
559
+ * (the default) emits `budget:warning` and resolves — the run continues;
560
+ * block emits `budget:block` BEFORE throwing the enforcement error.
561
+ */
562
+ async evaluateCap(t, e, o, s, n) {
563
+ const i = this.resolveCapLimit(t, o), r = this.getSnapshotValue(e, t, n);
564
+ if (r < i)
565
+ return;
566
+ if ((t.mode ?? s ?? "warning") === "warning") {
567
+ await this.emitBudgetEvent("budget:warning", o, t, r, i);
568
+ return;
569
+ }
570
+ throw await this.emitBudgetEvent("budget:block", o, t, r, i), v(t.field, i, r, t.message);
359
571
  }
360
572
  // ── Enforcement: pre:model_request ────────────────────────────────────────
361
- enforcePreModel(e) {
362
- var u, m;
363
- const { taskId: t, sessionId: o, agentName: s } = e.executionContext, a = ((m = (u = e.executionContext.agentDefinition) == null ? void 0 : u.provider) == null ? void 0 : m.type) ?? "unknown", n = this.accumulators.get(t);
364
- if (!n)
573
+ async enforcePreModel(t) {
574
+ var d, k;
575
+ const { taskId: e, sessionId: o, agentName: s } = t.executionContext, n = ((k = (d = t.executionContext.agentDefinition) == null ? void 0 : d.provider) == null ? void 0 : k.type) ?? "unknown", i = this.accumulators.get(e);
576
+ if (!i)
365
577
  return;
366
- const { caps: i, scope: c } = this.resolveCaps(s, a), r = i.filter((d) => d.field !== "toolCalls");
367
- if (r.length === 0)
578
+ const { caps: r, mode: a, scope: l } = this.resolveCaps(
579
+ s,
580
+ n
581
+ ), p = r.filter(
582
+ (m) => !b.has(m.field)
583
+ );
584
+ if (p.length === 0)
368
585
  return;
369
- const l = this.buildSnapshot(
370
- r,
371
- n,
372
- t,
586
+ const f = this.buildSnapshot(
587
+ p,
588
+ i,
589
+ e,
373
590
  o,
374
591
  s,
375
- c
592
+ l,
593
+ R(t.messages, t.tools)
376
594
  );
377
- for (const d of r)
378
- this.evaluateCap(d, l);
595
+ for (const m of p)
596
+ await this.evaluateCap(m, f, t.executionContext, a, l);
379
597
  }
380
598
  // ── Enforcement: pre:tool_call ────────────────────────────────────────────
381
- enforcePreTool(e) {
382
- const { toolName: t, callId: o, executionContext: s } = e, {
383
- caps: a,
384
- mode: n,
385
- scope: i
386
- } = this.resolveCaps(s.agentName, "", t), c = this.accumulators.get(s.taskId);
387
- if (!c || a.length === 0)
599
+ async enforcePreTool(t) {
600
+ const { toolName: e, callId: o, executionContext: s } = t, {
601
+ caps: n,
602
+ mode: i,
603
+ scope: r
604
+ } = this.resolveCaps(s.agentName, "", e), a = this.accumulators.get(s.taskId);
605
+ if (!a)
606
+ return;
607
+ const l = n.filter((d) => d.field !== "context");
608
+ if (l.length === 0)
388
609
  return;
389
- const r = c.toolCalls.get(t) ?? 0, l = this.buildSnapshot(
610
+ const p = a.toolCalls.get(e) ?? 0, f = this.buildSnapshot(
611
+ l,
390
612
  a,
391
- c,
392
613
  s.taskId,
393
614
  s.sessionId,
394
615
  s.agentName,
395
- i
616
+ r
396
617
  );
397
- for (const u of a) {
398
- const m = u.field === "toolCalls" ? r : this.getSnapshotValue(l, u, i);
399
- if (m >= u.maximum) {
400
- const d = u.message ?? `tool "${t}": ${u.field} limit is ${u.maximum}, current value is ${Math.round(m)}`;
401
- throw (u.mode ?? n ?? "warning") === "warning" ? v(t, o, d) : T(
402
- `tool:${t}:${u.field}`,
403
- u.maximum,
618
+ for (const d of l) {
619
+ const k = this.resolveCapLimit(d, s), m = d.field === "toolCalls" ? p : this.getSnapshotValue(f, d, r);
620
+ if (m >= k) {
621
+ const h = d.message ?? `tool "${e}": ${d.field} limit is ${k}, current value is ${Math.round(m)}`;
622
+ throw (d.mode ?? i ?? "warning") === "warning" ? (await this.emitBudgetEvent(
623
+ "budget:warning",
624
+ s,
625
+ d,
626
+ m,
627
+ k,
628
+ h
629
+ ), $(e, o, h)) : (await this.emitBudgetEvent(
630
+ "budget:block",
631
+ s,
632
+ d,
633
+ m,
634
+ k,
635
+ h
636
+ ), v(
637
+ `tool:${e}:${d.field}`,
638
+ k,
404
639
  m,
405
- u.message
406
- );
640
+ d.message
641
+ ));
407
642
  }
408
643
  }
409
- c.toolCalls.set(t, r + 1);
644
+ a.toolCalls.set(e, p + 1);
410
645
  }
411
646
  // ── Enforcement: transform:tool_result (response size) ────────────────────
412
- enforceResponseSize(e) {
413
- const { toolName: t, result: o } = e;
647
+ enforceResponseSize(t) {
648
+ const { toolName: e, result: o } = t;
414
649
  if (typeof o != "object" || o === null)
415
650
  return;
416
- const { caps: s, mode: a } = this.resolveCaps("", "", t), n = s.filter((l) => l.field === "responseSize");
417
- if (n.length === 0)
651
+ const { caps: s, mode: n } = this.resolveCaps("", "", e), i = s.filter((p) => p.field === "responseSize");
652
+ if (i.length === 0)
418
653
  return;
419
- const i = o, c = i.content;
420
- if (!Array.isArray(c))
654
+ const r = o, a = r.content;
655
+ if (!Array.isArray(a))
421
656
  return;
422
- let r = 0;
423
- for (const l of c)
424
- if (typeof l == "object" && l !== null) {
425
- const u = l;
426
- u.type === "text" && (r += (u.text ?? "").length);
657
+ let l = 0;
658
+ for (const p of a)
659
+ if (typeof p == "object" && p !== null) {
660
+ const f = p;
661
+ f.type === "text" && (l += (f.text ?? "").length);
427
662
  }
428
- for (const l of n) {
429
- if (r <= l.maximum)
663
+ for (const p of i) {
664
+ const f = this.resolveCapLimit(p, t.executionContext);
665
+ if (l <= f)
430
666
  continue;
431
- if ((l.mode ?? a ?? "warning") === "block")
432
- i.content = [
667
+ if ((p.mode ?? n ?? "warning") === "block")
668
+ r.content = [
433
669
  {
434
670
  type: "text",
435
- text: l.message ?? `Response size (${r} chars) exceeds limit of ${l.maximum}. Use offset/limit or shell paging tools instead.`
671
+ text: p.message ?? `Response size (${l} chars) exceeds limit of ${f}. Use offset/limit or shell paging tools instead.`
436
672
  }
437
- ], e.isError = !0;
673
+ ], t.isError = !0;
438
674
  else {
439
- let m = l.maximum;
440
- const d = [];
441
- for (const k of c) {
442
- if (typeof k != "object" || k === null) {
443
- d.push(k);
675
+ let k = f;
676
+ const m = [];
677
+ for (const h of a) {
678
+ if (typeof h != "object" || h === null) {
679
+ m.push(h);
444
680
  continue;
445
681
  }
446
- const h = k;
447
- if (h.type !== "text") {
448
- d.push(k);
682
+ const C = h;
683
+ if (C.type !== "text") {
684
+ m.push(h);
449
685
  continue;
450
686
  }
451
- const C = h.text ?? "";
452
- if (C.length <= m)
453
- d.push(k), m -= C.length;
687
+ const T = C.text ?? "";
688
+ if (T.length <= k)
689
+ m.push(h), k -= T.length;
454
690
  else {
455
- d.push({ type: "text", text: C.slice(0, m) });
691
+ m.push({ type: "text", text: T.slice(0, k) });
456
692
  break;
457
693
  }
458
694
  }
459
- d.push({
695
+ m.push({
460
696
  type: "text",
461
697
  text: `
462
698
 
463
- [truncated: response was ${r} chars, limited to ${l.maximum}. ${l.message ?? "Use offset/limit or shell paging tools for full content."}]`
464
- }), i.content = d;
699
+ [truncated: response was ${l} chars, limited to ${p.maximum}. ${p.message ?? "Use offset/limit or shell paging tools for full content."}]`
700
+ }), r.content = m;
465
701
  }
466
702
  break;
467
703
  }
468
704
  }
469
705
  }
470
- const $ = ({ db: f, config: e }) => {
471
- var a, n;
472
- const t = _(e), o = ((a = t.defaults) == null ? void 0 : a.costPerInputToken) ?? 0, s = ((n = t.defaults) == null ? void 0 : n.costPerOutputToken) ?? 0;
473
- return new O(f, t, o, s);
706
+ const j = ({ db: c, config: t }) => {
707
+ var r, a, l, p;
708
+ const e = I(t), o = ((r = e.defaults) == null ? void 0 : r.costPerInputToken) ?? 0, s = ((a = e.defaults) == null ? void 0 : a.costPerOutputToken) ?? 0, n = ((l = e.defaults) == null ? void 0 : l.costPerCacheReadToken) ?? o, i = ((p = e.defaults) == null ? void 0 : p.costPerCacheWriteToken) ?? o;
709
+ return new L(c, e, o, s, n, i);
474
710
  };
475
711
  export {
476
- y as configSchema,
477
- $ as createPlugin,
478
- $ as default,
479
- w as pluginConfigSchema
712
+ N as configSchema,
713
+ j as createPlugin,
714
+ j as default,
715
+ _ as pluginConfigSchema
480
716
  };