@lobstack-ai/mcp 0.1.1 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -173,18 +173,32 @@ structured result carries:
173
173
 
174
174
  ### `lobstack_spend`
175
175
 
176
- What the organization has spent over `7d`, `14d`, `30d` or `90d`, grouped by
177
- `day`, `model`, `key` or `agent`, with request counts, tokens, errors and
178
- latency percentiles. Requires a key holding the `usage:read` scope.
179
-
180
- It also reports `unpriced_requests` and sets `is_floor`. The endpoint sums an
181
- unpriced row as zero — the only arithmetic available — so a total that includes
182
- one is a lower bound, not a total, and this tool says which.
183
-
184
- It does **not** report a savings total. Savings are reported per call, by
185
- `lobstack_chat`, where the API sends them with the reason attached. Adding them
186
- up client-side would mean pricing the org's tokens against a copy of the rate
187
- card, and a copy drifts.
176
+ What the organization has spent over `month` (the UTC calendar month to date,
177
+ the Console's default window), `7d`, `14d`, `30d` or `90d`, grouped by `day`,
178
+ `model`, `key` or `agent`, with request counts, tokens, errors and latency
179
+ percentiles. Requires a key holding the `usage:read` scope. The default range
180
+ is `7d`.
181
+
182
+ **The total is the Console's figure.** `/api/v1/usage` returns two totals from
183
+ two tables: `spend.cost_usd`, from the billing ledger that the allowance and
184
+ invoices are read from and that the Console's Spend shows, and
185
+ `summary.cost_usd`, the request trace's own copy of each price, kept for older
186
+ callers. They are written separately and can disagree. This tool reports
187
+ `spend.cost_usd` (and each group's `ledger_cost_usd`) as `cost_usd`, with
188
+ `cost_source: "ledger"`. Only when the ledger figure is null or missing does it
189
+ fall back to `summary.cost_usd`, set `cost_source: "trace"`, and say so in the
190
+ output.
191
+
192
+ It also sets `is_floor`. Unpriced rows sum as zero — the only arithmetic
193
+ available — so a total that includes one is a lower bound, not a total, and
194
+ this tool says how many there were, counted in the same table as the total.
195
+
196
+ **Routing savings are two figures, never one.** When the API sends `savings`,
197
+ the tool shows `named` (saved on models you asked for, measured) and
198
+ `plan_ceiling` (what `auto` requests would have cost on the best model your
199
+ plan allows: a comparison, not a saving) on separate lines. It never adds them
200
+ together, and it never computes a saving client-side from a copy of the rate
201
+ card.
188
202
 
189
203
  ## Two rules about the numbers
190
204
 
package/dist/gateway.js CHANGED
@@ -47,12 +47,12 @@ export async function gwFetch(cfg, url, opts = {}) {
47
47
  });
48
48
  }
49
49
  catch (e) {
50
- throw new GatewayError(`could not reach the gateway: ${scrub(e instanceof Error ? e.message : String(e), cfg.apiKey)}`, { hint: `Base URL in use: ${cfg.base.origin}` });
50
+ throw new GatewayError(`could not reach the Lobstack API: ${scrub(e instanceof Error ? e.message : String(e), cfg.apiKey)}`, { hint: `Base URL in use: ${cfg.base.origin}` });
51
51
  }
52
52
  if (res.status >= 300 && res.status < 400) {
53
53
  // Not followed, and not quietly.
54
54
  const location = res.headers.get("location");
55
- throw new GatewayError(`the gateway redirected (${res.status}) to ${location || "somewhere else"}; the request was not followed.`, {
55
+ throw new GatewayError(`the Lobstack API redirected (${res.status}) to ${location || "somewhere else"}; the request was not followed.`, {
56
56
  status: res.status,
57
57
  hint: "A redirect across hosts strips the Authorization header, so the key would never arrive. " +
58
58
  "Set LOBSTACK_BASE_URL to the host that answers directly — https://www.lobstack.ai, never the bare apex.",
package/dist/receipt.js CHANGED
@@ -108,7 +108,7 @@ export function describeReceipt({ receipt, usage, model, droppedParams }) {
108
108
  lines.push(` measured against ${saving.baseline_model}, the priciest model your plan allows — you sent auto, not that model`);
109
109
  }
110
110
  if (receipt && !receipt.priced) {
111
- lines.push(" the gateway could not price this model, so no cost is claimed");
111
+ lines.push(" the Lobstack API could not price this model, so no cost is claimed");
112
112
  }
113
113
  if (!receipt) {
114
114
  lines.push(" no receipt on this response — the endpoint did not send one");
package/dist/server.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * The MCP server: four tools over the Lobstack Gateway.
2
+ * The MCP server: four tools over the Lobstack API.
3
3
  *
4
4
  * Exported as a factory rather than wired straight to stdio so the tests can
5
5
  * drive it over an in-memory transport with a real MCP client on the other end,
@@ -16,7 +16,8 @@
16
16
  * what the product does before anybody has signed up.
17
17
  * lobstack_models the catalogue, with per-token prices and tiers.
18
18
  * lobstack_chat the actual completion, and the receipt for it.
19
- * lobstack_spend the ledger over a range.
19
+ * lobstack_spend the billing ledger over a range, as the Console
20
+ * shows it, with routing savings split in two.
20
21
  *
21
22
  * Nothing here mints, rotates or reads API keys, and nothing accepts a base URL
22
23
  * as an argument. This process holds a live credential for as long as the MCP
package/dist/server.js CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * The MCP server: four tools over the Lobstack Gateway.
2
+ * The MCP server: four tools over the Lobstack API.
3
3
  *
4
4
  * Exported as a factory rather than wired straight to stdio so the tests can
5
5
  * drive it over an in-memory transport with a real MCP client on the other end,
@@ -16,7 +16,8 @@
16
16
  * what the product does before anybody has signed up.
17
17
  * lobstack_models the catalogue, with per-token prices and tiers.
18
18
  * lobstack_chat the actual completion, and the receipt for it.
19
- * lobstack_spend the ledger over a range.
19
+ * lobstack_spend the billing ledger over a range, as the Console
20
+ * shows it, with routing savings split in two.
20
21
  *
21
22
  * Nothing here mints, rotates or reads API keys, and nothing accepts a base URL
22
23
  * as an argument. This process holds a live credential for as long as the MCP
@@ -48,7 +49,7 @@ export const SERVER_VERSION = (() => {
48
49
  return "0.0.0";
49
50
  }
50
51
  })();
51
- const INSTRUCTIONS = `Lobstack is a metered LLM gateway: one key reaches every major model, and every call
52
+ const INSTRUCTIONS = `The Lobstack API is a metered LLM gateway: one key reaches every major model, and every call
52
53
  comes back with a receipt saying which model served it and what it cost.
53
54
 
54
55
  - lobstack_route_preview needs NO API key. It scores a prompt against the same
@@ -56,7 +57,7 @@ comes back with a receipt saying which model served it and what it cost.
56
57
  estimated cost. Use it to choose a model, or to show what routing does.
57
58
  - lobstack_chat runs the completion. Send model "auto" to let the router pick
58
59
  the cheapest model that can handle the prompt.
59
- - Costs are reported as the gateway priced them. A null cost means the gateway
60
+ - Costs are reported as the API priced them. A null cost means the API
60
61
  could not price the call — it does not mean the call was free.
61
62
  - A saving labelled "saved" is like-for-like: the caller named a model and got
62
63
  something cheaper. A saving labelled "vs ceiling" is measured against the most
@@ -69,23 +70,23 @@ export function createServer(options = {}) {
69
70
  title: "Preview routing and cost",
70
71
  description: "Score a prompt and report which model the Lobstack router would serve it with, and what that would cost. " +
71
72
  "Runs no inference, spends nothing, and NEEDS NO API KEY — use it to pick a model before calling lobstack_chat, " +
72
- "or to show what the gateway does on a machine with no key configured. Token counts are estimates; the billed " +
73
+ "or to show what the Lobstack API does on a machine with no key configured. Token counts are estimates; the billed " +
73
74
  "figure comes from the provider's usage block on the real call.",
74
75
  inputSchema: routePreviewInput,
75
76
  outputSchema: routePreviewOutput,
76
77
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
77
78
  }, async (args) => runRoutePreview(cfg, args));
78
79
  server.registerTool("lobstack_models", {
79
- title: "List gateway models",
80
- description: "The models the Lobstack Gateway serves, with capability tier, provider, context window and USD price per " +
80
+ title: "List Lobstack API models",
81
+ description: "The models the Lobstack API serves, with capability tier, provider, context window and USD price per " +
81
82
  "million input and output tokens. A model the registry cannot price shows a null price, not zero.",
82
83
  inputSchema: modelsInput,
83
84
  outputSchema: modelsOutput,
84
85
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
85
86
  }, async (args) => runModels(cfg, args));
86
87
  server.registerTool("lobstack_chat", {
87
- title: "Chat through the gateway",
88
- description: "Send a prompt or conversation through the Lobstack Gateway and get the reply plus a receipt: the model that " +
88
+ title: "Chat through the Lobstack API",
89
+ description: "Send a prompt or conversation through the Lobstack API and get the reply plus a receipt: the model that " +
89
90
  'actually served it, token counts, USD cost, and any saving with the reason it may be claimed. Model "auto" ' +
90
91
  "(the default) lets the router pick the cheapest model that can handle the prompt. This call spends money " +
91
92
  "against the configured key's allowance.",
@@ -95,9 +96,12 @@ export function createServer(options = {}) {
95
96
  }, async (args) => runChat(cfg, args));
96
97
  server.registerTool("lobstack_spend", {
97
98
  title: "Read spend and usage",
98
- description: "What this organization has spent through the gateway over a range, broken down by day, model, key or agent, " +
99
- "with request counts, tokens, error counts and latency percentiles. Requires an API key with the usage:read " +
100
- "scope. Reports how many requests could not be priced, because a total that includes them is a floor.",
99
+ description: "What this organization has spent through the Lobstack API over a range, broken down by day, model, key or agent, " +
100
+ "with request counts, tokens, error counts and latency percentiles. The total is the billing ledger's figure, " +
101
+ "the same one the Console's Spend shows; if the ledger cannot be read it falls back to the request trace's " +
102
+ "legacy figure and says so. Also shows routing savings as two separate figures, never summed: saved on models " +
103
+ "you named, and a comparison against the best model your plan allows. Reports how many rows could not be " +
104
+ "priced, because a total that includes them is a floor. Requires an API key with the usage:read scope.",
101
105
  inputSchema: spendInput,
102
106
  outputSchema: spendOutput,
103
107
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
package/dist/sse.js CHANGED
@@ -90,7 +90,7 @@ export async function consume(body, onText) {
90
90
  const e = frame.error;
91
91
  // A 200 whose stream carries an error. Headers are long gone by then, so
92
92
  // this is the only place the gateway can report a mid-stream failure.
93
- throw new StreamError(e?.message || "the gateway reported an error mid-stream");
93
+ throw new StreamError(e?.message || "the Lobstack API reported an error mid-stream");
94
94
  }
95
95
  if (typeof frame.model === "string")
96
96
  model = frame.model;
@@ -68,7 +68,7 @@ export const chatOutput = {
68
68
  cost_usd: z
69
69
  .number()
70
70
  .nullable()
71
- .describe("USD the caller owes. NULL — never 0 — when the gateway could not price the call."),
71
+ .describe("USD the caller owes. NULL — never 0 — when the API could not price the call."),
72
72
  cost_display: z.string().describe('Human form. "unpriced" when cost_usd is null.'),
73
73
  priced: z.boolean(),
74
74
  savings: z
@@ -85,7 +85,7 @@ export const chatOutput = {
85
85
  })
86
86
  .nullable()
87
87
  .describe("Null when the endpoint sent no receipt at all."),
88
- quota: z.record(z.unknown()).nullable().describe("Allowance remaining, as the gateway reported it."),
88
+ quota: z.record(z.unknown()).nullable().describe("Allowance remaining, as the API reported it."),
89
89
  dropped_params: z.array(z.string()),
90
90
  };
91
91
  export async function runChat(cfg, args) {
@@ -41,13 +41,13 @@ export function fromThrown(cfg, e) {
41
41
  : "The key was rejected. It may have been revoked or have expired; mint a new one in Console → API keys.");
42
42
  }
43
43
  if (e.requestId)
44
- hints.push(`Gateway request id: ${e.requestId}`);
44
+ hints.push(`Request id: ${e.requestId}`);
45
45
  return failure(cfg, e.message, hints.join("\n"));
46
46
  }
47
47
  if (e instanceof ConfigError)
48
48
  return failure(cfg, e.message, e.hint);
49
49
  if (e instanceof StreamError) {
50
- return failure(cfg, `the gateway failed part-way through the answer: ${e.message}`);
50
+ return failure(cfg, `the Lobstack API failed part-way through the answer: ${e.message}`);
51
51
  }
52
52
  return failure(cfg, e instanceof Error ? e.message : String(e));
53
53
  }
@@ -2,47 +2,140 @@
2
2
  * lobstack_spend — what this organization has spent, over a range.
3
3
  *
4
4
  * `GET /api/v1/usage` is org-scoped and takes either a browser session or an
5
- * API key holding the `usage:read` scope. It is a sibling of the gateway
6
- * prefix, not under it, which is why the base URL here is kept as an origin
7
- * and paths are composed rather than concatenated onto a gateway URL.
8
- *
9
- * TWO THINGS THIS TOOL REPORTS THAT THE ENDPOINT BURIES
10
- *
11
- * `unpriced_requests` — rows the meter could not price. The endpoint's
12
- * `cost_usd` sums a NULL as zero, which is the only arithmetic available and
13
- * not the only truth: a total built partly from unpriced rows is a FLOOR. A
14
- * reader not told how many were unpriced reads it as exact, which is the same
15
- * mistake as a $0.00 receipt, one aggregation up.
16
- *
17
- * `truncated` — the endpoint pages to a cap. When it binds, the sums are a
18
- * floor for a second, independent reason.
19
- *
20
- * WHAT THIS TOOL DOES NOT REPORT (YET)
21
- *
22
- * A savings total. `/api/v1/usage` now computes one, server-side from the
23
- * priced ledger, as a top-level `savings` object split in two and never
24
- * summed: `named` (a measured saving against a model the caller asked for) and
25
- * `plan_ceiling` (a counterfactual against the priciest model the plan allows,
26
- * when the caller sent `auto`). `savings` is null when the ledger could not be
27
- * read. This tool does not read it yet; if it ever does, the two blocks must
28
- * stay separate and `plan_ceiling` must be labelled as a counterfactual. What
29
- * it must never do is add up per-call savings client-side from a rate card we
30
- * hold a copy of. Per-call savings are reported by lobstack_chat, where the
31
- * API sends them with the reason attached.
5
+ * API key holding the `usage:read` scope. It is a sibling of the API prefix,
6
+ * not under it, which is why the base URL here is kept as an origin and paths
7
+ * are composed rather than concatenated onto the API base URL.
8
+ *
9
+ * ONE MONEY FIGURE, AND IT IS THE CONSOLE'S
10
+ *
11
+ * The endpoint returns two totals of the same quantity, from two tables:
12
+ *
13
+ * spend.cost_usd the priced ledger (`token_usage`) — what the allowance
14
+ * is metered from, what invoices are cut from, and what
15
+ * the Console's Spend figure shows.
16
+ * summary.cost_usd the request trace's own copy of each price
17
+ * (`gateway_requests`), kept for older callers.
18
+ *
19
+ * The two are written by separate statements and can disagree: a request the
20
+ * trace recorded but the ledger lost is in the second and not the first, a
21
+ * ledger row whose trace is missing is the other way round, and the unpriced
22
+ * counts are taken over different rows. This tool used to print
23
+ * `summary.cost_usd`, so it could quote a different total from the Console for
24
+ * the same window. It now prints `spend.cost_usd`, and falls back to the trace
25
+ * figure only when the ledger figure is null (the ledger could not be read) or
26
+ * absent (an older deployment) — and says so in the output when it does.
27
+ *
28
+ * The same rule applies per group: `ledger_cost_usd` when it is there, the
29
+ * trace's `cost_usd` only when it is not, as the Console's breakdown does.
30
+ *
31
+ * WHEN THE TOTAL IS A FLOOR
32
+ *
33
+ * Unpriced rows sum as zero, which is the only arithmetic available and not
34
+ * the only truth: a total built partly from unpriced rows is a FLOOR. The
35
+ * unpriced count is taken from the same table as the total it qualifies —
36
+ * `spend.unpriced_rows` for the ledger, `summary.unpriced_requests` for the
37
+ * trace. A read that hit the row cap (`truncated`) is a floor for a second,
38
+ * independent reason.
39
+ *
40
+ * SAVINGS: TWO FIGURES, NEVER ONE
41
+ *
42
+ * The endpoint computes routing savings server-side from the priced ledger, as
43
+ * a `savings` object split in two: `named` (a measured saving against a model
44
+ * the caller asked for) and `plan_ceiling` (a counterfactual against the
45
+ * priciest model the plan allows, when the caller sent `auto`). This tool
46
+ * shows each one on its own line, labels `plan_ceiling` as a comparison rather
47
+ * than a saving, and never adds them together. It never computes a saving
48
+ * client-side from a copy of the rate card.
32
49
  */
33
50
  import { z } from "zod";
34
51
  import type { Config } from "../config.js";
35
52
  import { type ToolResult } from "./shared.js";
36
53
  export declare const spendInput: {
37
- range: z.ZodOptional<z.ZodEnum<["7d", "14d", "30d", "90d"]>>;
54
+ range: z.ZodOptional<z.ZodEnum<["month", "7d", "14d", "30d", "90d"]>>;
38
55
  group_by: z.ZodOptional<z.ZodEnum<["day", "model", "key", "agent"]>>;
39
56
  };
40
57
  export declare const spendOutput: {
41
58
  enabled: z.ZodBoolean;
42
59
  range: z.ZodString;
43
60
  group_by: z.ZodString;
61
+ cost_usd: z.ZodNullable<z.ZodNumber>;
62
+ cost_source: z.ZodNullable<z.ZodEnum<["ledger", "trace"]>>;
63
+ spend: z.ZodNullable<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
44
64
  summary: z.ZodNullable<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
45
65
  groups: z.ZodArray<z.ZodRecord<z.ZodString, z.ZodUnknown>, "many">;
66
+ savings: z.ZodNullable<z.ZodObject<{
67
+ named: z.ZodObject<{
68
+ requests: z.ZodNumber;
69
+ served_cost_usd: z.ZodNumber;
70
+ baseline_cost_usd: z.ZodNumber;
71
+ difference_usd: z.ZodNumber;
72
+ baseline_models: z.ZodArray<z.ZodString, "many">;
73
+ }, "strip", z.ZodTypeAny, {
74
+ baseline_cost_usd: number;
75
+ requests: number;
76
+ served_cost_usd: number;
77
+ difference_usd: number;
78
+ baseline_models: string[];
79
+ }, {
80
+ baseline_cost_usd: number;
81
+ requests: number;
82
+ served_cost_usd: number;
83
+ difference_usd: number;
84
+ baseline_models: string[];
85
+ }>;
86
+ plan_ceiling: z.ZodObject<{
87
+ requests: z.ZodNumber;
88
+ served_cost_usd: z.ZodNumber;
89
+ baseline_cost_usd: z.ZodNumber;
90
+ difference_usd: z.ZodNumber;
91
+ baseline_models: z.ZodArray<z.ZodString, "many">;
92
+ }, "strip", z.ZodTypeAny, {
93
+ baseline_cost_usd: number;
94
+ requests: number;
95
+ served_cost_usd: number;
96
+ difference_usd: number;
97
+ baseline_models: string[];
98
+ }, {
99
+ baseline_cost_usd: number;
100
+ requests: number;
101
+ served_cost_usd: number;
102
+ difference_usd: number;
103
+ baseline_models: string[];
104
+ }>;
105
+ unpriced_routed_requests: z.ZodNumber;
106
+ }, "strip", z.ZodTypeAny, {
107
+ named: {
108
+ baseline_cost_usd: number;
109
+ requests: number;
110
+ served_cost_usd: number;
111
+ difference_usd: number;
112
+ baseline_models: string[];
113
+ };
114
+ plan_ceiling: {
115
+ baseline_cost_usd: number;
116
+ requests: number;
117
+ served_cost_usd: number;
118
+ difference_usd: number;
119
+ baseline_models: string[];
120
+ };
121
+ unpriced_routed_requests: number;
122
+ }, {
123
+ named: {
124
+ baseline_cost_usd: number;
125
+ requests: number;
126
+ served_cost_usd: number;
127
+ difference_usd: number;
128
+ baseline_models: string[];
129
+ };
130
+ plan_ceiling: {
131
+ baseline_cost_usd: number;
132
+ requests: number;
133
+ served_cost_usd: number;
134
+ difference_usd: number;
135
+ baseline_models: string[];
136
+ };
137
+ unpriced_routed_requests: number;
138
+ }>>;
46
139
  is_floor: z.ZodBoolean;
47
140
  message: z.ZodNullable<z.ZodString>;
48
141
  };
@@ -2,51 +2,104 @@
2
2
  * lobstack_spend — what this organization has spent, over a range.
3
3
  *
4
4
  * `GET /api/v1/usage` is org-scoped and takes either a browser session or an
5
- * API key holding the `usage:read` scope. It is a sibling of the gateway
6
- * prefix, not under it, which is why the base URL here is kept as an origin
7
- * and paths are composed rather than concatenated onto a gateway URL.
5
+ * API key holding the `usage:read` scope. It is a sibling of the API prefix,
6
+ * not under it, which is why the base URL here is kept as an origin and paths
7
+ * are composed rather than concatenated onto the API base URL.
8
8
  *
9
- * TWO THINGS THIS TOOL REPORTS THAT THE ENDPOINT BURIES
9
+ * ONE MONEY FIGURE, AND IT IS THE CONSOLE'S
10
10
  *
11
- * `unpriced_requests` — rows the meter could not price. The endpoint's
12
- * `cost_usd` sums a NULL as zero, which is the only arithmetic available and
13
- * not the only truth: a total built partly from unpriced rows is a FLOOR. A
14
- * reader not told how many were unpriced reads it as exact, which is the same
15
- * mistake as a $0.00 receipt, one aggregation up.
11
+ * The endpoint returns two totals of the same quantity, from two tables:
16
12
  *
17
- * `truncated` — the endpoint pages to a cap. When it binds, the sums are a
18
- * floor for a second, independent reason.
13
+ * spend.cost_usd the priced ledger (`token_usage`) — what the allowance
14
+ * is metered from, what invoices are cut from, and what
15
+ * the Console's Spend figure shows.
16
+ * summary.cost_usd the request trace's own copy of each price
17
+ * (`gateway_requests`), kept for older callers.
19
18
  *
20
- * WHAT THIS TOOL DOES NOT REPORT (YET)
19
+ * The two are written by separate statements and can disagree: a request the
20
+ * trace recorded but the ledger lost is in the second and not the first, a
21
+ * ledger row whose trace is missing is the other way round, and the unpriced
22
+ * counts are taken over different rows. This tool used to print
23
+ * `summary.cost_usd`, so it could quote a different total from the Console for
24
+ * the same window. It now prints `spend.cost_usd`, and falls back to the trace
25
+ * figure only when the ledger figure is null (the ledger could not be read) or
26
+ * absent (an older deployment) — and says so in the output when it does.
21
27
  *
22
- * A savings total. `/api/v1/usage` now computes one, server-side from the
23
- * priced ledger, as a top-level `savings` object split in two and never
24
- * summed: `named` (a measured saving against a model the caller asked for) and
25
- * `plan_ceiling` (a counterfactual against the priciest model the plan allows,
26
- * when the caller sent `auto`). `savings` is null when the ledger could not be
27
- * read. This tool does not read it yet; if it ever does, the two blocks must
28
- * stay separate and `plan_ceiling` must be labelled as a counterfactual. What
29
- * it must never do is add up per-call savings client-side from a rate card we
30
- * hold a copy of. Per-call savings are reported by lobstack_chat, where the
31
- * API sends them with the reason attached.
28
+ * The same rule applies per group: `ledger_cost_usd` when it is there, the
29
+ * trace's `cost_usd` only when it is not, as the Console's breakdown does.
30
+ *
31
+ * WHEN THE TOTAL IS A FLOOR
32
+ *
33
+ * Unpriced rows sum as zero, which is the only arithmetic available and not
34
+ * the only truth: a total built partly from unpriced rows is a FLOOR. The
35
+ * unpriced count is taken from the same table as the total it qualifies —
36
+ * `spend.unpriced_rows` for the ledger, `summary.unpriced_requests` for the
37
+ * trace. A read that hit the row cap (`truncated`) is a floor for a second,
38
+ * independent reason.
39
+ *
40
+ * SAVINGS: TWO FIGURES, NEVER ONE
41
+ *
42
+ * The endpoint computes routing savings server-side from the priced ledger, as
43
+ * a `savings` object split in two: `named` (a measured saving against a model
44
+ * the caller asked for) and `plan_ceiling` (a counterfactual against the
45
+ * priciest model the plan allows, when the caller sent `auto`). This tool
46
+ * shows each one on its own line, labels `plan_ceiling` as a comparison rather
47
+ * than a saving, and never adds them together. It never computes a saving
48
+ * client-side from a copy of the rate card.
32
49
  */
33
50
  import { z } from "zod";
34
51
  import { apiUrl } from "../config.js";
35
52
  import { errorFrom, gwFetch } from "../gateway.js";
36
53
  import { money } from "../receipt.js";
37
54
  import { baseNote, fromThrown, ok, requireKey } from "./shared.js";
38
- const RANGES = ["7d", "14d", "30d", "90d"];
55
+ const RANGES = ["month", "7d", "14d", "30d", "90d"];
39
56
  const GROUPS = ["day", "model", "key", "agent"];
40
57
  export const spendInput = {
41
- range: z.enum(RANGES).optional().describe("How far back to look. Defaults to 7d."),
58
+ range: z
59
+ .enum(RANGES)
60
+ .optional()
61
+ .describe('How far back to look. "month" is the UTC calendar month to date, the Console\'s default window. Defaults to 7d.'),
42
62
  group_by: z.enum(GROUPS).optional().describe("How to break the total down. Defaults to model."),
43
63
  };
64
+ const savingsBlock = z.object({
65
+ requests: z.number(),
66
+ served_cost_usd: z.number(),
67
+ baseline_cost_usd: z.number(),
68
+ difference_usd: z.number(),
69
+ baseline_models: z.array(z.string()),
70
+ });
44
71
  export const spendOutput = {
45
72
  enabled: z.boolean().describe("False when request tracing is not enabled on this deployment."),
46
73
  range: z.string(),
47
74
  group_by: z.string(),
48
- summary: z.record(z.unknown()).nullable(),
75
+ cost_usd: z
76
+ .number()
77
+ .nullable()
78
+ .describe("The spend total. The billing ledger's figure (spend.cost_usd), the same one the Console shows, unless " +
79
+ "cost_source says otherwise. Null when neither figure is available."),
80
+ cost_source: z
81
+ .enum(["ledger", "trace"])
82
+ .nullable()
83
+ .describe('"ledger" when cost_usd is the billing ledger\'s figure, as in the Console. "trace" when the ledger figure was ' +
84
+ "null or not reported and cost_usd fell back to the request trace's legacy copy (summary.cost_usd), which " +
85
+ "can differ from the Console."),
86
+ spend: z
87
+ .record(z.unknown())
88
+ .nullable()
89
+ .describe("The billing ledger's spend block as the API returned it. Null when the ledger could not be read."),
90
+ summary: z
91
+ .record(z.unknown())
92
+ .nullable()
93
+ .describe("Requests, errors, tokens and latency from the request trace. Its cost_usd is the legacy figure."),
49
94
  groups: z.array(z.record(z.unknown())),
95
+ savings: z
96
+ .object({
97
+ named: savingsBlock.describe("Measured: requests that named a model and were routed to a cheaper one."),
98
+ plan_ceiling: savingsBlock.describe("A comparison, not a saving: auto requests against the priciest model the plan allows, which nobody asked for."),
99
+ unpriced_routed_requests: z.number(),
100
+ })
101
+ .nullable()
102
+ .describe("Routing savings from the ledger, as two separate figures. Never add them together. Null when not reported."),
50
103
  is_floor: z
51
104
  .boolean()
52
105
  .describe("True when some rows were unpriced or the row cap bound, so the totals are a lower bound, not a total."),
@@ -54,6 +107,42 @@ export const spendOutput = {
54
107
  };
55
108
  const pad = (s, n) => (s.length >= n ? s : s + " ".repeat(n - s.length));
56
109
  const padStart = (s, n) => (s.length >= n ? s : " ".repeat(n - s.length) + s);
110
+ const plural = (n, one, many = `${one}s`) => `${n} ${n === 1 ? one : many}`;
111
+ const isNum = (v) => typeof v === "number" && Number.isFinite(v);
112
+ /** A savings block as a complete, typed object, or null when it is not there. */
113
+ function block(b) {
114
+ if (!b || typeof b !== "object")
115
+ return null;
116
+ return {
117
+ requests: isNum(b.requests) ? b.requests : 0,
118
+ served_cost_usd: isNum(b.served_cost_usd) ? b.served_cost_usd : 0,
119
+ baseline_cost_usd: isNum(b.baseline_cost_usd) ? b.baseline_cost_usd : 0,
120
+ difference_usd: isNum(b.difference_usd) ? b.difference_usd : 0,
121
+ baseline_models: Array.isArray(b.baseline_models) ? b.baseline_models.filter((m) => typeof m === "string") : [],
122
+ };
123
+ }
124
+ /** The savings lines. Each figure on its own line; there is no line that adds them. */
125
+ function savingsLines(named, ceiling, unpriced) {
126
+ const against = (b, fallback) => b.baseline_models.length ? b.baseline_models.join(", ") : fallback;
127
+ const lines = [];
128
+ if (named.requests > 0) {
129
+ const d = named.difference_usd;
130
+ lines.push(d < 0
131
+ ? `Models you named: routing cost ${money(-d)} more on ${plural(named.requests, "request")} ` +
132
+ `(${money(named.served_cost_usd)} instead of ${money(named.baseline_cost_usd)} on ${against(named, "the named models")}).`
133
+ : `Saved on models you named: ${money(d)} on ${plural(named.requests, "request")} ` +
134
+ `(${money(named.served_cost_usd)} instead of ${money(named.baseline_cost_usd)} on ${against(named, "the named models")}).`);
135
+ }
136
+ if (ceiling.requests > 0) {
137
+ lines.push(`Compared with the best model your plan allows: ${money(ceiling.difference_usd)} on ` +
138
+ `${plural(ceiling.requests, "auto request")}. A comparison, not a saving: nothing was named, and ` +
139
+ `${against(ceiling, "your plan's top model")} was not requested.`);
140
+ }
141
+ if (unpriced > 0) {
142
+ lines.push(`${plural(unpriced, "routed request")} carry a baseline that could not be priced and ${unpriced === 1 ? "is" : "are"} in neither figure.`);
143
+ }
144
+ return lines.length ? ["Routing savings (two separate figures, never added together):", ...lines.map((l) => ` ${l}`)] : [];
145
+ }
57
146
  export async function runSpend(cfg, args) {
58
147
  const missing = requireKey(cfg);
59
148
  if (missing)
@@ -75,45 +164,97 @@ export async function runSpend(cfg, args) {
75
164
  enabled: false,
76
165
  range,
77
166
  group_by: groupBy,
167
+ cost_usd: null,
168
+ cost_source: null,
169
+ spend: null,
78
170
  summary: null,
79
171
  groups: [],
172
+ savings: null,
80
173
  is_floor: false,
81
174
  message: b.message ?? null,
82
175
  });
83
176
  }
84
177
  const s = b.summary ?? {};
85
- const unpriced = s.unpriced_requests ?? 0;
86
- const truncated = b.truncated === true;
178
+ const spend = b.spend && typeof b.spend === "object" ? b.spend : null;
179
+ /*
180
+ * The Console's figure, or the legacy one and a sentence saying so. The
181
+ * unpriced count and its denominator always come from the same table as
182
+ * the total they qualify.
183
+ */
184
+ const fromLedger = spend !== null && isNum(spend.cost_usd);
185
+ const traceCost = isNum(s.cost_usd) ? s.cost_usd : null;
186
+ const cost = fromLedger ? spend.cost_usd : traceCost;
187
+ const costSource = fromLedger ? "ledger" : traceCost !== null ? "trace" : null;
188
+ const unpriced = fromLedger ? spend.unpriced_rows ?? 0 : s.unpriced_requests ?? 0;
189
+ const truncated = b.truncated === true || (fromLedger && spend.truncated === true);
87
190
  const isFloor = unpriced > 0 || truncated;
88
191
  const groups = b.groups ?? [];
89
- const head = `Last ${b.range ?? range} · ${s.requests ?? 0} request${s.requests === 1 ? "" : "s"} · ` +
90
- `${money(typeof s.cost_usd === "number" ? s.cost_usd : null)}` +
192
+ /* As the Console renders it: a window in which no row carried a price has
193
+ an unknown total, not a total of zero. */
194
+ const priceable = fromLedger ? spend.rows ?? 0 : s.requests ?? 0;
195
+ const costText = cost === null ? "spend unknown" : priceable > 0 && unpriced >= priceable ? "unpriced" : money(cost);
196
+ const shown = b.range ?? range;
197
+ const head = `${shown === "month" ? "This month (UTC)" : `Last ${shown}`} · ${plural(s.requests ?? 0, "request")} · ` +
198
+ costText +
91
199
  (s.total_tokens ? ` · ${s.total_tokens.toLocaleString("en-US")} tokens` : "") +
92
- (s.errors ? ` · ${s.errors} error${s.errors === 1 ? "" : "s"}` : "");
200
+ (s.errors ? ` · ${plural(s.errors, "error")}` : "");
201
+ /* Per group, the same rule as the Console's breakdown: the ledger's spend
202
+ when the group carries it, the trace's copy only when it does not. */
203
+ const groupCost = (g) => (isNum(g.ledger_cost_usd) ? g.ledger_cost_usd : isNum(g.cost_usd) ? g.cost_usd : null);
204
+ const groupUnpriced = (g) => isNum(g.ledger_cost_usd) ? g.ledger_unpriced_rows ?? 0 : g.unpriced_requests ?? 0;
93
205
  const w = Math.max(3, ...groups.map((g) => String(g.key ?? "").length));
94
206
  const table = groups.length
95
207
  ? [
96
208
  "",
97
209
  `${pad(String(groupBy).toUpperCase(), w)} ${padStart("REQS", 6)} ${padStart("COST", 11)}`,
98
- ...groups.map((g) => `${pad(String(g.key ?? ""), w)} ${padStart(String(g.requests ?? 0), 6)} ` +
99
- `${padStart(money(typeof g.cost_usd === "number" ? g.cost_usd : null), 11)}` +
100
- (g.unpriced_requests ? ` (${g.unpriced_requests} unpriced)` : "")),
210
+ ...groups.map((g) => {
211
+ const u = groupUnpriced(g);
212
+ return (`${pad(String(g.key ?? ""), w)} ${padStart(String(g.requests ?? 0), 6)} ` +
213
+ `${padStart(money(groupCost(g)), 11)}` +
214
+ (u ? ` (${u} unpriced)` : ""));
215
+ }),
101
216
  ].join("\n")
102
217
  : "\nNo requests in this window.";
218
+ const named = block(b.savings?.named);
219
+ const ceiling = block(b.savings?.plan_ceiling);
220
+ const savings = named && ceiling
221
+ ? {
222
+ named,
223
+ plan_ceiling: ceiling,
224
+ unpriced_routed_requests: isNum(b.savings?.unpriced_routed_requests) ? b.savings.unpriced_routed_requests : 0,
225
+ }
226
+ : null;
227
+ const unmetered = b.ledger?.unmetered_requests ?? 0;
103
228
  const notes = [
229
+ !fromLedger
230
+ ? spend === null && "spend" in b
231
+ ? "The billing ledger could not be read, so this total is the request trace's copy of each price (the legacy summary.cost_usd). It can differ from the Console's Spend."
232
+ : "This deployment does not report the billing ledger's spend, so this total is the request trace's copy of each price (the legacy summary.cost_usd). It can differ from the Console's Spend."
233
+ : null,
234
+ fromLedger && (spend.byok_cost_usd ?? 0) > 0
235
+ ? `${money(spend.managed_cost_usd ?? 0)} billed by Lobstack · ${money(spend.byok_cost_usd ?? 0)} on your own provider key.`
236
+ : null,
104
237
  unpriced
105
- ? `${unpriced} request${unpriced === 1 ? "" : "s"} could not be priced. Those rows sum as zero, so the total above is a floor, not a total.`
238
+ ? `${plural(unpriced, fromLedger ? "row" : "request")} could not be priced. Those sum as zero, so the total above is a floor, not a total.`
106
239
  : null,
107
240
  truncated ? "The row cap bound on this range, so older requests are not counted here." : null,
241
+ fromLedger && unmetered > 0
242
+ ? `${plural(unmetered, "served request")} ${unmetered === 1 ? "has" : "have"} no billing ledger row, so ${unmetered === 1 ? "it is" : "they are"} not in this total.`
243
+ : null,
108
244
  typeof s.p95_latency_ms === "number" ? `p50 ${s.p50_latency_ms ?? "—"}ms · p95 ${s.p95_latency_ms}ms` : null,
109
245
  baseNote(cfg),
110
- ].filter((n) => n !== null);
111
- return ok([head + table + (notes.length ? "\n\n" + notes.join("\n") : "")], {
246
+ ].filter((n) => typeof n === "string");
247
+ const saved = savings ? savingsLines(savings.named, savings.plan_ceiling, savings.unpriced_routed_requests) : [];
248
+ return ok([head + table + (saved.length ? "\n\n" + saved.join("\n") : "") + (notes.length ? "\n\n" + notes.join("\n") : "")], {
112
249
  enabled: true,
113
250
  range: b.range ?? range,
114
251
  group_by: b.group_by ?? groupBy,
252
+ cost_usd: cost,
253
+ cost_source: costSource,
254
+ spend: (spend ?? null),
115
255
  summary: (b.summary ?? null),
116
256
  groups: groups,
257
+ savings,
117
258
  is_floor: isFloor,
118
259
  message: null,
119
260
  });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lobstack-ai/mcp",
3
- "version": "0.1.1",
3
+ "version": "0.1.3",
4
4
  "description": "The Lobstack API as an MCP server: one key across many model providers, and what each call cost.",
5
5
  "type": "module",
6
6
  "bin": {