@lobstack-ai/mcp 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -12
- package/dist/gateway.js +2 -2
- package/dist/receipt.js +1 -1
- package/dist/server.d.ts +3 -2
- package/dist/server.js +16 -12
- package/dist/sse.js +1 -1
- package/dist/tools/chat.js +2 -2
- package/dist/tools/shared.js +2 -2
- package/dist/tools/spend.d.ts +121 -28
- package/dist/tools/spend.js +177 -36
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -173,18 +173,32 @@ structured result carries:
|
|
|
173
173
|
|
|
174
174
|
### `lobstack_spend`
|
|
175
175
|
|
|
176
|
-
What the organization has spent over `
|
|
177
|
-
`
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
`
|
|
186
|
-
|
|
187
|
-
|
|
176
|
+
What the organization has spent over `month` (the UTC calendar month to date,
|
|
177
|
+
the Console's default window), `7d`, `14d`, `30d` or `90d`, grouped by `day`,
|
|
178
|
+
`model`, `key` or `agent`, with request counts, tokens, errors and latency
|
|
179
|
+
percentiles. Requires a key holding the `usage:read` scope. The default range
|
|
180
|
+
is `7d`.
|
|
181
|
+
|
|
182
|
+
**The total is the Console's figure.** `/api/v1/usage` returns two totals from
|
|
183
|
+
two tables: `spend.cost_usd`, from the billing ledger that the allowance and
|
|
184
|
+
invoices are read from and that the Console's Spend shows, and
|
|
185
|
+
`summary.cost_usd`, the request trace's own copy of each price, kept for older
|
|
186
|
+
callers. They are written separately and can disagree. This tool reports
|
|
187
|
+
`spend.cost_usd` (and each group's `ledger_cost_usd`) as `cost_usd`, with
|
|
188
|
+
`cost_source: "ledger"`. Only when the ledger figure is null or missing does it
|
|
189
|
+
fall back to `summary.cost_usd`, set `cost_source: "trace"`, and say so in the
|
|
190
|
+
output.
|
|
191
|
+
|
|
192
|
+
It also sets `is_floor`. Unpriced rows sum as zero — the only arithmetic
|
|
193
|
+
available — so a total that includes one is a lower bound, not a total, and
|
|
194
|
+
this tool says how many there were, counted in the same table as the total.
|
|
195
|
+
|
|
196
|
+
**Routing savings are two figures, never one.** When the API sends `savings`,
|
|
197
|
+
the tool shows `named` (saved on models you asked for, measured) and
|
|
198
|
+
`plan_ceiling` (what `auto` requests would have cost on the best model your
|
|
199
|
+
plan allows: a comparison, not a saving) on separate lines. It never adds them
|
|
200
|
+
together, and it never computes a saving client-side from a copy of the rate
|
|
201
|
+
card.
|
|
188
202
|
|
|
189
203
|
## Two rules about the numbers
|
|
190
204
|
|
package/dist/gateway.js
CHANGED
|
@@ -47,12 +47,12 @@ export async function gwFetch(cfg, url, opts = {}) {
|
|
|
47
47
|
});
|
|
48
48
|
}
|
|
49
49
|
catch (e) {
|
|
50
|
-
throw new GatewayError(`could not reach the
|
|
50
|
+
throw new GatewayError(`could not reach the Lobstack API: ${scrub(e instanceof Error ? e.message : String(e), cfg.apiKey)}`, { hint: `Base URL in use: ${cfg.base.origin}` });
|
|
51
51
|
}
|
|
52
52
|
if (res.status >= 300 && res.status < 400) {
|
|
53
53
|
// Not followed, and not quietly.
|
|
54
54
|
const location = res.headers.get("location");
|
|
55
|
-
throw new GatewayError(`the
|
|
55
|
+
throw new GatewayError(`the Lobstack API redirected (${res.status}) to ${location || "somewhere else"}; the request was not followed.`, {
|
|
56
56
|
status: res.status,
|
|
57
57
|
hint: "A redirect across hosts strips the Authorization header, so the key would never arrive. " +
|
|
58
58
|
"Set LOBSTACK_BASE_URL to the host that answers directly — https://www.lobstack.ai, never the bare apex.",
|
package/dist/receipt.js
CHANGED
|
@@ -108,7 +108,7 @@ export function describeReceipt({ receipt, usage, model, droppedParams }) {
|
|
|
108
108
|
lines.push(` measured against ${saving.baseline_model}, the priciest model your plan allows — you sent auto, not that model`);
|
|
109
109
|
}
|
|
110
110
|
if (receipt && !receipt.priced) {
|
|
111
|
-
lines.push(" the
|
|
111
|
+
lines.push(" the Lobstack API could not price this model, so no cost is claimed");
|
|
112
112
|
}
|
|
113
113
|
if (!receipt) {
|
|
114
114
|
lines.push(" no receipt on this response — the endpoint did not send one");
|
package/dist/server.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The MCP server: four tools over the Lobstack
|
|
2
|
+
* The MCP server: four tools over the Lobstack API.
|
|
3
3
|
*
|
|
4
4
|
* Exported as a factory rather than wired straight to stdio so the tests can
|
|
5
5
|
* drive it over an in-memory transport with a real MCP client on the other end,
|
|
@@ -16,7 +16,8 @@
|
|
|
16
16
|
* what the product does before anybody has signed up.
|
|
17
17
|
* lobstack_models the catalogue, with per-token prices and tiers.
|
|
18
18
|
* lobstack_chat the actual completion, and the receipt for it.
|
|
19
|
-
* lobstack_spend the ledger over a range
|
|
19
|
+
* lobstack_spend the billing ledger over a range, as the Console
|
|
20
|
+
* shows it, with routing savings split in two.
|
|
20
21
|
*
|
|
21
22
|
* Nothing here mints, rotates or reads API keys, and nothing accepts a base URL
|
|
22
23
|
* as an argument. This process holds a live credential for as long as the MCP
|
package/dist/server.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The MCP server: four tools over the Lobstack
|
|
2
|
+
* The MCP server: four tools over the Lobstack API.
|
|
3
3
|
*
|
|
4
4
|
* Exported as a factory rather than wired straight to stdio so the tests can
|
|
5
5
|
* drive it over an in-memory transport with a real MCP client on the other end,
|
|
@@ -16,7 +16,8 @@
|
|
|
16
16
|
* what the product does before anybody has signed up.
|
|
17
17
|
* lobstack_models the catalogue, with per-token prices and tiers.
|
|
18
18
|
* lobstack_chat the actual completion, and the receipt for it.
|
|
19
|
-
* lobstack_spend the ledger over a range
|
|
19
|
+
* lobstack_spend the billing ledger over a range, as the Console
|
|
20
|
+
* shows it, with routing savings split in two.
|
|
20
21
|
*
|
|
21
22
|
* Nothing here mints, rotates or reads API keys, and nothing accepts a base URL
|
|
22
23
|
* as an argument. This process holds a live credential for as long as the MCP
|
|
@@ -48,7 +49,7 @@ export const SERVER_VERSION = (() => {
|
|
|
48
49
|
return "0.0.0";
|
|
49
50
|
}
|
|
50
51
|
})();
|
|
51
|
-
const INSTRUCTIONS = `Lobstack is a metered LLM gateway: one key reaches every major model, and every call
|
|
52
|
+
const INSTRUCTIONS = `The Lobstack API is a metered LLM gateway: one key reaches every major model, and every call
|
|
52
53
|
comes back with a receipt saying which model served it and what it cost.
|
|
53
54
|
|
|
54
55
|
- lobstack_route_preview needs NO API key. It scores a prompt against the same
|
|
@@ -56,7 +57,7 @@ comes back with a receipt saying which model served it and what it cost.
|
|
|
56
57
|
estimated cost. Use it to choose a model, or to show what routing does.
|
|
57
58
|
- lobstack_chat runs the completion. Send model "auto" to let the router pick
|
|
58
59
|
the cheapest model that can handle the prompt.
|
|
59
|
-
- Costs are reported as the
|
|
60
|
+
- Costs are reported as the API priced them. A null cost means the API
|
|
60
61
|
could not price the call — it does not mean the call was free.
|
|
61
62
|
- A saving labelled "saved" is like-for-like: the caller named a model and got
|
|
62
63
|
something cheaper. A saving labelled "vs ceiling" is measured against the most
|
|
@@ -69,23 +70,23 @@ export function createServer(options = {}) {
|
|
|
69
70
|
title: "Preview routing and cost",
|
|
70
71
|
description: "Score a prompt and report which model the Lobstack router would serve it with, and what that would cost. " +
|
|
71
72
|
"Runs no inference, spends nothing, and NEEDS NO API KEY — use it to pick a model before calling lobstack_chat, " +
|
|
72
|
-
"or to show what the
|
|
73
|
+
"or to show what the Lobstack API does on a machine with no key configured. Token counts are estimates; the billed " +
|
|
73
74
|
"figure comes from the provider's usage block on the real call.",
|
|
74
75
|
inputSchema: routePreviewInput,
|
|
75
76
|
outputSchema: routePreviewOutput,
|
|
76
77
|
annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
|
|
77
78
|
}, async (args) => runRoutePreview(cfg, args));
|
|
78
79
|
server.registerTool("lobstack_models", {
|
|
79
|
-
title: "List
|
|
80
|
-
description: "The models the Lobstack
|
|
80
|
+
title: "List Lobstack API models",
|
|
81
|
+
description: "The models the Lobstack API serves, with capability tier, provider, context window and USD price per " +
|
|
81
82
|
"million input and output tokens. A model the registry cannot price shows a null price, not zero.",
|
|
82
83
|
inputSchema: modelsInput,
|
|
83
84
|
outputSchema: modelsOutput,
|
|
84
85
|
annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
|
|
85
86
|
}, async (args) => runModels(cfg, args));
|
|
86
87
|
server.registerTool("lobstack_chat", {
|
|
87
|
-
title: "Chat through the
|
|
88
|
-
description: "Send a prompt or conversation through the Lobstack
|
|
88
|
+
title: "Chat through the Lobstack API",
|
|
89
|
+
description: "Send a prompt or conversation through the Lobstack API and get the reply plus a receipt: the model that " +
|
|
89
90
|
'actually served it, token counts, USD cost, and any saving with the reason it may be claimed. Model "auto" ' +
|
|
90
91
|
"(the default) lets the router pick the cheapest model that can handle the prompt. This call spends money " +
|
|
91
92
|
"against the configured key's allowance.",
|
|
@@ -95,9 +96,12 @@ export function createServer(options = {}) {
|
|
|
95
96
|
}, async (args) => runChat(cfg, args));
|
|
96
97
|
server.registerTool("lobstack_spend", {
|
|
97
98
|
title: "Read spend and usage",
|
|
98
|
-
description: "What this organization has spent through the
|
|
99
|
-
"with request counts, tokens, error counts and latency percentiles.
|
|
100
|
-
"
|
|
99
|
+
description: "What this organization has spent through the Lobstack API over a range, broken down by day, model, key or agent, " +
|
|
100
|
+
"with request counts, tokens, error counts and latency percentiles. The total is the billing ledger's figure, " +
|
|
101
|
+
"the same one the Console's Spend shows; if the ledger cannot be read it falls back to the request trace's " +
|
|
102
|
+
"legacy figure and says so. Also shows routing savings as two separate figures, never summed: saved on models " +
|
|
103
|
+
"you named, and a comparison against the best model your plan allows. Reports how many rows could not be " +
|
|
104
|
+
"priced, because a total that includes them is a floor. Requires an API key with the usage:read scope.",
|
|
101
105
|
inputSchema: spendInput,
|
|
102
106
|
outputSchema: spendOutput,
|
|
103
107
|
annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
|
package/dist/sse.js
CHANGED
|
@@ -90,7 +90,7 @@ export async function consume(body, onText) {
|
|
|
90
90
|
const e = frame.error;
|
|
91
91
|
// A 200 whose stream carries an error. Headers are long gone by then, so
|
|
92
92
|
// this is the only place the gateway can report a mid-stream failure.
|
|
93
|
-
throw new StreamError(e?.message || "the
|
|
93
|
+
throw new StreamError(e?.message || "the Lobstack API reported an error mid-stream");
|
|
94
94
|
}
|
|
95
95
|
if (typeof frame.model === "string")
|
|
96
96
|
model = frame.model;
|
package/dist/tools/chat.js
CHANGED
|
@@ -68,7 +68,7 @@ export const chatOutput = {
|
|
|
68
68
|
cost_usd: z
|
|
69
69
|
.number()
|
|
70
70
|
.nullable()
|
|
71
|
-
.describe("USD the caller owes. NULL — never 0 — when the
|
|
71
|
+
.describe("USD the caller owes. NULL — never 0 — when the API could not price the call."),
|
|
72
72
|
cost_display: z.string().describe('Human form. "unpriced" when cost_usd is null.'),
|
|
73
73
|
priced: z.boolean(),
|
|
74
74
|
savings: z
|
|
@@ -85,7 +85,7 @@ export const chatOutput = {
|
|
|
85
85
|
})
|
|
86
86
|
.nullable()
|
|
87
87
|
.describe("Null when the endpoint sent no receipt at all."),
|
|
88
|
-
quota: z.record(z.unknown()).nullable().describe("Allowance remaining, as the
|
|
88
|
+
quota: z.record(z.unknown()).nullable().describe("Allowance remaining, as the API reported it."),
|
|
89
89
|
dropped_params: z.array(z.string()),
|
|
90
90
|
};
|
|
91
91
|
export async function runChat(cfg, args) {
|
package/dist/tools/shared.js
CHANGED
|
@@ -41,13 +41,13 @@ export function fromThrown(cfg, e) {
|
|
|
41
41
|
: "The key was rejected. It may have been revoked or have expired; mint a new one in Console → API keys.");
|
|
42
42
|
}
|
|
43
43
|
if (e.requestId)
|
|
44
|
-
hints.push(`
|
|
44
|
+
hints.push(`Request id: ${e.requestId}`);
|
|
45
45
|
return failure(cfg, e.message, hints.join("\n"));
|
|
46
46
|
}
|
|
47
47
|
if (e instanceof ConfigError)
|
|
48
48
|
return failure(cfg, e.message, e.hint);
|
|
49
49
|
if (e instanceof StreamError) {
|
|
50
|
-
return failure(cfg, `the
|
|
50
|
+
return failure(cfg, `the Lobstack API failed part-way through the answer: ${e.message}`);
|
|
51
51
|
}
|
|
52
52
|
return failure(cfg, e instanceof Error ? e.message : String(e));
|
|
53
53
|
}
|
package/dist/tools/spend.d.ts
CHANGED
|
@@ -2,47 +2,140 @@
|
|
|
2
2
|
* lobstack_spend — what this organization has spent, over a range.
|
|
3
3
|
*
|
|
4
4
|
* `GET /api/v1/usage` is org-scoped and takes either a browser session or an
|
|
5
|
-
* API key holding the `usage:read` scope. It is a sibling of the
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
5
|
+
* API key holding the `usage:read` scope. It is a sibling of the API prefix,
|
|
6
|
+
* not under it, which is why the base URL here is kept as an origin and paths
|
|
7
|
+
* are composed rather than concatenated onto the API base URL.
|
|
8
|
+
*
|
|
9
|
+
* ONE MONEY FIGURE, AND IT IS THE CONSOLE'S
|
|
10
|
+
*
|
|
11
|
+
* The endpoint returns two totals of the same quantity, from two tables:
|
|
12
|
+
*
|
|
13
|
+
* spend.cost_usd the priced ledger (`token_usage`) — what the allowance
|
|
14
|
+
* is metered from, what invoices are cut from, and what
|
|
15
|
+
* the Console's Spend figure shows.
|
|
16
|
+
* summary.cost_usd the request trace's own copy of each price
|
|
17
|
+
* (`gateway_requests`), kept for older callers.
|
|
18
|
+
*
|
|
19
|
+
* The two are written by separate statements and can disagree: a request the
|
|
20
|
+
* trace recorded but the ledger lost is in the second and not the first, a
|
|
21
|
+
* ledger row whose trace is missing is the other way round, and the unpriced
|
|
22
|
+
* counts are taken over different rows. This tool used to print
|
|
23
|
+
* `summary.cost_usd`, so it could quote a different total from the Console for
|
|
24
|
+
* the same window. It now prints `spend.cost_usd`, and falls back to the trace
|
|
25
|
+
* figure only when the ledger figure is null (the ledger could not be read) or
|
|
26
|
+
* absent (an older deployment) — and says so in the output when it does.
|
|
27
|
+
*
|
|
28
|
+
* The same rule applies per group: `ledger_cost_usd` when it is there, the
|
|
29
|
+
* trace's `cost_usd` only when it is not, as the Console's breakdown does.
|
|
30
|
+
*
|
|
31
|
+
* WHEN THE TOTAL IS A FLOOR
|
|
32
|
+
*
|
|
33
|
+
* Unpriced rows sum as zero, which is the only arithmetic available and not
|
|
34
|
+
* the only truth: a total built partly from unpriced rows is a FLOOR. The
|
|
35
|
+
* unpriced count is taken from the same table as the total it qualifies —
|
|
36
|
+
* `spend.unpriced_rows` for the ledger, `summary.unpriced_requests` for the
|
|
37
|
+
* trace. A read that hit the row cap (`truncated`) is a floor for a second,
|
|
38
|
+
* independent reason.
|
|
39
|
+
*
|
|
40
|
+
* SAVINGS: TWO FIGURES, NEVER ONE
|
|
41
|
+
*
|
|
42
|
+
* The endpoint computes routing savings server-side from the priced ledger, as
|
|
43
|
+
* a `savings` object split in two: `named` (a measured saving against a model
|
|
44
|
+
* the caller asked for) and `plan_ceiling` (a counterfactual against the
|
|
45
|
+
* priciest model the plan allows, when the caller sent `auto`). This tool
|
|
46
|
+
* shows each one on its own line, labels `plan_ceiling` as a comparison rather
|
|
47
|
+
* than a saving, and never adds them together. It never computes a saving
|
|
48
|
+
* client-side from a copy of the rate card.
|
|
32
49
|
*/
|
|
33
50
|
import { z } from "zod";
|
|
34
51
|
import type { Config } from "../config.js";
|
|
35
52
|
import { type ToolResult } from "./shared.js";
|
|
36
53
|
export declare const spendInput: {
|
|
37
|
-
range: z.ZodOptional<z.ZodEnum<["7d", "14d", "30d", "90d"]>>;
|
|
54
|
+
range: z.ZodOptional<z.ZodEnum<["month", "7d", "14d", "30d", "90d"]>>;
|
|
38
55
|
group_by: z.ZodOptional<z.ZodEnum<["day", "model", "key", "agent"]>>;
|
|
39
56
|
};
|
|
40
57
|
export declare const spendOutput: {
|
|
41
58
|
enabled: z.ZodBoolean;
|
|
42
59
|
range: z.ZodString;
|
|
43
60
|
group_by: z.ZodString;
|
|
61
|
+
cost_usd: z.ZodNullable<z.ZodNumber>;
|
|
62
|
+
cost_source: z.ZodNullable<z.ZodEnum<["ledger", "trace"]>>;
|
|
63
|
+
spend: z.ZodNullable<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
44
64
|
summary: z.ZodNullable<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
45
65
|
groups: z.ZodArray<z.ZodRecord<z.ZodString, z.ZodUnknown>, "many">;
|
|
66
|
+
savings: z.ZodNullable<z.ZodObject<{
|
|
67
|
+
named: z.ZodObject<{
|
|
68
|
+
requests: z.ZodNumber;
|
|
69
|
+
served_cost_usd: z.ZodNumber;
|
|
70
|
+
baseline_cost_usd: z.ZodNumber;
|
|
71
|
+
difference_usd: z.ZodNumber;
|
|
72
|
+
baseline_models: z.ZodArray<z.ZodString, "many">;
|
|
73
|
+
}, "strip", z.ZodTypeAny, {
|
|
74
|
+
baseline_cost_usd: number;
|
|
75
|
+
requests: number;
|
|
76
|
+
served_cost_usd: number;
|
|
77
|
+
difference_usd: number;
|
|
78
|
+
baseline_models: string[];
|
|
79
|
+
}, {
|
|
80
|
+
baseline_cost_usd: number;
|
|
81
|
+
requests: number;
|
|
82
|
+
served_cost_usd: number;
|
|
83
|
+
difference_usd: number;
|
|
84
|
+
baseline_models: string[];
|
|
85
|
+
}>;
|
|
86
|
+
plan_ceiling: z.ZodObject<{
|
|
87
|
+
requests: z.ZodNumber;
|
|
88
|
+
served_cost_usd: z.ZodNumber;
|
|
89
|
+
baseline_cost_usd: z.ZodNumber;
|
|
90
|
+
difference_usd: z.ZodNumber;
|
|
91
|
+
baseline_models: z.ZodArray<z.ZodString, "many">;
|
|
92
|
+
}, "strip", z.ZodTypeAny, {
|
|
93
|
+
baseline_cost_usd: number;
|
|
94
|
+
requests: number;
|
|
95
|
+
served_cost_usd: number;
|
|
96
|
+
difference_usd: number;
|
|
97
|
+
baseline_models: string[];
|
|
98
|
+
}, {
|
|
99
|
+
baseline_cost_usd: number;
|
|
100
|
+
requests: number;
|
|
101
|
+
served_cost_usd: number;
|
|
102
|
+
difference_usd: number;
|
|
103
|
+
baseline_models: string[];
|
|
104
|
+
}>;
|
|
105
|
+
unpriced_routed_requests: z.ZodNumber;
|
|
106
|
+
}, "strip", z.ZodTypeAny, {
|
|
107
|
+
named: {
|
|
108
|
+
baseline_cost_usd: number;
|
|
109
|
+
requests: number;
|
|
110
|
+
served_cost_usd: number;
|
|
111
|
+
difference_usd: number;
|
|
112
|
+
baseline_models: string[];
|
|
113
|
+
};
|
|
114
|
+
plan_ceiling: {
|
|
115
|
+
baseline_cost_usd: number;
|
|
116
|
+
requests: number;
|
|
117
|
+
served_cost_usd: number;
|
|
118
|
+
difference_usd: number;
|
|
119
|
+
baseline_models: string[];
|
|
120
|
+
};
|
|
121
|
+
unpriced_routed_requests: number;
|
|
122
|
+
}, {
|
|
123
|
+
named: {
|
|
124
|
+
baseline_cost_usd: number;
|
|
125
|
+
requests: number;
|
|
126
|
+
served_cost_usd: number;
|
|
127
|
+
difference_usd: number;
|
|
128
|
+
baseline_models: string[];
|
|
129
|
+
};
|
|
130
|
+
plan_ceiling: {
|
|
131
|
+
baseline_cost_usd: number;
|
|
132
|
+
requests: number;
|
|
133
|
+
served_cost_usd: number;
|
|
134
|
+
difference_usd: number;
|
|
135
|
+
baseline_models: string[];
|
|
136
|
+
};
|
|
137
|
+
unpriced_routed_requests: number;
|
|
138
|
+
}>>;
|
|
46
139
|
is_floor: z.ZodBoolean;
|
|
47
140
|
message: z.ZodNullable<z.ZodString>;
|
|
48
141
|
};
|
package/dist/tools/spend.js
CHANGED
|
@@ -2,51 +2,104 @@
|
|
|
2
2
|
* lobstack_spend — what this organization has spent, over a range.
|
|
3
3
|
*
|
|
4
4
|
* `GET /api/v1/usage` is org-scoped and takes either a browser session or an
|
|
5
|
-
* API key holding the `usage:read` scope. It is a sibling of the
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* API key holding the `usage:read` scope. It is a sibling of the API prefix,
|
|
6
|
+
* not under it, which is why the base URL here is kept as an origin and paths
|
|
7
|
+
* are composed rather than concatenated onto the API base URL.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
9
|
+
* ONE MONEY FIGURE, AND IT IS THE CONSOLE'S
|
|
10
10
|
*
|
|
11
|
-
*
|
|
12
|
-
* `cost_usd` sums a NULL as zero, which is the only arithmetic available and
|
|
13
|
-
* not the only truth: a total built partly from unpriced rows is a FLOOR. A
|
|
14
|
-
* reader not told how many were unpriced reads it as exact, which is the same
|
|
15
|
-
* mistake as a $0.00 receipt, one aggregation up.
|
|
11
|
+
* The endpoint returns two totals of the same quantity, from two tables:
|
|
16
12
|
*
|
|
17
|
-
* `
|
|
18
|
-
*
|
|
13
|
+
* spend.cost_usd the priced ledger (`token_usage`) — what the allowance
|
|
14
|
+
* is metered from, what invoices are cut from, and what
|
|
15
|
+
* the Console's Spend figure shows.
|
|
16
|
+
* summary.cost_usd the request trace's own copy of each price
|
|
17
|
+
* (`gateway_requests`), kept for older callers.
|
|
19
18
|
*
|
|
20
|
-
*
|
|
19
|
+
* The two are written by separate statements and can disagree: a request the
|
|
20
|
+
* trace recorded but the ledger lost is in the second and not the first, a
|
|
21
|
+
* ledger row whose trace is missing is the other way round, and the unpriced
|
|
22
|
+
* counts are taken over different rows. This tool used to print
|
|
23
|
+
* `summary.cost_usd`, so it could quote a different total from the Console for
|
|
24
|
+
* the same window. It now prints `spend.cost_usd`, and falls back to the trace
|
|
25
|
+
* figure only when the ledger figure is null (the ledger could not be read) or
|
|
26
|
+
* absent (an older deployment) — and says so in the output when it does.
|
|
21
27
|
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
28
|
+
* The same rule applies per group: `ledger_cost_usd` when it is there, the
|
|
29
|
+
* trace's `cost_usd` only when it is not, as the Console's breakdown does.
|
|
30
|
+
*
|
|
31
|
+
* WHEN THE TOTAL IS A FLOOR
|
|
32
|
+
*
|
|
33
|
+
* Unpriced rows sum as zero, which is the only arithmetic available and not
|
|
34
|
+
* the only truth: a total built partly from unpriced rows is a FLOOR. The
|
|
35
|
+
* unpriced count is taken from the same table as the total it qualifies —
|
|
36
|
+
* `spend.unpriced_rows` for the ledger, `summary.unpriced_requests` for the
|
|
37
|
+
* trace. A read that hit the row cap (`truncated`) is a floor for a second,
|
|
38
|
+
* independent reason.
|
|
39
|
+
*
|
|
40
|
+
* SAVINGS: TWO FIGURES, NEVER ONE
|
|
41
|
+
*
|
|
42
|
+
* The endpoint computes routing savings server-side from the priced ledger, as
|
|
43
|
+
* a `savings` object split in two: `named` (a measured saving against a model
|
|
44
|
+
* the caller asked for) and `plan_ceiling` (a counterfactual against the
|
|
45
|
+
* priciest model the plan allows, when the caller sent `auto`). This tool
|
|
46
|
+
* shows each one on its own line, labels `plan_ceiling` as a comparison rather
|
|
47
|
+
* than a saving, and never adds them together. It never computes a saving
|
|
48
|
+
* client-side from a copy of the rate card.
|
|
32
49
|
*/
|
|
33
50
|
import { z } from "zod";
|
|
34
51
|
import { apiUrl } from "../config.js";
|
|
35
52
|
import { errorFrom, gwFetch } from "../gateway.js";
|
|
36
53
|
import { money } from "../receipt.js";
|
|
37
54
|
import { baseNote, fromThrown, ok, requireKey } from "./shared.js";
|
|
38
|
-
const RANGES = ["7d", "14d", "30d", "90d"];
|
|
55
|
+
const RANGES = ["month", "7d", "14d", "30d", "90d"];
|
|
39
56
|
const GROUPS = ["day", "model", "key", "agent"];
|
|
40
57
|
export const spendInput = {
|
|
41
|
-
range: z
|
|
58
|
+
range: z
|
|
59
|
+
.enum(RANGES)
|
|
60
|
+
.optional()
|
|
61
|
+
.describe('How far back to look. "month" is the UTC calendar month to date, the Console\'s default window. Defaults to 7d.'),
|
|
42
62
|
group_by: z.enum(GROUPS).optional().describe("How to break the total down. Defaults to model."),
|
|
43
63
|
};
|
|
64
|
+
const savingsBlock = z.object({
|
|
65
|
+
requests: z.number(),
|
|
66
|
+
served_cost_usd: z.number(),
|
|
67
|
+
baseline_cost_usd: z.number(),
|
|
68
|
+
difference_usd: z.number(),
|
|
69
|
+
baseline_models: z.array(z.string()),
|
|
70
|
+
});
|
|
44
71
|
export const spendOutput = {
|
|
45
72
|
enabled: z.boolean().describe("False when request tracing is not enabled on this deployment."),
|
|
46
73
|
range: z.string(),
|
|
47
74
|
group_by: z.string(),
|
|
48
|
-
|
|
75
|
+
cost_usd: z
|
|
76
|
+
.number()
|
|
77
|
+
.nullable()
|
|
78
|
+
.describe("The spend total. The billing ledger's figure (spend.cost_usd), the same one the Console shows, unless " +
|
|
79
|
+
"cost_source says otherwise. Null when neither figure is available."),
|
|
80
|
+
cost_source: z
|
|
81
|
+
.enum(["ledger", "trace"])
|
|
82
|
+
.nullable()
|
|
83
|
+
.describe('"ledger" when cost_usd is the billing ledger\'s figure, as in the Console. "trace" when the ledger figure was ' +
|
|
84
|
+
"null or not reported and cost_usd fell back to the request trace's legacy copy (summary.cost_usd), which " +
|
|
85
|
+
"can differ from the Console."),
|
|
86
|
+
spend: z
|
|
87
|
+
.record(z.unknown())
|
|
88
|
+
.nullable()
|
|
89
|
+
.describe("The billing ledger's spend block as the API returned it. Null when the ledger could not be read."),
|
|
90
|
+
summary: z
|
|
91
|
+
.record(z.unknown())
|
|
92
|
+
.nullable()
|
|
93
|
+
.describe("Requests, errors, tokens and latency from the request trace. Its cost_usd is the legacy figure."),
|
|
49
94
|
groups: z.array(z.record(z.unknown())),
|
|
95
|
+
savings: z
|
|
96
|
+
.object({
|
|
97
|
+
named: savingsBlock.describe("Measured: requests that named a model and were routed to a cheaper one."),
|
|
98
|
+
plan_ceiling: savingsBlock.describe("A comparison, not a saving: auto requests against the priciest model the plan allows, which nobody asked for."),
|
|
99
|
+
unpriced_routed_requests: z.number(),
|
|
100
|
+
})
|
|
101
|
+
.nullable()
|
|
102
|
+
.describe("Routing savings from the ledger, as two separate figures. Never add them together. Null when not reported."),
|
|
50
103
|
is_floor: z
|
|
51
104
|
.boolean()
|
|
52
105
|
.describe("True when some rows were unpriced or the row cap bound, so the totals are a lower bound, not a total."),
|
|
@@ -54,6 +107,42 @@ export const spendOutput = {
|
|
|
54
107
|
};
|
|
55
108
|
const pad = (s, n) => (s.length >= n ? s : s + " ".repeat(n - s.length));
|
|
56
109
|
const padStart = (s, n) => (s.length >= n ? s : " ".repeat(n - s.length) + s);
|
|
110
|
+
const plural = (n, one, many = `${one}s`) => `${n} ${n === 1 ? one : many}`;
|
|
111
|
+
const isNum = (v) => typeof v === "number" && Number.isFinite(v);
|
|
112
|
+
/** A savings block as a complete, typed object, or null when it is not there. */
|
|
113
|
+
function block(b) {
|
|
114
|
+
if (!b || typeof b !== "object")
|
|
115
|
+
return null;
|
|
116
|
+
return {
|
|
117
|
+
requests: isNum(b.requests) ? b.requests : 0,
|
|
118
|
+
served_cost_usd: isNum(b.served_cost_usd) ? b.served_cost_usd : 0,
|
|
119
|
+
baseline_cost_usd: isNum(b.baseline_cost_usd) ? b.baseline_cost_usd : 0,
|
|
120
|
+
difference_usd: isNum(b.difference_usd) ? b.difference_usd : 0,
|
|
121
|
+
baseline_models: Array.isArray(b.baseline_models) ? b.baseline_models.filter((m) => typeof m === "string") : [],
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
/** The savings lines. Each figure on its own line; there is no line that adds them. */
|
|
125
|
+
function savingsLines(named, ceiling, unpriced) {
|
|
126
|
+
const against = (b, fallback) => b.baseline_models.length ? b.baseline_models.join(", ") : fallback;
|
|
127
|
+
const lines = [];
|
|
128
|
+
if (named.requests > 0) {
|
|
129
|
+
const d = named.difference_usd;
|
|
130
|
+
lines.push(d < 0
|
|
131
|
+
? `Models you named: routing cost ${money(-d)} more on ${plural(named.requests, "request")} ` +
|
|
132
|
+
`(${money(named.served_cost_usd)} instead of ${money(named.baseline_cost_usd)} on ${against(named, "the named models")}).`
|
|
133
|
+
: `Saved on models you named: ${money(d)} on ${plural(named.requests, "request")} ` +
|
|
134
|
+
`(${money(named.served_cost_usd)} instead of ${money(named.baseline_cost_usd)} on ${against(named, "the named models")}).`);
|
|
135
|
+
}
|
|
136
|
+
if (ceiling.requests > 0) {
|
|
137
|
+
lines.push(`Compared with the best model your plan allows: ${money(ceiling.difference_usd)} on ` +
|
|
138
|
+
`${plural(ceiling.requests, "auto request")}. A comparison, not a saving: nothing was named, and ` +
|
|
139
|
+
`${against(ceiling, "your plan's top model")} was not requested.`);
|
|
140
|
+
}
|
|
141
|
+
if (unpriced > 0) {
|
|
142
|
+
lines.push(`${plural(unpriced, "routed request")} carry a baseline that could not be priced and ${unpriced === 1 ? "is" : "are"} in neither figure.`);
|
|
143
|
+
}
|
|
144
|
+
return lines.length ? ["Routing savings (two separate figures, never added together):", ...lines.map((l) => ` ${l}`)] : [];
|
|
145
|
+
}
|
|
57
146
|
export async function runSpend(cfg, args) {
|
|
58
147
|
const missing = requireKey(cfg);
|
|
59
148
|
if (missing)
|
|
@@ -75,45 +164,97 @@ export async function runSpend(cfg, args) {
|
|
|
75
164
|
enabled: false,
|
|
76
165
|
range,
|
|
77
166
|
group_by: groupBy,
|
|
167
|
+
cost_usd: null,
|
|
168
|
+
cost_source: null,
|
|
169
|
+
spend: null,
|
|
78
170
|
summary: null,
|
|
79
171
|
groups: [],
|
|
172
|
+
savings: null,
|
|
80
173
|
is_floor: false,
|
|
81
174
|
message: b.message ?? null,
|
|
82
175
|
});
|
|
83
176
|
}
|
|
84
177
|
const s = b.summary ?? {};
|
|
85
|
-
const
|
|
86
|
-
|
|
178
|
+
const spend = b.spend && typeof b.spend === "object" ? b.spend : null;
|
|
179
|
+
/*
|
|
180
|
+
* The Console's figure, or the legacy one and a sentence saying so. The
|
|
181
|
+
* unpriced count and its denominator always come from the same table as
|
|
182
|
+
* the total they qualify.
|
|
183
|
+
*/
|
|
184
|
+
const fromLedger = spend !== null && isNum(spend.cost_usd);
|
|
185
|
+
const traceCost = isNum(s.cost_usd) ? s.cost_usd : null;
|
|
186
|
+
const cost = fromLedger ? spend.cost_usd : traceCost;
|
|
187
|
+
const costSource = fromLedger ? "ledger" : traceCost !== null ? "trace" : null;
|
|
188
|
+
const unpriced = fromLedger ? spend.unpriced_rows ?? 0 : s.unpriced_requests ?? 0;
|
|
189
|
+
const truncated = b.truncated === true || (fromLedger && spend.truncated === true);
|
|
87
190
|
const isFloor = unpriced > 0 || truncated;
|
|
88
191
|
const groups = b.groups ?? [];
|
|
89
|
-
|
|
90
|
-
|
|
192
|
+
/* As the Console renders it: a window in which no row carried a price has
|
|
193
|
+
an unknown total, not a total of zero. */
|
|
194
|
+
const priceable = fromLedger ? spend.rows ?? 0 : s.requests ?? 0;
|
|
195
|
+
const costText = cost === null ? "spend unknown" : priceable > 0 && unpriced >= priceable ? "unpriced" : money(cost);
|
|
196
|
+
const shown = b.range ?? range;
|
|
197
|
+
const head = `${shown === "month" ? "This month (UTC)" : `Last ${shown}`} · ${plural(s.requests ?? 0, "request")} · ` +
|
|
198
|
+
costText +
|
|
91
199
|
(s.total_tokens ? ` · ${s.total_tokens.toLocaleString("en-US")} tokens` : "") +
|
|
92
|
-
(s.errors ? ` · ${s.errors
|
|
200
|
+
(s.errors ? ` · ${plural(s.errors, "error")}` : "");
|
|
201
|
+
/* Per group, the same rule as the Console's breakdown: the ledger's spend
|
|
202
|
+
when the group carries it, the trace's copy only when it does not. */
|
|
203
|
+
const groupCost = (g) => (isNum(g.ledger_cost_usd) ? g.ledger_cost_usd : isNum(g.cost_usd) ? g.cost_usd : null);
|
|
204
|
+
const groupUnpriced = (g) => isNum(g.ledger_cost_usd) ? g.ledger_unpriced_rows ?? 0 : g.unpriced_requests ?? 0;
|
|
93
205
|
const w = Math.max(3, ...groups.map((g) => String(g.key ?? "").length));
|
|
94
206
|
const table = groups.length
|
|
95
207
|
? [
|
|
96
208
|
"",
|
|
97
209
|
`${pad(String(groupBy).toUpperCase(), w)} ${padStart("REQS", 6)} ${padStart("COST", 11)}`,
|
|
98
|
-
...groups.map((g) =>
|
|
99
|
-
|
|
100
|
-
(g.
|
|
210
|
+
...groups.map((g) => {
|
|
211
|
+
const u = groupUnpriced(g);
|
|
212
|
+
return (`${pad(String(g.key ?? ""), w)} ${padStart(String(g.requests ?? 0), 6)} ` +
|
|
213
|
+
`${padStart(money(groupCost(g)), 11)}` +
|
|
214
|
+
(u ? ` (${u} unpriced)` : ""));
|
|
215
|
+
}),
|
|
101
216
|
].join("\n")
|
|
102
217
|
: "\nNo requests in this window.";
|
|
218
|
+
const named = block(b.savings?.named);
|
|
219
|
+
const ceiling = block(b.savings?.plan_ceiling);
|
|
220
|
+
const savings = named && ceiling
|
|
221
|
+
? {
|
|
222
|
+
named,
|
|
223
|
+
plan_ceiling: ceiling,
|
|
224
|
+
unpriced_routed_requests: isNum(b.savings?.unpriced_routed_requests) ? b.savings.unpriced_routed_requests : 0,
|
|
225
|
+
}
|
|
226
|
+
: null;
|
|
227
|
+
const unmetered = b.ledger?.unmetered_requests ?? 0;
|
|
103
228
|
const notes = [
|
|
229
|
+
!fromLedger
|
|
230
|
+
? spend === null && "spend" in b
|
|
231
|
+
? "The billing ledger could not be read, so this total is the request trace's copy of each price (the legacy summary.cost_usd). It can differ from the Console's Spend."
|
|
232
|
+
: "This deployment does not report the billing ledger's spend, so this total is the request trace's copy of each price (the legacy summary.cost_usd). It can differ from the Console's Spend."
|
|
233
|
+
: null,
|
|
234
|
+
fromLedger && (spend.byok_cost_usd ?? 0) > 0
|
|
235
|
+
? `${money(spend.managed_cost_usd ?? 0)} billed by Lobstack · ${money(spend.byok_cost_usd ?? 0)} on your own provider key.`
|
|
236
|
+
: null,
|
|
104
237
|
unpriced
|
|
105
|
-
? `${unpriced
|
|
238
|
+
? `${plural(unpriced, fromLedger ? "row" : "request")} could not be priced. Those sum as zero, so the total above is a floor, not a total.`
|
|
106
239
|
: null,
|
|
107
240
|
truncated ? "The row cap bound on this range, so older requests are not counted here." : null,
|
|
241
|
+
fromLedger && unmetered > 0
|
|
242
|
+
? `${plural(unmetered, "served request")} ${unmetered === 1 ? "has" : "have"} no billing ledger row, so ${unmetered === 1 ? "it is" : "they are"} not in this total.`
|
|
243
|
+
: null,
|
|
108
244
|
typeof s.p95_latency_ms === "number" ? `p50 ${s.p50_latency_ms ?? "—"}ms · p95 ${s.p95_latency_ms}ms` : null,
|
|
109
245
|
baseNote(cfg),
|
|
110
|
-
].filter((n) => n
|
|
111
|
-
|
|
246
|
+
].filter((n) => typeof n === "string");
|
|
247
|
+
const saved = savings ? savingsLines(savings.named, savings.plan_ceiling, savings.unpriced_routed_requests) : [];
|
|
248
|
+
return ok([head + table + (saved.length ? "\n\n" + saved.join("\n") : "") + (notes.length ? "\n\n" + notes.join("\n") : "")], {
|
|
112
249
|
enabled: true,
|
|
113
250
|
range: b.range ?? range,
|
|
114
251
|
group_by: b.group_by ?? groupBy,
|
|
252
|
+
cost_usd: cost,
|
|
253
|
+
cost_source: costSource,
|
|
254
|
+
spend: (spend ?? null),
|
|
115
255
|
summary: (b.summary ?? null),
|
|
116
256
|
groups: groups,
|
|
257
|
+
savings,
|
|
117
258
|
is_floor: isFloor,
|
|
118
259
|
message: null,
|
|
119
260
|
});
|