@lobstack-ai/mcp 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +238 -0
- package/dist/config.d.ts +75 -0
- package/dist/config.js +111 -0
- package/dist/gateway.d.ts +61 -0
- package/dist/gateway.js +126 -0
- package/dist/index.d.ts +14 -0
- package/dist/index.js +28 -0
- package/dist/receipt.d.ts +84 -0
- package/dist/receipt.js +122 -0
- package/dist/server.d.ts +44 -0
- package/dist/server.js +106 -0
- package/dist/sse.d.ts +50 -0
- package/dist/sse.js +112 -0
- package/dist/tools/chat.d.ts +133 -0
- package/dist/tools/chat.js +177 -0
- package/dist/tools/models.d.ts +56 -0
- package/dist/tools/models.js +90 -0
- package/dist/tools/route-preview.d.ts +75 -0
- package/dist/tools/route-preview.js +155 -0
- package/dist/tools/shared.d.ts +38 -0
- package/dist/tools/shared.js +64 -0
- package/dist/tools/spend.d.ts +49 -0
- package/dist/tools/spend.js +121 -0
- package/package.json +65 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lobstack_chat — one completion, and what it cost.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS STREAMS WHEN NOTHING IS STREAMED TO
|
|
5
|
+
*
|
|
6
|
+
* An MCP tool result is a single message; there is no partial delivery to an
|
|
7
|
+
* agent mid-call. So the whole answer is accumulated here before it is
|
|
8
|
+
* returned, and `stream: true` looks pointless.
|
|
9
|
+
*
|
|
10
|
+
* It is not. On the buffered path the Gateway reports the price in response
|
|
11
|
+
* HEADERS, and on the streamed path it reports it in the trailing SSE frame as
|
|
12
|
+
* `x_lobstack`. The frame is the better source: the streaming path meters
|
|
13
|
+
* *before* emitting that frame, so the figure is the ledger's, not an estimate,
|
|
14
|
+
* and it arrives as one structured object rather than eight headers, three of
|
|
15
|
+
* which encode "unpriced" as an empty string. Requesting SSE and reading it to
|
|
16
|
+
* the end is how this tool returns a receipt it did not compute itself.
|
|
17
|
+
*
|
|
18
|
+
* `stream_options.include_usage` is not optional: without it the Gateway has no
|
|
19
|
+
* trailing frame to attach `x_lobstack` to, and the price never arrives.
|
|
20
|
+
*/
|
|
21
|
+
import { z } from "zod";
|
|
22
|
+
import { gatewayUrl } from "../config.js";
|
|
23
|
+
import { gwFetch, errorFrom, quotaFromHeaders, describeQuota } from "../gateway.js";
|
|
24
|
+
import { consume } from "../sse.js";
|
|
25
|
+
import { describeReceipt, money, parseReceipt, savingsLabel } from "../receipt.js";
|
|
26
|
+
import { baseNote, droppedParams, failure, fromThrown, ok, requireKey } from "./shared.js";
|
|
27
|
+
const Message = z.object({
|
|
28
|
+
role: z.enum(["system", "user", "assistant"]),
|
|
29
|
+
content: z.string(),
|
|
30
|
+
});
|
|
31
|
+
export const chatInput = {
|
|
32
|
+
prompt: z
|
|
33
|
+
.string()
|
|
34
|
+
.min(1)
|
|
35
|
+
.optional()
|
|
36
|
+
.describe("A single user message. Use this or `messages`, not both."),
|
|
37
|
+
messages: z
|
|
38
|
+
.array(Message)
|
|
39
|
+
.min(1)
|
|
40
|
+
.optional()
|
|
41
|
+
.describe("A full conversation, OpenAI-shaped. Use this or `prompt`, not both."),
|
|
42
|
+
model: z
|
|
43
|
+
.string()
|
|
44
|
+
.optional()
|
|
45
|
+
.describe('Lobstack model key, e.g. "claude-sonnet-5". Defaults to "auto", which lets Token Intelligence pick the cheapest model that can handle the prompt. lobstack_models lists the keys.'),
|
|
46
|
+
system: z.string().optional().describe("System prompt, prepended to the conversation."),
|
|
47
|
+
max_tokens: z.number().int().positive().max(200_000).optional().describe("Cap on the reply length."),
|
|
48
|
+
temperature: z
|
|
49
|
+
.number()
|
|
50
|
+
.min(0)
|
|
51
|
+
.max(2)
|
|
52
|
+
.optional()
|
|
53
|
+
.describe("Sampling temperature. Some served models do not accept it; the receipt says when it was dropped."),
|
|
54
|
+
};
|
|
55
|
+
export const chatOutput = {
|
|
56
|
+
text: z.string().describe("The assistant's reply."),
|
|
57
|
+
model: z.object({
|
|
58
|
+
requested: z.string().nullable(),
|
|
59
|
+
served: z.string().nullable().describe("The model that actually answered."),
|
|
60
|
+
routed: z.boolean().nullable().describe("True when the served model differs from the one requested."),
|
|
61
|
+
}),
|
|
62
|
+
usage: z
|
|
63
|
+
.object({ prompt_tokens: z.number(), completion_tokens: z.number(), total_tokens: z.number() })
|
|
64
|
+
.nullable(),
|
|
65
|
+
receipt: z
|
|
66
|
+
.object({
|
|
67
|
+
request_id: z.string().nullable(),
|
|
68
|
+
cost_usd: z
|
|
69
|
+
.number()
|
|
70
|
+
.nullable()
|
|
71
|
+
.describe("USD the caller owes. NULL — never 0 — when the gateway could not price the call."),
|
|
72
|
+
cost_display: z.string().describe('Human form. "unpriced" when cost_usd is null.'),
|
|
73
|
+
priced: z.boolean(),
|
|
74
|
+
savings: z
|
|
75
|
+
.object({
|
|
76
|
+
amount_usd: z.number(),
|
|
77
|
+
label: z.string().describe('"saved" for a like-for-like comparison, "vs ceiling" otherwise.'),
|
|
78
|
+
named: z
|
|
79
|
+
.boolean()
|
|
80
|
+
.describe("True only when the caller asked for baseline_model and got something cheaper."),
|
|
81
|
+
baseline_model: z.string().nullable(),
|
|
82
|
+
baseline_reason: z.string().nullable(),
|
|
83
|
+
})
|
|
84
|
+
.nullable(),
|
|
85
|
+
})
|
|
86
|
+
.nullable()
|
|
87
|
+
.describe("Null when the endpoint sent no receipt at all."),
|
|
88
|
+
quota: z.record(z.unknown()).nullable().describe("Allowance remaining, as the gateway reported it."),
|
|
89
|
+
dropped_params: z.array(z.string()),
|
|
90
|
+
};
|
|
91
|
+
export async function runChat(cfg, args) {
|
|
92
|
+
const missing = requireKey(cfg);
|
|
93
|
+
if (missing)
|
|
94
|
+
return missing;
|
|
95
|
+
if (!args.prompt && !args.messages?.length) {
|
|
96
|
+
return failure(cfg, "Give either `prompt` (a single message) or `messages` (a conversation).");
|
|
97
|
+
}
|
|
98
|
+
if (args.prompt && args.messages?.length) {
|
|
99
|
+
return failure(cfg, "Give either `prompt` or `messages`, not both — it is ambiguous which one to send.");
|
|
100
|
+
}
|
|
101
|
+
const messages = [
|
|
102
|
+
...(args.system ? [{ role: "system", content: args.system }] : []),
|
|
103
|
+
...(args.messages ?? [{ role: "user", content: args.prompt }]),
|
|
104
|
+
];
|
|
105
|
+
const requestedModel = args.model?.trim() || "auto";
|
|
106
|
+
try {
|
|
107
|
+
const res = await gwFetch(cfg, gatewayUrl(cfg.base.origin, "/chat/completions"), {
|
|
108
|
+
method: "POST",
|
|
109
|
+
body: JSON.stringify({
|
|
110
|
+
model: requestedModel,
|
|
111
|
+
messages,
|
|
112
|
+
stream: true,
|
|
113
|
+
// Without this there is no trailing frame, and so no price.
|
|
114
|
+
stream_options: { include_usage: true },
|
|
115
|
+
...(args.max_tokens !== undefined ? { max_tokens: args.max_tokens } : {}),
|
|
116
|
+
...(args.temperature !== undefined ? { temperature: args.temperature } : {}),
|
|
117
|
+
}),
|
|
118
|
+
});
|
|
119
|
+
if (!res.ok || !res.body) {
|
|
120
|
+
throw await errorFrom(cfg, res, res.status === 402
|
|
121
|
+
? "The allowance on this key is exhausted. Top up or wait for the reset in Console → Billing."
|
|
122
|
+
: res.status === 403
|
|
123
|
+
? 'This key needs the "inference" scope. Mint one in Console → API keys.'
|
|
124
|
+
: undefined);
|
|
125
|
+
}
|
|
126
|
+
const quota = quotaFromHeaders(res.headers);
|
|
127
|
+
const dropped = droppedParams(res.headers);
|
|
128
|
+
const stream = await consume(res.body);
|
|
129
|
+
const receipt = parseReceipt(stream.receipt);
|
|
130
|
+
const saving = savingsLabel(receipt);
|
|
131
|
+
const structured = {
|
|
132
|
+
text: stream.text,
|
|
133
|
+
model: {
|
|
134
|
+
requested: receipt?.requested_model ?? requestedModel,
|
|
135
|
+
served: receipt?.served_model ?? stream.model,
|
|
136
|
+
routed: receipt?.routed ?? null,
|
|
137
|
+
},
|
|
138
|
+
usage: stream.usage,
|
|
139
|
+
receipt: receipt
|
|
140
|
+
? {
|
|
141
|
+
request_id: receipt.request_id,
|
|
142
|
+
// Null stays null all the way out. A consumer that wants a number
|
|
143
|
+
// has to decide for itself what "we could not price this" means.
|
|
144
|
+
cost_usd: receipt.cost_usd,
|
|
145
|
+
cost_display: money(receipt.cost_usd),
|
|
146
|
+
priced: receipt.priced,
|
|
147
|
+
savings: saving
|
|
148
|
+
? {
|
|
149
|
+
amount_usd: saving.amount,
|
|
150
|
+
label: saving.label,
|
|
151
|
+
named: saving.named,
|
|
152
|
+
baseline_model: saving.baseline_model,
|
|
153
|
+
baseline_reason: saving.baseline_reason,
|
|
154
|
+
}
|
|
155
|
+
: null,
|
|
156
|
+
}
|
|
157
|
+
: null,
|
|
158
|
+
quota: quota ? quota : null,
|
|
159
|
+
dropped_params: dropped,
|
|
160
|
+
};
|
|
161
|
+
// Two blocks, not one. The answer is the thing that was asked for; the
|
|
162
|
+
// receipt is a fact about the call. Concatenating them puts a price line
|
|
163
|
+
// into whatever consumes the answer — the same reason the CLI puts the
|
|
164
|
+
// receipt on stderr.
|
|
165
|
+
const receiptBlock = [
|
|
166
|
+
describeReceipt({ receipt, usage: stream.usage, model: stream.model, droppedParams: dropped }),
|
|
167
|
+
describeQuota(quota),
|
|
168
|
+
baseNote(cfg),
|
|
169
|
+
]
|
|
170
|
+
.filter(Boolean)
|
|
171
|
+
.join("\n");
|
|
172
|
+
return ok([stream.text, receiptBlock], structured);
|
|
173
|
+
}
|
|
174
|
+
catch (e) {
|
|
175
|
+
return fromThrown(cfg, e);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lobstack_models — the catalogue, with prices.
|
|
3
|
+
*
|
|
4
|
+
* `GET /api/gateway/v1/models` is OpenAI-shaped with Lobstack extensions on
|
|
5
|
+
* each row: `tier`, `context_window`, `price_per_mtok` and `managed`. It takes
|
|
6
|
+
* no credential on lobstack.ai, and this tool sends one only if the server was
|
|
7
|
+
* given one — a self-hosted deployment is entitled to put auth in front of it.
|
|
8
|
+
*
|
|
9
|
+
* A model whose price the registry does not know arrives with an empty or
|
|
10
|
+
* partial `price_per_mtok`. It is rendered as "—", never as "$0": a catalogue
|
|
11
|
+
* that shows an unpriced model as free teaches the reader that some of our
|
|
12
|
+
* routing is free, which is the same lie as a $0.00 receipt one layer up.
|
|
13
|
+
*/
|
|
14
|
+
import { z } from "zod";
|
|
15
|
+
import type { Config } from "../config.js";
|
|
16
|
+
import { type ToolResult } from "./shared.js";
|
|
17
|
+
export declare const modelsInput: {
|
|
18
|
+
tier: z.ZodOptional<z.ZodEnum<["nano", "small", "standard", "premium", "flagship"]>>;
|
|
19
|
+
provider: z.ZodOptional<z.ZodString>;
|
|
20
|
+
};
|
|
21
|
+
export declare const modelsOutput: {
|
|
22
|
+
count: z.ZodNumber;
|
|
23
|
+
unpriced_count: z.ZodNumber;
|
|
24
|
+
models: z.ZodArray<z.ZodObject<{
|
|
25
|
+
id: z.ZodString;
|
|
26
|
+
label: z.ZodNullable<z.ZodString>;
|
|
27
|
+
tier: z.ZodNullable<z.ZodString>;
|
|
28
|
+
provider: z.ZodNullable<z.ZodString>;
|
|
29
|
+
context_window: z.ZodNullable<z.ZodNumber>;
|
|
30
|
+
input_usd_per_mtok: z.ZodNullable<z.ZodNumber>;
|
|
31
|
+
output_usd_per_mtok: z.ZodNullable<z.ZodNumber>;
|
|
32
|
+
managed: z.ZodNullable<z.ZodBoolean>;
|
|
33
|
+
}, "strip", z.ZodTypeAny, {
|
|
34
|
+
label: string | null;
|
|
35
|
+
id: string;
|
|
36
|
+
tier: string | null;
|
|
37
|
+
provider: string | null;
|
|
38
|
+
context_window: number | null;
|
|
39
|
+
input_usd_per_mtok: number | null;
|
|
40
|
+
output_usd_per_mtok: number | null;
|
|
41
|
+
managed: boolean | null;
|
|
42
|
+
}, {
|
|
43
|
+
label: string | null;
|
|
44
|
+
id: string;
|
|
45
|
+
tier: string | null;
|
|
46
|
+
provider: string | null;
|
|
47
|
+
context_window: number | null;
|
|
48
|
+
input_usd_per_mtok: number | null;
|
|
49
|
+
output_usd_per_mtok: number | null;
|
|
50
|
+
managed: boolean | null;
|
|
51
|
+
}>, "many">;
|
|
52
|
+
};
|
|
53
|
+
export declare function runModels(cfg: Config, args: {
|
|
54
|
+
tier?: string;
|
|
55
|
+
provider?: string;
|
|
56
|
+
}): Promise<ToolResult>;
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lobstack_models — the catalogue, with prices.
|
|
3
|
+
*
|
|
4
|
+
* `GET /api/gateway/v1/models` is OpenAI-shaped with Lobstack extensions on
|
|
5
|
+
* each row: `tier`, `context_window`, `price_per_mtok` and `managed`. It takes
|
|
6
|
+
* no credential on lobstack.ai, and this tool sends one only if the server was
|
|
7
|
+
* given one — a self-hosted deployment is entitled to put auth in front of it.
|
|
8
|
+
*
|
|
9
|
+
* A model whose price the registry does not know arrives with an empty or
|
|
10
|
+
* partial `price_per_mtok`. It is rendered as "—", never as "$0": a catalogue
|
|
11
|
+
* that shows an unpriced model as free teaches the reader that some of our
|
|
12
|
+
* routing is free, which is the same lie as a $0.00 receipt one layer up.
|
|
13
|
+
*/
|
|
14
|
+
import { z } from "zod";
|
|
15
|
+
import { gatewayUrl } from "../config.js";
|
|
16
|
+
import { errorFrom, gwFetch } from "../gateway.js";
|
|
17
|
+
import { asMoney } from "../receipt.js";
|
|
18
|
+
import { baseNote, fromThrown, ok } from "./shared.js";
|
|
19
|
+
const TIERS = ["nano", "small", "standard", "premium", "flagship"];
|
|
20
|
+
export const modelsInput = {
|
|
21
|
+
tier: z
|
|
22
|
+
.enum(TIERS)
|
|
23
|
+
.optional()
|
|
24
|
+
.describe("Only models in this capability tier. Token Intelligence walks these from cheapest upward."),
|
|
25
|
+
provider: z
|
|
26
|
+
.string()
|
|
27
|
+
.optional()
|
|
28
|
+
.describe('Only models from this provider, e.g. "anthropic", "openai", "google".'),
|
|
29
|
+
};
|
|
30
|
+
const ModelRow = z.object({
|
|
31
|
+
id: z.string(),
|
|
32
|
+
label: z.string().nullable(),
|
|
33
|
+
tier: z.string().nullable(),
|
|
34
|
+
provider: z.string().nullable(),
|
|
35
|
+
context_window: z.number().nullable(),
|
|
36
|
+
input_usd_per_mtok: z.number().nullable().describe("Null when the registry cannot price this model."),
|
|
37
|
+
output_usd_per_mtok: z.number().nullable(),
|
|
38
|
+
managed: z.boolean().nullable(),
|
|
39
|
+
});
|
|
40
|
+
export const modelsOutput = {
|
|
41
|
+
count: z.number(),
|
|
42
|
+
unpriced_count: z.number().describe("Models the registry could not price. Their prices are null, not zero."),
|
|
43
|
+
models: z.array(ModelRow),
|
|
44
|
+
};
|
|
45
|
+
const pad = (s, n) => (s.length >= n ? s : s + " ".repeat(n - s.length));
|
|
46
|
+
const priceCell = (v) => (v === null ? "—" : `$${v}`);
|
|
47
|
+
export async function runModels(cfg, args) {
|
|
48
|
+
try {
|
|
49
|
+
const res = await gwFetch(cfg, gatewayUrl(cfg.base.origin, "/models"));
|
|
50
|
+
if (!res.ok)
|
|
51
|
+
throw await errorFrom(cfg, res);
|
|
52
|
+
const body = (await res.json());
|
|
53
|
+
const all = (body.data ?? []).map((m) => ({
|
|
54
|
+
id: String(m.id ?? ""),
|
|
55
|
+
label: m.label ?? null,
|
|
56
|
+
tier: m.tier ?? null,
|
|
57
|
+
provider: m.owned_by ?? null,
|
|
58
|
+
context_window: typeof m.context_window === "number" ? m.context_window : null,
|
|
59
|
+
input_usd_per_mtok: asMoney(m.price_per_mtok?.input),
|
|
60
|
+
output_usd_per_mtok: asMoney(m.price_per_mtok?.output),
|
|
61
|
+
managed: typeof m.managed === "boolean" ? m.managed : null,
|
|
62
|
+
}));
|
|
63
|
+
const rows = all.filter((m) => (!args.tier || m.tier === args.tier) &&
|
|
64
|
+
(!args.provider || (m.provider ?? "").toLowerCase() === args.provider.toLowerCase()));
|
|
65
|
+
const unpriced = rows.filter((m) => m.input_usd_per_mtok === null || m.output_usd_per_mtok === null).length;
|
|
66
|
+
const w = Math.max(5, ...rows.map((m) => m.id.length));
|
|
67
|
+
const lines = [
|
|
68
|
+
`${pad("MODEL", w)} ${pad("TIER", 9)} ${pad("PROVIDER", 10)} ${pad("IN/Mtok", 9)} OUT/Mtok`,
|
|
69
|
+
...rows.map((m) => `${pad(m.id, w)} ${pad(m.tier ?? "—", 9)} ${pad(m.provider ?? "—", 10)} ` +
|
|
70
|
+
`${pad(priceCell(m.input_usd_per_mtok), 9)} ${priceCell(m.output_usd_per_mtok)}`),
|
|
71
|
+
];
|
|
72
|
+
const footer = [
|
|
73
|
+
"",
|
|
74
|
+
`${rows.length} model${rows.length === 1 ? "" : "s"}${rows.length !== all.length ? ` of ${all.length}` : ""}. ` +
|
|
75
|
+
'Send model "auto" to lobstack_chat and the router picks one, then tells you which.',
|
|
76
|
+
unpriced ? `${unpriced} of them carry no price in the registry; they are shown as — and are not free.` : null,
|
|
77
|
+
baseNote(cfg),
|
|
78
|
+
]
|
|
79
|
+
.filter((l) => l !== null)
|
|
80
|
+
.join("\n");
|
|
81
|
+
return ok([lines.join("\n") + "\n" + footer], {
|
|
82
|
+
count: rows.length,
|
|
83
|
+
unpriced_count: unpriced,
|
|
84
|
+
models: rows,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
catch (e) {
|
|
88
|
+
return fromThrown(cfg, e);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lobstack_route_preview — where a prompt would go, and what it would cost,
|
|
3
|
+
* without spending a token and without a key.
|
|
4
|
+
*
|
|
5
|
+
* `POST /api/gateway/v1/route-preview` is unauthenticated on purpose: it runs
|
|
6
|
+
* no inference, calls no provider, writes nothing and reads no per-user state.
|
|
7
|
+
* It runs the same `selectModel()` and the same price registry the paid path
|
|
8
|
+
* uses, so the answer is the decision rather than a simulation of it.
|
|
9
|
+
*
|
|
10
|
+
* This tool therefore sends NO Authorization header — `anonymous: true` below.
|
|
11
|
+
* That is not an oversight, it is the point: an MCP client with this server
|
|
12
|
+
* installed and no key at all can still answer "which model would this go to
|
|
13
|
+
* and what would it cost", which is the whole product visible before anybody
|
|
14
|
+
* signs up.
|
|
15
|
+
*
|
|
16
|
+
* Two honesty constraints carried over from the endpoint.
|
|
17
|
+
*
|
|
18
|
+
* Token counts are ESTIMATES — roughly four characters per token, with output
|
|
19
|
+
* assumed at half the prompt unless the caller says otherwise. The response
|
|
20
|
+
* says so in `token_estimate.estimated` and so does the text here. The billed
|
|
21
|
+
* figure always comes from the provider's own usage block on the real call.
|
|
22
|
+
*
|
|
23
|
+
* `baseline` exists only when the caller named a model and the router went
|
|
24
|
+
* somewhere else. On "auto" it is null, because there is no model anybody asked
|
|
25
|
+
* for to compare against, and inventing one — against the flagship, say — is
|
|
26
|
+
* how every savings claim in this category gets manufactured.
|
|
27
|
+
*/
|
|
28
|
+
import { z } from "zod";
|
|
29
|
+
import type { Config } from "../config.js";
|
|
30
|
+
import { type ToolResult } from "./shared.js";
|
|
31
|
+
export declare const routePreviewInput: {
|
|
32
|
+
prompt: z.ZodString;
|
|
33
|
+
requested_model: z.ZodOptional<z.ZodString>;
|
|
34
|
+
plan_tier: z.ZodOptional<z.ZodString>;
|
|
35
|
+
expected_output_tokens: z.ZodOptional<z.ZodNumber>;
|
|
36
|
+
conversation_length: z.ZodOptional<z.ZodNumber>;
|
|
37
|
+
};
|
|
38
|
+
export declare const routePreviewOutput: {
|
|
39
|
+
requested_model: z.ZodString;
|
|
40
|
+
served_model: z.ZodString;
|
|
41
|
+
served_label: z.ZodNullable<z.ZodString>;
|
|
42
|
+
provider: z.ZodNullable<z.ZodString>;
|
|
43
|
+
tier: z.ZodNullable<z.ZodString>;
|
|
44
|
+
complexity: z.ZodNullable<z.ZodNumber>;
|
|
45
|
+
routed: z.ZodNullable<z.ZodBoolean>;
|
|
46
|
+
reason: z.ZodNullable<z.ZodString>;
|
|
47
|
+
estimated_cost_usd: z.ZodNullable<z.ZodNumber>;
|
|
48
|
+
estimated_cost_display: z.ZodString;
|
|
49
|
+
token_estimate: z.ZodNullable<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
50
|
+
baseline: z.ZodNullable<z.ZodObject<{
|
|
51
|
+
model: z.ZodString;
|
|
52
|
+
label: z.ZodNullable<z.ZodString>;
|
|
53
|
+
cost_usd: z.ZodNullable<z.ZodNumber>;
|
|
54
|
+
saving_usd: z.ZodNullable<z.ZodNumber>;
|
|
55
|
+
}, "strip", z.ZodTypeAny, {
|
|
56
|
+
model: string;
|
|
57
|
+
cost_usd: number | null;
|
|
58
|
+
label: string | null;
|
|
59
|
+
saving_usd: number | null;
|
|
60
|
+
}, {
|
|
61
|
+
model: string;
|
|
62
|
+
cost_usd: number | null;
|
|
63
|
+
label: string | null;
|
|
64
|
+
saving_usd: number | null;
|
|
65
|
+
}>>;
|
|
66
|
+
managed_key_configured: z.ZodNullable<z.ZodBoolean>;
|
|
67
|
+
estimated: z.ZodLiteral<true>;
|
|
68
|
+
};
|
|
69
|
+
export declare function runRoutePreview(cfg: Config, args: {
|
|
70
|
+
prompt: string;
|
|
71
|
+
requested_model?: string;
|
|
72
|
+
plan_tier?: string;
|
|
73
|
+
expected_output_tokens?: number;
|
|
74
|
+
conversation_length?: number;
|
|
75
|
+
}): Promise<ToolResult>;
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lobstack_route_preview — where a prompt would go, and what it would cost,
|
|
3
|
+
* without spending a token and without a key.
|
|
4
|
+
*
|
|
5
|
+
* `POST /api/gateway/v1/route-preview` is unauthenticated on purpose: it runs
|
|
6
|
+
* no inference, calls no provider, writes nothing and reads no per-user state.
|
|
7
|
+
* It runs the same `selectModel()` and the same price registry the paid path
|
|
8
|
+
* uses, so the answer is the decision rather than a simulation of it.
|
|
9
|
+
*
|
|
10
|
+
* This tool therefore sends NO Authorization header — `anonymous: true` below.
|
|
11
|
+
* That is not an oversight, it is the point: an MCP client with this server
|
|
12
|
+
* installed and no key at all can still answer "which model would this go to
|
|
13
|
+
* and what would it cost", which is the whole product visible before anybody
|
|
14
|
+
* signs up.
|
|
15
|
+
*
|
|
16
|
+
* Two honesty constraints carried over from the endpoint.
|
|
17
|
+
*
|
|
18
|
+
* Token counts are ESTIMATES — roughly four characters per token, with output
|
|
19
|
+
* assumed at half the prompt unless the caller says otherwise. The response
|
|
20
|
+
* says so in `token_estimate.estimated` and so does the text here. The billed
|
|
21
|
+
* figure always comes from the provider's own usage block on the real call.
|
|
22
|
+
*
|
|
23
|
+
* `baseline` exists only when the caller named a model and the router went
|
|
24
|
+
* somewhere else. On "auto" it is null, because there is no model anybody asked
|
|
25
|
+
* for to compare against, and inventing one — against the flagship, say — is
|
|
26
|
+
* how every savings claim in this category gets manufactured.
|
|
27
|
+
*/
|
|
28
|
+
import { z } from "zod";
|
|
29
|
+
import { gatewayUrl } from "../config.js";
|
|
30
|
+
import { errorFrom, gwFetch } from "../gateway.js";
|
|
31
|
+
import { asMoney, money } from "../receipt.js";
|
|
32
|
+
import { baseNote, fromThrown, ok } from "./shared.js";
|
|
33
|
+
export const routePreviewInput = {
|
|
34
|
+
prompt: z.string().min(1).max(8000).describe("The prompt to score. Scored, never sent to a model."),
|
|
35
|
+
requested_model: z
|
|
36
|
+
.string()
|
|
37
|
+
.optional()
|
|
38
|
+
.describe('A model key to compare against, e.g. "claude-opus-5". Defaults to "auto".'),
|
|
39
|
+
plan_tier: z
|
|
40
|
+
.string()
|
|
41
|
+
.optional()
|
|
42
|
+
.describe("Plan id, which sets the ceiling the router may reach. Defaults to the pro ceiling."),
|
|
43
|
+
expected_output_tokens: z
|
|
44
|
+
.number()
|
|
45
|
+
.int()
|
|
46
|
+
.positive()
|
|
47
|
+
.max(200_000)
|
|
48
|
+
.optional()
|
|
49
|
+
.describe("How long the reply is expected to be. Defaults to half the prompt, which is the common chat ratio."),
|
|
50
|
+
conversation_length: z
|
|
51
|
+
.number()
|
|
52
|
+
.int()
|
|
53
|
+
.nonnegative()
|
|
54
|
+
.optional()
|
|
55
|
+
.describe("Messages already in the conversation; it feeds the complexity score."),
|
|
56
|
+
};
|
|
57
|
+
export const routePreviewOutput = {
|
|
58
|
+
requested_model: z.string(),
|
|
59
|
+
served_model: z.string().describe("The model this prompt would actually be routed to."),
|
|
60
|
+
served_label: z.string().nullable(),
|
|
61
|
+
provider: z.string().nullable(),
|
|
62
|
+
tier: z.string().nullable(),
|
|
63
|
+
complexity: z.number().nullable(),
|
|
64
|
+
routed: z.boolean().nullable().describe("True when the router would serve something other than what was requested."),
|
|
65
|
+
reason: z.string().nullable(),
|
|
66
|
+
estimated_cost_usd: z.number().nullable().describe("Null when the registry cannot price the model."),
|
|
67
|
+
estimated_cost_display: z.string(),
|
|
68
|
+
token_estimate: z.record(z.unknown()).nullable(),
|
|
69
|
+
baseline: z
|
|
70
|
+
.object({
|
|
71
|
+
model: z.string(),
|
|
72
|
+
label: z.string().nullable(),
|
|
73
|
+
cost_usd: z.number().nullable(),
|
|
74
|
+
saving_usd: z.number().nullable(),
|
|
75
|
+
})
|
|
76
|
+
.nullable()
|
|
77
|
+
.describe('Only present when a model was named and the router moved away from it — the "named" case.'),
|
|
78
|
+
managed_key_configured: z.boolean().nullable(),
|
|
79
|
+
estimated: z.literal(true).describe("Always true. These are estimates, not a bill."),
|
|
80
|
+
};
|
|
81
|
+
export async function runRoutePreview(cfg, args) {
|
|
82
|
+
try {
|
|
83
|
+
const res = await gwFetch(cfg, gatewayUrl(cfg.base.origin, "/route-preview"), {
|
|
84
|
+
method: "POST",
|
|
85
|
+
// No credential. This endpoint needs none, and sending one anyway would
|
|
86
|
+
// turn a public demonstration into an authenticated call for no reason.
|
|
87
|
+
anonymous: true,
|
|
88
|
+
body: JSON.stringify({
|
|
89
|
+
prompt: args.prompt,
|
|
90
|
+
...(args.requested_model ? { requested_model: args.requested_model } : {}),
|
|
91
|
+
...(args.plan_tier ? { plan_tier: args.plan_tier } : {}),
|
|
92
|
+
...(args.expected_output_tokens ? { expected_output_tokens: args.expected_output_tokens } : {}),
|
|
93
|
+
...(args.conversation_length !== undefined ? { conversation_length: args.conversation_length } : {}),
|
|
94
|
+
}),
|
|
95
|
+
});
|
|
96
|
+
if (!res.ok)
|
|
97
|
+
throw await errorFrom(cfg, res);
|
|
98
|
+
const b = (await res.json());
|
|
99
|
+
const cost = asMoney(b.cost_usd);
|
|
100
|
+
const served = b.model?.key ?? "unknown";
|
|
101
|
+
const baselineCost = asMoney(b.baseline?.cost_usd);
|
|
102
|
+
const baselineSaving = asMoney(b.baseline?.saving_usd);
|
|
103
|
+
const structured = {
|
|
104
|
+
requested_model: b.requested_model ?? args.requested_model ?? "auto",
|
|
105
|
+
served_model: served,
|
|
106
|
+
served_label: b.model?.label ?? null,
|
|
107
|
+
provider: b.model?.provider ?? null,
|
|
108
|
+
tier: b.tier ?? null,
|
|
109
|
+
complexity: typeof b.complexity === "number" ? b.complexity : null,
|
|
110
|
+
routed: typeof b.routed === "boolean" ? b.routed : null,
|
|
111
|
+
reason: b.reason ?? null,
|
|
112
|
+
estimated_cost_usd: cost,
|
|
113
|
+
estimated_cost_display: money(cost),
|
|
114
|
+
token_estimate: b.token_estimate ?? null,
|
|
115
|
+
baseline: b.baseline?.model
|
|
116
|
+
? {
|
|
117
|
+
model: b.baseline.model,
|
|
118
|
+
label: b.baseline.label ?? null,
|
|
119
|
+
cost_usd: baselineCost,
|
|
120
|
+
saving_usd: baselineSaving,
|
|
121
|
+
}
|
|
122
|
+
: null,
|
|
123
|
+
managed_key_configured: b.model?.managed_key_configured ?? null,
|
|
124
|
+
estimated: true,
|
|
125
|
+
};
|
|
126
|
+
const est = b.token_estimate;
|
|
127
|
+
const lines = [
|
|
128
|
+
`This prompt would be served by ${served}${b.model?.label ? ` (${b.model.label})` : ""}` +
|
|
129
|
+
`${b.tier ? `, ${b.tier} tier` : ""}${b.model?.provider ? `, via ${b.model.provider}` : ""}.`,
|
|
130
|
+
b.reason ? ` why: ${b.reason}${typeof b.complexity === "number" ? ` (complexity ${b.complexity})` : ""}` : null,
|
|
131
|
+
` estimated cost ${money(cost)}` +
|
|
132
|
+
(est?.input !== undefined ? ` on ~${est.input} in / ~${est.output} out tokens` : ""),
|
|
133
|
+
// A saving is only ever quoted here for a model the caller named. The
|
|
134
|
+
// endpoint returns null on "auto" for exactly that reason.
|
|
135
|
+
structured.baseline && baselineSaving !== null && baselineSaving > 0
|
|
136
|
+
? ` ${money(baselineSaving)} cheaper than ${structured.baseline.model}, which you named` +
|
|
137
|
+
(baselineCost !== null ? ` and which would have cost ${money(baselineCost)}` : "")
|
|
138
|
+
: null,
|
|
139
|
+
b.model?.managed_key_configured === false
|
|
140
|
+
? ` note: no managed provider key for ${b.model?.provider ?? "this provider"} is configured on this deployment, so a real call may route elsewhere`
|
|
141
|
+
: null,
|
|
142
|
+
"",
|
|
143
|
+
` Estimated, not billed. ${est?.method ? `Tokens: ${est.method}. ` : ""}` +
|
|
144
|
+
"The charged figure comes from the provider's own usage block on the real request.",
|
|
145
|
+
" This preview needs no API key.",
|
|
146
|
+
baseNote(cfg),
|
|
147
|
+
]
|
|
148
|
+
.filter((l) => l !== null)
|
|
149
|
+
.join("\n");
|
|
150
|
+
return ok([lines], structured);
|
|
151
|
+
}
|
|
152
|
+
catch (e) {
|
|
153
|
+
return fromThrown(cfg, e);
|
|
154
|
+
}
|
|
155
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turning a failure into something an agent can act on.
|
|
3
|
+
*
|
|
4
|
+
* Every tool in this repo answers a failure the same way: `isError: true` with
|
|
5
|
+
* a sentence and, where there is one, a next step. Not a thrown McpError —
|
|
6
|
+
* a protocol-level error tells the client the call was malformed, which is a
|
|
7
|
+
* different claim from "your key lacks a scope", and clients surface the two
|
|
8
|
+
* very differently.
|
|
9
|
+
*
|
|
10
|
+
* Everything that goes through here is scrubbed. The key is in this process's
|
|
11
|
+
* environment and nowhere else; it must not turn up in a tool result, which is
|
|
12
|
+
* transcript, which is context, which is somewhere it will be read.
|
|
13
|
+
*/
|
|
14
|
+
import { type Config } from "../config.js";
|
|
15
|
+
export interface ToolResult {
|
|
16
|
+
/** The SDK's CallToolResult is open-ended; this keeps us assignable to it. */
|
|
17
|
+
[x: string]: unknown;
|
|
18
|
+
content: Array<{
|
|
19
|
+
type: "text";
|
|
20
|
+
text: string;
|
|
21
|
+
}>;
|
|
22
|
+
structuredContent?: Record<string, unknown>;
|
|
23
|
+
isError?: boolean;
|
|
24
|
+
}
|
|
25
|
+
export declare const text: (...blocks: Array<string | null | undefined>) => Array<{
|
|
26
|
+
type: "text";
|
|
27
|
+
text: string;
|
|
28
|
+
}>;
|
|
29
|
+
export declare function ok(blocks: Array<string | null | undefined>, structured: Record<string, unknown>): ToolResult;
|
|
30
|
+
export declare function failure(cfg: Config, message: string, hint?: string): ToolResult;
|
|
31
|
+
/** The message for a tool that needs a credential and has not been given one. */
|
|
32
|
+
export declare function requireKey(cfg: Config): ToolResult | null;
|
|
33
|
+
/** Turn any thrown thing into a tool result, with the key removed. */
|
|
34
|
+
export declare function fromThrown(cfg: Config, e: unknown): ToolResult;
|
|
35
|
+
/** `x-lobstack-dropped-params`, split. Empty when the header is absent. */
|
|
36
|
+
export declare function droppedParams(h: Headers): string[];
|
|
37
|
+
/** A note about the apex rewrite, wherever a base URL was used. */
|
|
38
|
+
export declare function baseNote(cfg: Config): string | null;
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turning a failure into something an agent can act on.
|
|
3
|
+
*
|
|
4
|
+
* Every tool in this repo answers a failure the same way: `isError: true` with
|
|
5
|
+
* a sentence and, where there is one, a next step. Not a thrown McpError —
|
|
6
|
+
* a protocol-level error tells the client the call was malformed, which is a
|
|
7
|
+
* different claim from "your key lacks a scope", and clients surface the two
|
|
8
|
+
* very differently.
|
|
9
|
+
*
|
|
10
|
+
* Everything that goes through here is scrubbed. The key is in this process's
|
|
11
|
+
* environment and nowhere else; it must not turn up in a tool result, which is
|
|
12
|
+
* transcript, which is context, which is somewhere it will be read.
|
|
13
|
+
*/
|
|
14
|
+
import { ConfigError, scrub } from "../config.js";
|
|
15
|
+
import { GatewayError } from "../gateway.js";
|
|
16
|
+
import { StreamError } from "../sse.js";
|
|
17
|
+
export const text = (...blocks) => blocks.filter((b) => typeof b === "string" && b.length > 0).map((t) => ({ type: "text", text: t }));
|
|
18
|
+
export function ok(blocks, structured) {
|
|
19
|
+
return { content: text(...blocks), structuredContent: structured };
|
|
20
|
+
}
|
|
21
|
+
export function failure(cfg, message, hint) {
|
|
22
|
+
const body = hint ? `${scrub(message, cfg.apiKey)}\n\n${scrub(hint, cfg.apiKey)}` : scrub(message, cfg.apiKey);
|
|
23
|
+
return { content: text(body), isError: true };
|
|
24
|
+
}
|
|
25
|
+
/** The message for a tool that needs a credential and has not been given one. */
|
|
26
|
+
export function requireKey(cfg) {
|
|
27
|
+
if (cfg.apiKey)
|
|
28
|
+
return null;
|
|
29
|
+
return failure(cfg, "No Lobstack API key is configured, so this call cannot be made.", "Set LOBSTACK_API_KEY in this server's environment (mint a key in Console → API keys) and restart the MCP client. " +
|
|
30
|
+
"lobstack_route_preview needs no key and works right now.");
|
|
31
|
+
}
|
|
32
|
+
/** Turn any thrown thing into a tool result, with the key removed. */
|
|
33
|
+
export function fromThrown(cfg, e) {
|
|
34
|
+
if (e instanceof GatewayError) {
|
|
35
|
+
const hints = [];
|
|
36
|
+
if (e.hint)
|
|
37
|
+
hints.push(e.hint);
|
|
38
|
+
if (e.status === 401) {
|
|
39
|
+
hints.push(cfg.keyShapeUnrecognised
|
|
40
|
+
? "LOBSTACK_API_KEY is set but is not shaped like a Lobstack API key (lsk_live_… or lsk_test_…). Check it was copied whole."
|
|
41
|
+
: "The key was rejected. It may have been revoked or have expired; mint a new one in Console → API keys.");
|
|
42
|
+
}
|
|
43
|
+
if (e.requestId)
|
|
44
|
+
hints.push(`Gateway request id: ${e.requestId}`);
|
|
45
|
+
return failure(cfg, e.message, hints.join("\n"));
|
|
46
|
+
}
|
|
47
|
+
if (e instanceof ConfigError)
|
|
48
|
+
return failure(cfg, e.message, e.hint);
|
|
49
|
+
if (e instanceof StreamError) {
|
|
50
|
+
return failure(cfg, `the gateway failed part-way through the answer: ${e.message}`);
|
|
51
|
+
}
|
|
52
|
+
return failure(cfg, e instanceof Error ? e.message : String(e));
|
|
53
|
+
}
|
|
54
|
+
/** `x-lobstack-dropped-params`, split. Empty when the header is absent. */
|
|
55
|
+
export function droppedParams(h) {
|
|
56
|
+
const raw = h.get("x-lobstack-dropped-params");
|
|
57
|
+
return raw ? raw.split(",").map((s) => s.trim()).filter(Boolean) : [];
|
|
58
|
+
}
|
|
59
|
+
/** A note about the apex rewrite, wherever a base URL was used. */
|
|
60
|
+
export function baseNote(cfg) {
|
|
61
|
+
return cfg.base.corrected
|
|
62
|
+
? ` note: the bare apex lobstack.ai redirects and would strip your key; ${cfg.base.origin} was used instead`
|
|
63
|
+
: null;
|
|
64
|
+
}
|