runbios-mcp 0.2.14-dev.245 → 0.2.14-dev.247
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +34 -12
- package/dist/api-client.d.ts +17 -5
- package/dist/api-client.d.ts.map +1 -1
- package/dist/api-client.js +266 -19
- package/dist/api-client.js.map +1 -1
- package/dist/http/mcp-handler.d.ts.map +1 -1
- package/dist/http/mcp-handler.js +3 -0
- package/dist/http/mcp-handler.js.map +1 -1
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +112 -33
- package/dist/server.js.map +1 -1
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -43,18 +43,26 @@ For interactive JWT authentication, omit `RUNBIOS_API_KEY` and set
|
|
|
43
43
|
`RUNBIOS_ACCESS_TOKEN`. The server never guesses credential type from a prefix:
|
|
44
44
|
API keys use `X-API-Key`, while access tokens use `Authorization: Bearer`.
|
|
45
45
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
`
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
46
|
+
The inference tools use `RUNBIOS_INFERENCE_KEY` for a dedicated deployment, or
|
|
47
|
+
a workspace `RUNBIOS_API_KEY` whose scopes allow serverless or dedicated
|
|
48
|
+
invocation through the unified `/v1` gateway;
|
|
49
|
+
no serving credential is accepted as a tool argument or returned in
|
|
50
|
+
model-visible text. `chat_with_inference` and `message_with_inference` work on
|
|
51
|
+
models that serve the respective OpenAI-chat or Anthropic-Messages dialect.
|
|
52
|
+
`complete_with_inference`, `embed_with_inference`, and
|
|
53
|
+
`rerank_with_inference` require a dedicated deployment that advertises the
|
|
54
|
+
matching task: the serverless pool has **no** completions/embeddings/rerank
|
|
55
|
+
routes. Choose by the server's declared model task, not a guess from the name.
|
|
56
|
+
|
|
57
|
+
Set `RUNBIOS_INFERENCE_BASE_URL=https://api-dev.runbios.ai` explicitly in dev.
|
|
58
|
+
The chat tool validates function schemas and matching tool responses. Chat
|
|
59
|
+
streaming assembles deltas; completions and Messages streaming preserve ordered
|
|
60
|
+
SSE events in one MCP result (up to 2 MB/4096 events; use an SDK for longer
|
|
61
|
+
streams). MCP cancellation propagates to HTTP, an error or response echoing the
|
|
62
|
+
configured key is suppressed, and a dispatched inference POST is never retried
|
|
63
|
+
automatically. `RUNBIOS_INFERENCE_TIMEOUT_MS` defaults to `900000` (15 minutes).
|
|
64
|
+
Caller-provided idempotency headers are forwarded without promising server-side
|
|
65
|
+
replay/deduplication.
|
|
58
66
|
|
|
59
67
|
## Workspace serverless observability
|
|
60
68
|
|
|
@@ -285,6 +293,16 @@ tokens are intentionally excluded from MCP tool inputs and no integration is
|
|
|
285
293
|
needed for models; Hugging Face integrations remain for dataset imports via
|
|
286
294
|
`import_huggingface_dataset`.
|
|
287
295
|
|
|
296
|
+
`list_inference_models` is a different read: it uses `/v1/models` and the
|
|
297
|
+
connection's selected inference credential to list the serverless pool and
|
|
298
|
+
workspace deployments it can actually invoke. `get_inference_model` retrieves
|
|
299
|
+
one exact id through the same scope and workspace boundary, including
|
|
300
|
+
`author/name` model ids. A partially readable list names
|
|
301
|
+
`usf_unreachable_sources`; never infer an unreadable source has no models.
|
|
302
|
+
The registry's `list_models` / `search_models` tools remain the source for
|
|
303
|
+
trainable/deployable base models, not the invocable roster. Neither new tool
|
|
304
|
+
accepts a credential argument or emits credential fields in its result.
|
|
305
|
+
|
|
288
306
|
`create_training_job` and `resume_training_job` accept an optional
|
|
289
307
|
`idempotency_key`. Reuse the same value after a timeout so the API replays the
|
|
290
308
|
original acknowledgement instead of creating another job, wallet hold, queue
|
|
@@ -393,6 +411,10 @@ deployed, not a roadmap item:
|
|
|
393
411
|
to the model. `upload_dataset` remains unavailable on hosted MCP because the
|
|
394
412
|
hosted process cannot read a file from the client's machine; use the console
|
|
395
413
|
for local uploads.
|
|
414
|
+
- `complete_with_inference`, `embed_with_inference`, `rerank_with_inference`
|
|
415
|
+
and `message_with_inference` are withheld from hosted `tools/list` until
|
|
416
|
+
OAuth-bound routes for those tasks are implemented and tested; stdio may
|
|
417
|
+
use the configured inference credential now.
|
|
396
418
|
|
|
397
419
|
Any documentation still describing the hosted connector as "planned" is stale.
|
|
398
420
|
|
package/dist/api-client.d.ts
CHANGED
|
@@ -60,8 +60,8 @@ export interface BiosClientOptions {
|
|
|
60
60
|
/** Authorization header value for inference calls (may be empty). */
|
|
61
61
|
inferenceAuthHeader?: string;
|
|
62
62
|
/**
|
|
63
|
-
* Fallback Authorization header for /v1/chat/completions
|
|
64
|
-
* deployment key
|
|
63
|
+
* Fallback Authorization header for serverless /v1/chat/completions and
|
|
64
|
+
* /v1/messages when no dedicated deployment key is set. A platform key with
|
|
65
65
|
* serverless scope authenticates model-ID-routed serverless catalog calls;
|
|
66
66
|
* the unified /v1 gateway routes by the request's `model`. Dedicated
|
|
67
67
|
* deployment keys still take precedence when present.
|
|
@@ -72,6 +72,7 @@ export interface BiosClientOptions {
|
|
|
72
72
|
/** User-Agent header; defaults to "bios-mcp/<VERSION>". */
|
|
73
73
|
userAgent?: string;
|
|
74
74
|
}
|
|
75
|
+
type InferenceVerbPath = "/v1/chat/completions" | "/v1/completions" | "/v1/embeddings" | "/v1/rerank" | "/v1/messages";
|
|
75
76
|
type ServerlessReadPath = "/api/serverless/limits" | "/api/serverless/usage/overview" | "/api/serverless/usage/by-model" | "/api/serverless/usage/requests" | "/api/serverless/usage/timeseries" | "/api/serverless/usage/daily" | "/api/serverless/usage/savings";
|
|
76
77
|
type ServerlessReadShape = "limits" | "overview" | "rows" | "timeseries" | "daily" | "savings";
|
|
77
78
|
/**
|
|
@@ -119,6 +120,9 @@ export declare class BiosClient {
|
|
|
119
120
|
* The caller never chooses an account and no credential value is exposed.
|
|
120
121
|
*/
|
|
121
122
|
resolvedWorkspaceId(): Promise<string>;
|
|
123
|
+
private inferenceModelRead;
|
|
124
|
+
inferenceModels(): Promise<unknown>;
|
|
125
|
+
inferenceModel(modelId: string): Promise<unknown>;
|
|
122
126
|
serverlessRead(path: ServerlessReadPath, shape: ServerlessReadShape, params?: Record<string, string | undefined>): Promise<unknown>;
|
|
123
127
|
/**
|
|
124
128
|
* Tri-state probe against the public model registry (model-service):
|
|
@@ -130,13 +134,16 @@ export declare class BiosClient {
|
|
|
130
134
|
modelRegistryStatus(modelId: string): Promise<"hosted" | "not_hosted" | "unknown">;
|
|
131
135
|
api<T = unknown>(path: string, opts?: ApiOpts): Promise<T>;
|
|
132
136
|
/**
|
|
133
|
-
* Select the Authorization header for /v1
|
|
134
|
-
* deployment key (inferenceAuthHeader) wins when present; otherwise a
|
|
137
|
+
* Select the Authorization header for the unified /v1 inference verbs. A
|
|
138
|
+
* dedicated deployment key (inferenceAuthHeader) wins when present; otherwise a
|
|
135
139
|
* serverless-scoped platform key (serverlessAuthHeader) or a hosted OAuth
|
|
136
140
|
* grant authenticates model-ID-routed calls on the unified endpoint.
|
|
137
141
|
*/
|
|
138
142
|
private resolveInferenceAuth;
|
|
139
|
-
|
|
143
|
+
private configuredInferenceKeys;
|
|
144
|
+
private redactInferenceError;
|
|
145
|
+
private containsInferenceCredential;
|
|
146
|
+
inferenceApi(body: Record<string, unknown>, idempotencyKey?: string, requestId?: string, callerSignal?: AbortSignal, path?: InferenceVerbPath, anthropicVersion?: string): Promise<unknown>;
|
|
140
147
|
/**
|
|
141
148
|
* POST a streaming /v1/chat/completions request, consume the SSE, and
|
|
142
149
|
* aggregate the deltas into ONE OpenAI-shaped completion. Content and
|
|
@@ -148,6 +155,7 @@ export declare class BiosClient {
|
|
|
148
155
|
* assembled message.
|
|
149
156
|
*/
|
|
150
157
|
streamInferenceApi(body: Record<string, unknown>, idempotencyKey?: string, requestId?: string, callerSignal?: AbortSignal): Promise<unknown>;
|
|
158
|
+
private streamTaskInferenceApi;
|
|
151
159
|
}
|
|
152
160
|
/**
|
|
153
161
|
* Parse the OpenAI chat-completion SSE from `stream` and fold its deltas into a
|
|
@@ -156,5 +164,9 @@ export declare class BiosClient {
|
|
|
156
164
|
* finish_reason, role, id, model, and the final usage row are captured.
|
|
157
165
|
*/
|
|
158
166
|
export declare function aggregateChatStream(stream: ReadableStream<Uint8Array>): Promise<Record<string, unknown>>;
|
|
167
|
+
export declare function collectInferenceEvents(stream: ReadableStream<Uint8Array>, containsCredential?: (content: string) => boolean, termination?: "openai" | "anthropic"): Promise<{
|
|
168
|
+
object: string;
|
|
169
|
+
events: Array<Record<string, unknown>>;
|
|
170
|
+
}>;
|
|
159
171
|
export {};
|
|
160
172
|
//# sourceMappingURL=api-client.d.ts.map
|
package/dist/api-client.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"api-client.d.ts","sourceRoot":"","sources":["../src/api-client.ts"],"names":[],"mappings":"AAMA;;;;GAIG;AACH,eAAO,MAAM,yBAAyB,yBAAyB,CAAC;AAEhE;;;GAGG;AACH,eAAO,MAAM,gCAAgC,6BAA6B,CAAC;AAE3E;;;;;;;GAOG;AACH,eAAO,MAAM,mBAAmB,yGAKtB,CAAC;AAeX,uDAAuD;AACvD,wBAAgB,mBAAmB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE3D;AAED,mEAAmE;AACnE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAEzD;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE7D;AAED,uEAAuE;AACvE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAIzD;AAED;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,IAAI,EAAE;IAC5C,SAAS,EAAE,OAAO,CAAC;IACnB,eAAe,EAAE,OAAO,CAAC;IACzB,mFAAmF;IACnF,IAAI,CAAC,EAAE,MAAM,CAAC;CACf,GAAG,MAAM,CAaT;AAID,MAAM,WAAW,OAAO;IACtB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,IAAI,CAAC,EAAE,OAAO,CAAC;IAIf,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,MAAM,EAAE,GAAG,SAAS,CAAC,CAAC;IACvD,QAAQ,CAAC,EAAE,QAAQ,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAClC;AAED,MAAM,WAAW,iBAAiB;IAChC,iDAAiD;IACjD,OAAO,EAAE,MAAM,CAAC;IAChB,uFAAuF;IACvF,WAAW,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,yDAAyD;IACzD,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,qEAAqE;IACrE,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;;;;OAMG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,qEAAqE;IACrE,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,2DAA2D;IAC3D,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,KAAK,kBAAkB,GAAG,wBAAwB,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,kCAAkC,GAAG,6BAA6B,GAAG,+BAA+B,CAAC;AACnQ,KAAK,mBAAmB,GAAG,QAAQ,GAAG,UAAU,GAAG,MAAM,GAAG,YAAY,GAAG,OAAO,GAAG,SAAS,CAAC;AAkM/F;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAajF;AAgHD,qBAAa,UAAU;IACrB,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,QAAQ,CAAC,WAAW,CAAyB;IACrD,OAAO,CAAC,QAAQ,CAAC,KAAK,CAAS;IAC/B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAO7B,OAAO,CAAC,uBAAuB,CAAC,CAAS;IACzC,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAS;IAC1C,OAAO,CAAC,QAAQ,CAAC,mBAAmB,CAAS;IAC7C,OAAO,CAAC,QAAQ,CAAC,oBAAoB,CAAS;IAC9C,OAAO,CAAC,QAAQ,CAAC,kBAAkB,CAAS;IAC5C,OAAO,CAAC,QAAQ,CAAC,SAAS,CAAS;gBAEvB,IAAI,EAAE,iBAAiB;IAYnC;;;;;;OAMG;IACG,mBAAmB,IAAI,OAAO,CAAC,MAAM,CAAC;
|
|
1
|
+
{"version":3,"file":"api-client.d.ts","sourceRoot":"","sources":["../src/api-client.ts"],"names":[],"mappings":"AAMA;;;;GAIG;AACH,eAAO,MAAM,yBAAyB,yBAAyB,CAAC;AAEhE;;;GAGG;AACH,eAAO,MAAM,gCAAgC,6BAA6B,CAAC;AAE3E;;;;;;;GAOG;AACH,eAAO,MAAM,mBAAmB,yGAKtB,CAAC;AAeX,uDAAuD;AACvD,wBAAgB,mBAAmB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE3D;AAED,mEAAmE;AACnE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAEzD;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE7D;AAED,uEAAuE;AACvE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAIzD;AAED;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,IAAI,EAAE;IAC5C,SAAS,EAAE,OAAO,CAAC;IACnB,eAAe,EAAE,OAAO,CAAC;IACzB,mFAAmF;IACnF,IAAI,CAAC,EAAE,MAAM,CAAC;CACf,GAAG,MAAM,CAaT;AAID,MAAM,WAAW,OAAO;IACtB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,IAAI,CAAC,EAAE,OAAO,CAAC;IAIf,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,MAAM,EAAE,GAAG,SAAS,CAAC,CAAC;IACvD,QAAQ,CAAC,EAAE,QAAQ,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAClC;AAED,MAAM,WAAW,iBAAiB;IAChC,iDAAiD;IACjD,OAAO,EAAE,MAAM,CAAC;IAChB,uFAAuF;IACvF,WAAW,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,yDAAyD;IACzD,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,qEAAqE;IACrE,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;;;;OAMG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,qEAAqE;IACrE,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,2DAA2D;IAC3D,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,KAAK,iBAAiB,GAAG,sBAAsB,GAAG,iBAAiB,GAAG,gBAAgB,GAAG,YAAY,GAAG,cAAc,CAAC;AACvH,KAAK,kBAAkB,GAAG,wBAAwB,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,kCAAkC,GAAG,6BAA6B,GAAG,+BAA+B,CAAC;AACnQ,KAAK,mBAAmB,GAAG,QAAQ,GAAG,UAAU,GAAG,MAAM,GAAG,YAAY,GAAG,OAAO,GAAG,SAAS,CAAC;AAkM/F;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAajF;AAgHD,qBAAa,UAAU;IACrB,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,QAAQ,CAAC,WAAW,CAAyB;IACrD,OAAO,CAAC,QAAQ,CAAC,KAAK,CAAS;IAC/B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAO7B,OAAO,CAAC,uBAAuB,CAAC,CAAS;IACzC,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAS;IAC1C,OAAO,CAAC,QAAQ,CAAC,mBAAmB,CAAS;IAC7C,OAAO,CAAC,QAAQ,CAAC,oBAAoB,CAAS;IAC9C,OAAO,CAAC,QAAQ,CAAC,kBAAkB,CAAS;IAC5C,OAAO,CAAC,QAAQ,CAAC,SAAS,CAAS;gBAEvB,IAAI,EAAE,iBAAiB;IAYnC;;;;;;OAMG;IACG,mBAAmB,IAAI,OAAO,CAAC,MAAM,CAAC;YAY9B,kBAAkB;IAsD1B,eAAe,IAAI,OAAO,CAAC,OAAO,CAAC;IAWnC,cAAc,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC;IAWjD,cAAc,CAClB,IAAI,EAAE,kBAAkB,EACxB,KAAK,EAAE,mBAAmB,EAC1B,MAAM,GAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,SAAS,CAAM,GAC9C,OAAO,CAAC,OAAO,CAAC;IAwEnB;;;;;;OAMG;IACG,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IA0BlF,GAAG,CAAC,CAAC,GAAG,OAAO,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,GAAE,OAAY,GAAG,OAAO,CAAC,CAAC,CAAC;IAiIpE;;;;;OAKG;IACH,OAAO,CAAC,oBAAoB;IAY5B,OAAO,CAAC,uBAAuB;IAM/B,OAAO,CAAC,oBAAoB;IAI5B,OAAO,CAAC,2BAA2B;IAI7B,YAAY,CAChB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,EAC1B,IAAI,GAAE,iBAA0C,EAChD,gBAAgB,CAAC,EAAE,MAAM,GACxB,OAAO,CAAC,OAAO,CAAC;IAsEnB;;;;;;;;;OASG;IACG,kBAAkB,CACtB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,GACzB,OAAO,CAAC,OAAO,CAAC;YAiDL,sBAAsB;CAoDrC;AAUD;;;;;GAKG;AACH,wBAAsB,mBAAmB,CAAC,MAAM,EAAE,cAAc,CAAC,UAAU,CAAC,GAAG,OAAO,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAmG9G;AAgBD,wBAAsB,sBAAsB,CAC1C,MAAM,EAAE,cAAc,CAAC,UAAU,CAAC,EAClC,kBAAkB,CAAC,EAAE,CAAC,OAAO,EAAE,MAAM,KAAK,OAAO,EACjD,WAAW,GAAE,QAAQ,GAAG,WAAsB,GAC7C,OAAO,CAAC;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,KAAK,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAA;CAAE,CAAC,CA6DrE"}
|
package/dist/api-client.js
CHANGED
|
@@ -450,6 +450,87 @@ export class BiosClient {
|
|
|
450
450
|
this.introspectedWorkspaceId = id.trim();
|
|
451
451
|
return this.introspectedWorkspaceId;
|
|
452
452
|
}
|
|
453
|
+
async inferenceModelRead(path) {
|
|
454
|
+
const platformKey = this.authHeaders["X-API-Key"];
|
|
455
|
+
const auth = this.inferenceAuthHeader || this.serverlessAuthHeader || this.authHeaders.Authorization
|
|
456
|
+
|| (platformKey ? `Bearer ${platformKey}` : "");
|
|
457
|
+
const headers = { "User-Agent": this.userAgent, Accept: "application/json" };
|
|
458
|
+
if (auth)
|
|
459
|
+
headers.Authorization = auth;
|
|
460
|
+
if (this.workspaceId)
|
|
461
|
+
headers["X-Workspace-ID"] = this.workspaceId;
|
|
462
|
+
if (this.orgId)
|
|
463
|
+
headers["X-Org-ID"] = this.orgId;
|
|
464
|
+
let response;
|
|
465
|
+
try {
|
|
466
|
+
response = await fetch(`${this.inferenceBaseUrl}${path}`, { method: "GET", headers });
|
|
467
|
+
}
|
|
468
|
+
catch {
|
|
469
|
+
throw new Error(JSON.stringify({
|
|
470
|
+
code: "MODELS_UNAVAILABLE", message: "Inference model discovery could not reach the API.",
|
|
471
|
+
instruction: "Retry shortly; do not treat this as an empty model roster.",
|
|
472
|
+
}));
|
|
473
|
+
}
|
|
474
|
+
const text = await response.text();
|
|
475
|
+
if (response.ok) {
|
|
476
|
+
for (const raw of [this.inferenceAuthHeader, this.serverlessAuthHeader, this.authHeaders.Authorization, platformKey]) {
|
|
477
|
+
const credential = raw?.replace(/^Bearer\s+/i, "").trim();
|
|
478
|
+
if (credential && credential.length >= 12 && text.includes(credential)) {
|
|
479
|
+
throw new Error(JSON.stringify({
|
|
480
|
+
code: "MODEL_RESPONSE_SUPPRESSED",
|
|
481
|
+
message: "The model catalog response contained authentication material and was withheld.",
|
|
482
|
+
instruction: "Contact support; do not quote or reuse this response.",
|
|
483
|
+
}));
|
|
484
|
+
}
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
if (!response.ok) {
|
|
488
|
+
let code = `HTTP_${response.status}`;
|
|
489
|
+
try {
|
|
490
|
+
const parsed = JSON.parse(text);
|
|
491
|
+
const candidate = parsed.error?.code ?? parsed.code;
|
|
492
|
+
if (typeof candidate === "string" && /^[A-Za-z_][A-Za-z0-9_-]{0,63}$/.test(candidate))
|
|
493
|
+
code = candidate;
|
|
494
|
+
}
|
|
495
|
+
catch { }
|
|
496
|
+
throw new Error(JSON.stringify({
|
|
497
|
+
code, status: response.status, message: "Inference model discovery failed.",
|
|
498
|
+
instruction: response.status === 404 && path.startsWith("/v1/models/")
|
|
499
|
+
? "Call list_inference_models for ids this credential can invoke in its bound workspace; do not guess another workspace's id."
|
|
500
|
+
: statusGuidance(response.status),
|
|
501
|
+
}));
|
|
502
|
+
}
|
|
503
|
+
try {
|
|
504
|
+
return JSON.parse(text);
|
|
505
|
+
}
|
|
506
|
+
catch {
|
|
507
|
+
throw new Error(JSON.stringify({
|
|
508
|
+
code: "INVALID_MODELS_RESPONSE", message: "Inference model discovery returned invalid JSON.",
|
|
509
|
+
instruction: "Retry shortly; do not treat this as an empty model roster.",
|
|
510
|
+
}));
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
async inferenceModels() {
|
|
514
|
+
const response = await this.inferenceModelRead("/v1/models");
|
|
515
|
+
const roster = response && typeof response === "object" ? response : null;
|
|
516
|
+
const rows = roster?.data;
|
|
517
|
+
if (roster?.object !== "list" || !Array.isArray(rows) ||
|
|
518
|
+
!rows.every(row => row && typeof row.id === "string" && row.id.trim())) {
|
|
519
|
+
throw new Error(JSON.stringify({ code: "INVALID_MODELS_RESPONSE", message: "The inference model roster had no usable data array.", instruction: "Retry shortly; an unavailable source is not an empty roster." }));
|
|
520
|
+
}
|
|
521
|
+
return response;
|
|
522
|
+
}
|
|
523
|
+
async inferenceModel(modelId) {
|
|
524
|
+
const id = typeof modelId === "string" ? modelId.trim() : "";
|
|
525
|
+
if (!id)
|
|
526
|
+
throw new Error("A model id is required; list_inference_models gives the ids this credential can invoke.");
|
|
527
|
+
const response = await this.inferenceModelRead(`/v1/models/${encodeURIComponent(id)}`);
|
|
528
|
+
if (!response || typeof response !== "object" || response.object !== "model" ||
|
|
529
|
+
typeof response.id !== "string") {
|
|
530
|
+
throw new Error(JSON.stringify({ code: "INVALID_MODEL_RESPONSE", message: "Inference model retrieval returned no model id.", instruction: "Retry or list the models this credential can invoke." }));
|
|
531
|
+
}
|
|
532
|
+
return response;
|
|
533
|
+
}
|
|
453
534
|
async serverlessRead(path, shape, params = {}) {
|
|
454
535
|
const credentials = [this.authHeaders["X-API-Key"], this.authHeaders.Authorization, this.inferenceAuthHeader, this.serverlessAuthHeader]
|
|
455
536
|
.map(raw => raw?.replace(/^Bearer\s+/i, "").trim() ?? "")
|
|
@@ -688,8 +769,8 @@ export class BiosClient {
|
|
|
688
769
|
return (parsed ?? {});
|
|
689
770
|
}
|
|
690
771
|
/**
|
|
691
|
-
* Select the Authorization header for /v1
|
|
692
|
-
* deployment key (inferenceAuthHeader) wins when present; otherwise a
|
|
772
|
+
* Select the Authorization header for the unified /v1 inference verbs. A
|
|
773
|
+
* dedicated deployment key (inferenceAuthHeader) wins when present; otherwise a
|
|
693
774
|
* serverless-scoped platform key (serverlessAuthHeader) or a hosted OAuth
|
|
694
775
|
* grant authenticates model-ID-routed calls on the unified endpoint.
|
|
695
776
|
*/
|
|
@@ -697,16 +778,28 @@ export class BiosClient {
|
|
|
697
778
|
const oauth = this.authHeaders.Authorization?.startsWith("Bearer mcp_at_") ? this.authHeaders.Authorization : "";
|
|
698
779
|
const auth = this.inferenceAuthHeader || this.serverlessAuthHeader || oauth;
|
|
699
780
|
if (!auth) {
|
|
700
|
-
throw new Error("
|
|
701
|
-
+ "or provide a workspace platform key (RUNBIOS_API_KEY with serverless scope) to call catalog models by id");
|
|
781
|
+
throw new Error("An inference credential is required: set RUNBIOS_INFERENCE_KEY for a dedicated deployment, "
|
|
782
|
+
+ "or provide a workspace platform key (RUNBIOS_API_KEY with serverless scope) to call compatible catalog models by id");
|
|
702
783
|
}
|
|
703
784
|
return auth;
|
|
704
785
|
}
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
786
|
+
configuredInferenceKeys() {
|
|
787
|
+
return [this.inferenceAuthHeader, this.serverlessAuthHeader, this.authHeaders.Authorization, this.authHeaders["X-API-Key"]]
|
|
788
|
+
.map(raw => raw?.replace(/^Bearer\s+/i, "").trim() ?? "")
|
|
789
|
+
.filter(key => key.length >= 12);
|
|
790
|
+
}
|
|
791
|
+
redactInferenceError(message) {
|
|
792
|
+
return this.configuredInferenceKeys().reduce((text, key) => text.replaceAll(key, "[redacted]"), redactBuildIdentity(message));
|
|
793
|
+
}
|
|
794
|
+
containsInferenceCredential(content) {
|
|
795
|
+
return this.configuredInferenceKeys().some(key => content.includes(key));
|
|
796
|
+
}
|
|
797
|
+
async inferenceApi(body, idempotencyKey, requestId, callerSignal, path = "/v1/chat/completions", anthropicVersion) {
|
|
798
|
+
// Aggregate chat deltas; preserve other task-specific SSE events intact.
|
|
708
799
|
if (body.stream === true) {
|
|
709
|
-
return
|
|
800
|
+
return path === "/v1/chat/completions"
|
|
801
|
+
? this.streamInferenceApi(body, idempotencyKey, requestId, callerSignal)
|
|
802
|
+
: this.streamTaskInferenceApi(path, body, idempotencyKey, requestId, callerSignal, anthropicVersion);
|
|
710
803
|
}
|
|
711
804
|
const headers = {
|
|
712
805
|
Authorization: this.resolveInferenceAuth(),
|
|
@@ -717,6 +810,8 @@ export class BiosClient {
|
|
|
717
810
|
};
|
|
718
811
|
if (idempotencyKey)
|
|
719
812
|
headers["Idempotency-Key"] = idempotencyKey;
|
|
813
|
+
if (path === "/v1/messages")
|
|
814
|
+
headers["Anthropic-Version"] = anthropicVersion || "2023-06-01";
|
|
720
815
|
// Inference POSTs are intentionally dispatched once. An MCP caller that
|
|
721
816
|
// retries can reuse idempotency_key; this client never replays implicitly.
|
|
722
817
|
const controller = new AbortController();
|
|
@@ -729,7 +824,7 @@ export class BiosClient {
|
|
|
729
824
|
let response;
|
|
730
825
|
let text;
|
|
731
826
|
try {
|
|
732
|
-
response = await fetch(`${this.inferenceBaseUrl}
|
|
827
|
+
response = await fetch(`${this.inferenceBaseUrl}${path}`, {
|
|
733
828
|
method: "POST",
|
|
734
829
|
headers,
|
|
735
830
|
body: JSON.stringify(body),
|
|
@@ -746,14 +841,17 @@ export class BiosClient {
|
|
|
746
841
|
// MCP client sees them as the tool error.
|
|
747
842
|
try {
|
|
748
843
|
const parsed = JSON.parse(text);
|
|
749
|
-
throw new Error(
|
|
844
|
+
throw new Error(this.redactInferenceError(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
|
|
750
845
|
}
|
|
751
846
|
catch (error) {
|
|
752
847
|
if (error instanceof SyntaxError)
|
|
753
|
-
throw new Error(
|
|
848
|
+
throw new Error(this.redactInferenceError(text || `Inference API error ${response.status}`));
|
|
754
849
|
throw error;
|
|
755
850
|
}
|
|
756
851
|
}
|
|
852
|
+
if (this.containsInferenceCredential(text)) {
|
|
853
|
+
throw new Error("Inference result contained authentication material and was withheld.");
|
|
854
|
+
}
|
|
757
855
|
let parsed;
|
|
758
856
|
try {
|
|
759
857
|
parsed = JSON.parse(text);
|
|
@@ -806,17 +904,76 @@ export class BiosClient {
|
|
|
806
904
|
const text = await response.text().catch(() => "");
|
|
807
905
|
try {
|
|
808
906
|
const parsed = JSON.parse(text);
|
|
809
|
-
throw new Error(
|
|
907
|
+
throw new Error(this.redactInferenceError(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
|
|
810
908
|
}
|
|
811
909
|
catch (error) {
|
|
812
910
|
if (error instanceof SyntaxError)
|
|
813
|
-
throw new Error(
|
|
911
|
+
throw new Error(this.redactInferenceError(text || `Inference API error ${response.status}`));
|
|
814
912
|
throw error;
|
|
815
913
|
}
|
|
816
914
|
}
|
|
817
915
|
if (!response.body)
|
|
818
916
|
throw new Error("Inference stream returned no body");
|
|
819
|
-
|
|
917
|
+
const result = await aggregateChatStream(response.body);
|
|
918
|
+
if (this.containsInferenceCredential(JSON.stringify(result))) {
|
|
919
|
+
throw new Error("Inference result contained authentication material and was withheld.");
|
|
920
|
+
}
|
|
921
|
+
return result;
|
|
922
|
+
}
|
|
923
|
+
catch (error) {
|
|
924
|
+
if (error instanceof Error)
|
|
925
|
+
throw new Error(this.redactInferenceError(error.message));
|
|
926
|
+
throw error;
|
|
927
|
+
}
|
|
928
|
+
finally {
|
|
929
|
+
clearTimeout(timeout);
|
|
930
|
+
callerSignal?.removeEventListener("abort", relayAbort);
|
|
931
|
+
}
|
|
932
|
+
}
|
|
933
|
+
async streamTaskInferenceApi(path, body, idempotencyKey, requestId, callerSignal, anthropicVersion) {
|
|
934
|
+
const headers = {
|
|
935
|
+
Authorization: this.resolveInferenceAuth(),
|
|
936
|
+
Accept: "text/event-stream",
|
|
937
|
+
"Content-Type": "application/json",
|
|
938
|
+
"X-Request-ID": requestId || randomUUID(),
|
|
939
|
+
"User-Agent": this.userAgent,
|
|
940
|
+
};
|
|
941
|
+
if (idempotencyKey)
|
|
942
|
+
headers["Idempotency-Key"] = idempotencyKey;
|
|
943
|
+
if (path === "/v1/messages")
|
|
944
|
+
headers["Anthropic-Version"] = anthropicVersion || "2023-06-01";
|
|
945
|
+
const controller = new AbortController();
|
|
946
|
+
const relayAbort = () => controller.abort(callerSignal?.reason);
|
|
947
|
+
if (callerSignal?.aborted)
|
|
948
|
+
relayAbort();
|
|
949
|
+
else
|
|
950
|
+
callerSignal?.addEventListener("abort", relayAbort, { once: true });
|
|
951
|
+
const timeout = setTimeout(() => controller.abort(new Error(`Inference timed out after ${this.inferenceTimeoutMs}ms`)), this.inferenceTimeoutMs);
|
|
952
|
+
try {
|
|
953
|
+
const response = await fetch(`${this.inferenceBaseUrl}${path}`, {
|
|
954
|
+
method: "POST", headers, body: JSON.stringify(body), signal: controller.signal,
|
|
955
|
+
});
|
|
956
|
+
if (!response.ok) {
|
|
957
|
+
const text = await response.text().catch(() => "");
|
|
958
|
+
let message = text || `Inference API error ${response.status}`;
|
|
959
|
+
try {
|
|
960
|
+
const parsed = JSON.parse(text);
|
|
961
|
+
message = parsed?.error?.message || parsed?.message || message;
|
|
962
|
+
}
|
|
963
|
+
catch { }
|
|
964
|
+
throw new Error(this.redactInferenceError(message));
|
|
965
|
+
}
|
|
966
|
+
if (!response.headers.get("content-type")?.toLowerCase().includes("text/event-stream")) {
|
|
967
|
+
throw new Error("Inference endpoint did not return an event stream");
|
|
968
|
+
}
|
|
969
|
+
if (!response.body)
|
|
970
|
+
throw new Error("Inference stream returned no body");
|
|
971
|
+
return await collectInferenceEvents(response.body, data => this.containsInferenceCredential(data), path === "/v1/messages" ? "anthropic" : "openai");
|
|
972
|
+
}
|
|
973
|
+
catch (error) {
|
|
974
|
+
if (error instanceof Error)
|
|
975
|
+
throw new Error(this.redactInferenceError(error.message));
|
|
976
|
+
throw error;
|
|
820
977
|
}
|
|
821
978
|
finally {
|
|
822
979
|
clearTimeout(timeout);
|
|
@@ -842,10 +999,15 @@ export async function aggregateChatStream(stream) {
|
|
|
842
999
|
let model;
|
|
843
1000
|
let created;
|
|
844
1001
|
let usage;
|
|
1002
|
+
let completed = false;
|
|
845
1003
|
const toolCalls = [];
|
|
846
1004
|
const consume = (payload) => {
|
|
847
1005
|
const data = payload.trim();
|
|
848
|
-
if (
|
|
1006
|
+
if (data === "[DONE]") {
|
|
1007
|
+
completed = true;
|
|
1008
|
+
return;
|
|
1009
|
+
}
|
|
1010
|
+
if (!data)
|
|
849
1011
|
return;
|
|
850
1012
|
let event;
|
|
851
1013
|
try {
|
|
@@ -897,7 +1059,7 @@ export async function aggregateChatStream(stream) {
|
|
|
897
1059
|
}
|
|
898
1060
|
};
|
|
899
1061
|
try {
|
|
900
|
-
|
|
1062
|
+
while (!completed) {
|
|
901
1063
|
const { done, value } = await reader.read();
|
|
902
1064
|
if (done)
|
|
903
1065
|
break;
|
|
@@ -909,16 +1071,22 @@ export async function aggregateChatStream(stream) {
|
|
|
909
1071
|
const frame = buffer.slice(0, match.index);
|
|
910
1072
|
buffer = buffer.slice(match.index + match[0].length);
|
|
911
1073
|
emitFrame(frame, consume);
|
|
1074
|
+
if (completed)
|
|
1075
|
+
break;
|
|
912
1076
|
}
|
|
913
1077
|
}
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
1078
|
+
if (!completed) {
|
|
1079
|
+
buffer += decoder.decode();
|
|
1080
|
+
if (buffer.trim())
|
|
1081
|
+
emitFrame(buffer, consume);
|
|
1082
|
+
}
|
|
917
1083
|
}
|
|
918
1084
|
finally {
|
|
919
1085
|
await reader.cancel().catch(() => undefined);
|
|
920
1086
|
reader.releaseLock();
|
|
921
1087
|
}
|
|
1088
|
+
if (!completed)
|
|
1089
|
+
throw new Error("Inference stream ended before completion marker [DONE]; no finished result was returned.");
|
|
922
1090
|
const message = { role, content: content || null };
|
|
923
1091
|
if (reasoning)
|
|
924
1092
|
message.reasoning_content = reasoning;
|
|
@@ -959,4 +1127,83 @@ function emitFrame(frame, consume) {
|
|
|
959
1127
|
if (data.length > 0)
|
|
960
1128
|
consume(data.join("\n"));
|
|
961
1129
|
}
|
|
1130
|
+
export async function collectInferenceEvents(stream, containsCredential, termination = "openai") {
|
|
1131
|
+
const reader = stream.getReader();
|
|
1132
|
+
const decoder = new TextDecoder();
|
|
1133
|
+
const events = [];
|
|
1134
|
+
let buffer = "";
|
|
1135
|
+
let size = 0;
|
|
1136
|
+
let completed = false;
|
|
1137
|
+
const consume = (payload) => {
|
|
1138
|
+
const data = payload.trim();
|
|
1139
|
+
if (data === "[DONE]") {
|
|
1140
|
+
if (termination === "anthropic")
|
|
1141
|
+
throw new Error("Anthropic stream ended without message_stop");
|
|
1142
|
+
completed = true;
|
|
1143
|
+
return;
|
|
1144
|
+
}
|
|
1145
|
+
if (!data)
|
|
1146
|
+
return;
|
|
1147
|
+
if (containsCredential?.(data))
|
|
1148
|
+
throw new Error("Inference result contained authentication material and was withheld.");
|
|
1149
|
+
if (size + data.length > 2_000_000 || events.length >= 4096) {
|
|
1150
|
+
throw new Error("Inference stream exceeded the MCP result limit; reduce max_tokens and retry explicitly.");
|
|
1151
|
+
}
|
|
1152
|
+
let event;
|
|
1153
|
+
try {
|
|
1154
|
+
event = JSON.parse(data);
|
|
1155
|
+
}
|
|
1156
|
+
catch {
|
|
1157
|
+
throw new Error("Inference stream returned invalid JSON");
|
|
1158
|
+
}
|
|
1159
|
+
if (!event || typeof event !== "object" || Array.isArray(event)) {
|
|
1160
|
+
throw new Error("Inference stream returned an invalid event");
|
|
1161
|
+
}
|
|
1162
|
+
const record = event;
|
|
1163
|
+
if (record.error) {
|
|
1164
|
+
const err = record.error;
|
|
1165
|
+
const candidate = err && typeof err === "object" ? err.code : undefined;
|
|
1166
|
+
const code = typeof candidate === "string" && /^[A-Za-z_][A-Za-z0-9_-]{0,63}$/.test(candidate) ? candidate : "stream_error";
|
|
1167
|
+
throw new Error(`Inference stream failed (${code}); no completed result was returned.`);
|
|
1168
|
+
}
|
|
1169
|
+
events.push(record);
|
|
1170
|
+
size += data.length;
|
|
1171
|
+
if (termination === "anthropic" && record.type === "message_stop")
|
|
1172
|
+
completed = true;
|
|
1173
|
+
};
|
|
1174
|
+
try {
|
|
1175
|
+
while (!completed) {
|
|
1176
|
+
const { done, value } = await reader.read();
|
|
1177
|
+
if (done)
|
|
1178
|
+
break;
|
|
1179
|
+
buffer += decoder.decode(value, { stream: true });
|
|
1180
|
+
if (buffer.length > 2_000_000)
|
|
1181
|
+
throw new Error("Inference stream exceeded the MCP result limit; reduce max_tokens.");
|
|
1182
|
+
for (;;) {
|
|
1183
|
+
const match = /\r\n\r\n|\n\n|\r\r/.exec(buffer);
|
|
1184
|
+
if (!match || match.index === undefined)
|
|
1185
|
+
break;
|
|
1186
|
+
const frame = buffer.slice(0, match.index);
|
|
1187
|
+
buffer = buffer.slice(match.index + match[0].length);
|
|
1188
|
+
emitFrame(frame, consume);
|
|
1189
|
+
if (completed)
|
|
1190
|
+
break;
|
|
1191
|
+
}
|
|
1192
|
+
}
|
|
1193
|
+
if (!completed) {
|
|
1194
|
+
buffer += decoder.decode();
|
|
1195
|
+
if (buffer.trim())
|
|
1196
|
+
emitFrame(buffer, consume);
|
|
1197
|
+
}
|
|
1198
|
+
}
|
|
1199
|
+
finally {
|
|
1200
|
+
await reader.cancel().catch(() => undefined);
|
|
1201
|
+
reader.releaseLock();
|
|
1202
|
+
}
|
|
1203
|
+
if (events.length === 0)
|
|
1204
|
+
throw new Error("Inference stream finished without a result");
|
|
1205
|
+
if (!completed)
|
|
1206
|
+
throw new Error("Inference stream ended before completion; no finished result was returned.");
|
|
1207
|
+
return { object: "inference.event_stream", events };
|
|
1208
|
+
}
|
|
962
1209
|
//# sourceMappingURL=api-client.js.map
|