runbios-mcp 0.2.14-dev.245 → 0.2.14-dev.247

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -43,18 +43,26 @@ For interactive JWT authentication, omit `RUNBIOS_API_KEY` and set
43
43
  `RUNBIOS_ACCESS_TOKEN`. The server never guesses credential type from a prefix:
44
44
  API keys use `X-API-Key`, while access tokens use `Authorization: Bearer`.
45
45
 
46
- Local `chat_with_inference` accepts `RUNBIOS_INFERENCE_KEY` for a dedicated
47
- endpoint, or a workspace platform key (`RUNBIOS_API_KEY`) whose scopes allow
48
- serverless or dedicated invocation through the unified `/v1` gateway. Credentials
49
- are never tool arguments or model-visible results. Set
50
- `RUNBIOS_INFERENCE_BASE_URL=https://api-dev.runbios.ai` explicitly for dev.
51
- The tool validates function schemas and matching tool response messages,
52
- propagates MCP cancellation, and makes one OpenAI-compatible request; with
53
- `stream=true` it consumes the SSE stream and returns the aggregated completion.
54
- It never retries a dispatched inference POST automatically.
55
- `RUNBIOS_INFERENCE_TIMEOUT_MS` defaults to `900000` (15 minutes). Its
56
- idempotency header is propagated, but replay/dedup is not claimed without an
57
- explicit acknowledgement from the endpoint.
46
+ The inference tools use `RUNBIOS_INFERENCE_KEY` for a dedicated deployment, or
47
+ a workspace `RUNBIOS_API_KEY` whose scopes allow serverless or dedicated
48
+ invocation through the unified `/v1` gateway;
49
+ no serving credential is accepted as a tool argument or returned in
50
+ model-visible text. `chat_with_inference` and `message_with_inference` work on
51
+ models that serve the respective OpenAI-chat or Anthropic-Messages dialect.
52
+ `complete_with_inference`, `embed_with_inference`, and
53
+ `rerank_with_inference` require a dedicated deployment that advertises the
54
+ matching task: the serverless pool has **no** completions/embeddings/rerank
55
+ routes. Choose by the server's declared model task, not a guess from the name.
56
+
57
+ Set `RUNBIOS_INFERENCE_BASE_URL=https://api-dev.runbios.ai` explicitly in dev.
58
+ The chat tool validates function schemas and matching tool responses. Chat
59
+ streaming assembles deltas; completions and Messages streaming preserve ordered
60
+ SSE events in one MCP result (up to 2 MB/4096 events; use an SDK for longer
61
+ streams). MCP cancellation propagates to HTTP, an error or response echoing the
62
+ configured key is suppressed, and a dispatched inference POST is never retried
63
+ automatically. `RUNBIOS_INFERENCE_TIMEOUT_MS` defaults to `900000` (15 minutes).
64
+ Caller-provided idempotency headers are forwarded without promising server-side
65
+ replay/deduplication.
58
66
 
59
67
  ## Workspace serverless observability
60
68
 
@@ -285,6 +293,16 @@ tokens are intentionally excluded from MCP tool inputs and no integration is
285
293
  needed for models; Hugging Face integrations remain for dataset imports via
286
294
  `import_huggingface_dataset`.
287
295
 
296
+ `list_inference_models` is a different read: it uses `/v1/models` and the
297
+ connection's selected inference credential to list the serverless pool and
298
+ workspace deployments it can actually invoke. `get_inference_model` retrieves
299
+ one exact id through the same scope and workspace boundary, including
300
+ `author/name` model ids. A partially readable list names
301
+ `usf_unreachable_sources`; never infer an unreadable source has no models.
302
+ The registry's `list_models` / `search_models` tools remain the source for
303
+ trainable/deployable base models, not the invocable roster. Neither new tool
304
+ accepts a credential argument or emits credential fields in its result.
305
+
288
306
  `create_training_job` and `resume_training_job` accept an optional
289
307
  `idempotency_key`. Reuse the same value after a timeout so the API replays the
290
308
  original acknowledgement instead of creating another job, wallet hold, queue
@@ -393,6 +411,10 @@ deployed, not a roadmap item:
393
411
  to the model. `upload_dataset` remains unavailable on hosted MCP because the
394
412
  hosted process cannot read a file from the client's machine; use the console
395
413
  for local uploads.
414
+ - `complete_with_inference`, `embed_with_inference`, `rerank_with_inference`
415
+ and `message_with_inference` are withheld from hosted `tools/list` until
416
+ OAuth-bound routes for those tasks are implemented and tested; stdio may
417
+ use the configured inference credential now.
396
418
 
397
419
  Any documentation still describing the hosted connector as "planned" is stale.
398
420
 
@@ -60,8 +60,8 @@ export interface BiosClientOptions {
60
60
  /** Authorization header value for inference calls (may be empty). */
61
61
  inferenceAuthHeader?: string;
62
62
  /**
63
- * Fallback Authorization header for /v1/chat/completions when no dedicated
64
- * deployment key (inferenceAuthHeader) is set. A workspace platform key with
63
+ * Fallback Authorization header for serverless /v1/chat/completions and
64
+ * /v1/messages when no dedicated deployment key is set. A platform key with
65
65
  * serverless scope authenticates model-ID-routed serverless catalog calls;
66
66
  * the unified /v1 gateway routes by the request's `model`. Dedicated
67
67
  * deployment keys still take precedence when present.
@@ -72,6 +72,7 @@ export interface BiosClientOptions {
72
72
  /** User-Agent header; defaults to "bios-mcp/<VERSION>". */
73
73
  userAgent?: string;
74
74
  }
75
+ type InferenceVerbPath = "/v1/chat/completions" | "/v1/completions" | "/v1/embeddings" | "/v1/rerank" | "/v1/messages";
75
76
  type ServerlessReadPath = "/api/serverless/limits" | "/api/serverless/usage/overview" | "/api/serverless/usage/by-model" | "/api/serverless/usage/requests" | "/api/serverless/usage/timeseries" | "/api/serverless/usage/daily" | "/api/serverless/usage/savings";
76
77
  type ServerlessReadShape = "limits" | "overview" | "rows" | "timeseries" | "daily" | "savings";
77
78
  /**
@@ -119,6 +120,9 @@ export declare class BiosClient {
119
120
  * The caller never chooses an account and no credential value is exposed.
120
121
  */
121
122
  resolvedWorkspaceId(): Promise<string>;
123
+ private inferenceModelRead;
124
+ inferenceModels(): Promise<unknown>;
125
+ inferenceModel(modelId: string): Promise<unknown>;
122
126
  serverlessRead(path: ServerlessReadPath, shape: ServerlessReadShape, params?: Record<string, string | undefined>): Promise<unknown>;
123
127
  /**
124
128
  * Tri-state probe against the public model registry (model-service):
@@ -130,13 +134,16 @@ export declare class BiosClient {
130
134
  modelRegistryStatus(modelId: string): Promise<"hosted" | "not_hosted" | "unknown">;
131
135
  api<T = unknown>(path: string, opts?: ApiOpts): Promise<T>;
132
136
  /**
133
- * Select the Authorization header for /v1/chat/completions. A dedicated
134
- * deployment key (inferenceAuthHeader) wins when present; otherwise a
137
+ * Select the Authorization header for the unified /v1 inference verbs. A
138
+ * dedicated deployment key (inferenceAuthHeader) wins when present; otherwise a
135
139
  * serverless-scoped platform key (serverlessAuthHeader) or a hosted OAuth
136
140
  * grant authenticates model-ID-routed calls on the unified endpoint.
137
141
  */
138
142
  private resolveInferenceAuth;
139
- inferenceApi(body: Record<string, unknown>, idempotencyKey?: string, requestId?: string, callerSignal?: AbortSignal): Promise<unknown>;
143
+ private configuredInferenceKeys;
144
+ private redactInferenceError;
145
+ private containsInferenceCredential;
146
+ inferenceApi(body: Record<string, unknown>, idempotencyKey?: string, requestId?: string, callerSignal?: AbortSignal, path?: InferenceVerbPath, anthropicVersion?: string): Promise<unknown>;
140
147
  /**
141
148
  * POST a streaming /v1/chat/completions request, consume the SSE, and
142
149
  * aggregate the deltas into ONE OpenAI-shaped completion. Content and
@@ -148,6 +155,7 @@ export declare class BiosClient {
148
155
  * assembled message.
149
156
  */
150
157
  streamInferenceApi(body: Record<string, unknown>, idempotencyKey?: string, requestId?: string, callerSignal?: AbortSignal): Promise<unknown>;
158
+ private streamTaskInferenceApi;
151
159
  }
152
160
  /**
153
161
  * Parse the OpenAI chat-completion SSE from `stream` and fold its deltas into a
@@ -156,5 +164,9 @@ export declare class BiosClient {
156
164
  * finish_reason, role, id, model, and the final usage row are captured.
157
165
  */
158
166
  export declare function aggregateChatStream(stream: ReadableStream<Uint8Array>): Promise<Record<string, unknown>>;
167
+ export declare function collectInferenceEvents(stream: ReadableStream<Uint8Array>, containsCredential?: (content: string) => boolean, termination?: "openai" | "anthropic"): Promise<{
168
+ object: string;
169
+ events: Array<Record<string, unknown>>;
170
+ }>;
159
171
  export {};
160
172
  //# sourceMappingURL=api-client.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"api-client.d.ts","sourceRoot":"","sources":["../src/api-client.ts"],"names":[],"mappings":"AAMA;;;;GAIG;AACH,eAAO,MAAM,yBAAyB,yBAAyB,CAAC;AAEhE;;;GAGG;AACH,eAAO,MAAM,gCAAgC,6BAA6B,CAAC;AAE3E;;;;;;;GAOG;AACH,eAAO,MAAM,mBAAmB,yGAKtB,CAAC;AAeX,uDAAuD;AACvD,wBAAgB,mBAAmB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE3D;AAED,mEAAmE;AACnE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAEzD;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE7D;AAED,uEAAuE;AACvE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAIzD;AAED;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,IAAI,EAAE;IAC5C,SAAS,EAAE,OAAO,CAAC;IACnB,eAAe,EAAE,OAAO,CAAC;IACzB,mFAAmF;IACnF,IAAI,CAAC,EAAE,MAAM,CAAC;CACf,GAAG,MAAM,CAaT;AAID,MAAM,WAAW,OAAO;IACtB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,IAAI,CAAC,EAAE,OAAO,CAAC;IAIf,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,MAAM,EAAE,GAAG,SAAS,CAAC,CAAC;IACvD,QAAQ,CAAC,EAAE,QAAQ,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAClC;AAED,MAAM,WAAW,iBAAiB;IAChC,iDAAiD;IACjD,OAAO,EAAE,MAAM,CAAC;IAChB,uFAAuF;IACvF,WAAW,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,yDAAyD;IACzD,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,qEAAqE;IACrE,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;;;;OAMG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,qEAAqE;IACrE,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,2DAA2D;IAC3D,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,KAAK,kBAAkB,GAAG,wBAAwB,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,kCAAkC,GAAG,6BAA6B,GAAG,+BAA+B,CAAC;AACnQ,KAAK,mBAAmB,GAAG,QAAQ,GAAG,UAAU,GAAG,MAAM,GAAG,YAAY,GAAG,OAAO,GAAG,SAAS,CAAC;AAkM/F;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAajF;AAgHD,qBAAa,UAAU;IACrB,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,QAAQ,CAAC,WAAW,CAAyB;IACrD,OAAO,CAAC,QAAQ,CAAC,KAAK,CAAS;IAC/B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAO7B,OAAO,CAAC,uBAAuB,CAAC,CAAS;IACzC,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAS;IAC1C,OAAO,CAAC,QAAQ,CAAC,mBAAmB,CAAS;IAC7C,OAAO,CAAC,QAAQ,CAAC,oBAAoB,CAAS;IAC9C,OAAO,CAAC,QAAQ,CAAC,kBAAkB,CAAS;IAC5C,OAAO,CAAC,QAAQ,CAAC,SAAS,CAAS;gBAEvB,IAAI,EAAE,iBAAiB;IAYnC;;;;;;OAMG;IACG,mBAAmB,IAAI,OAAO,CAAC,MAAM,CAAC;IAYtC,cAAc,CAClB,IAAI,EAAE,kBAAkB,EACxB,KAAK,EAAE,mBAAmB,EAC1B,MAAM,GAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,SAAS,CAAM,GAC9C,OAAO,CAAC,OAAO,CAAC;IAwEnB;;;;;;OAMG;IACG,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IA0BlF,GAAG,CAAC,CAAC,GAAG,OAAO,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,GAAE,OAAY,GAAG,OAAO,CAAC,CAAC,CAAC;IAiIpE;;;;;OAKG;IACH,OAAO,CAAC,oBAAoB;IAYtB,YAAY,CAChB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,GACzB,OAAO,CAAC,OAAO,CAAC;IAiEnB;;;;;;;;;OASG;IACG,kBAAkB,CACtB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,GACzB,OAAO,CAAC,OAAO,CAAC;CAyCpB;AAUD;;;;;GAKG;AACH,wBAAsB,mBAAmB,CAAC,MAAM,EAAE,cAAc,CAAC,UAAU,CAAC,GAAG,OAAO,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CA6F9G"}
1
+ {"version":3,"file":"api-client.d.ts","sourceRoot":"","sources":["../src/api-client.ts"],"names":[],"mappings":"AAMA;;;;GAIG;AACH,eAAO,MAAM,yBAAyB,yBAAyB,CAAC;AAEhE;;;GAGG;AACH,eAAO,MAAM,gCAAgC,6BAA6B,CAAC;AAE3E;;;;;;;GAOG;AACH,eAAO,MAAM,mBAAmB,yGAKtB,CAAC;AAeX,uDAAuD;AACvD,wBAAgB,mBAAmB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE3D;AAED,mEAAmE;AACnE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAEzD;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAE7D;AAED,uEAAuE;AACvE,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,OAAO,GAAG,OAAO,CAIzD;AAED;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,IAAI,EAAE;IAC5C,SAAS,EAAE,OAAO,CAAC;IACnB,eAAe,EAAE,OAAO,CAAC;IACzB,mFAAmF;IACnF,IAAI,CAAC,EAAE,MAAM,CAAC;CACf,GAAG,MAAM,CAaT;AAID,MAAM,WAAW,OAAO;IACtB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,IAAI,CAAC,EAAE,OAAO,CAAC;IAIf,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,MAAM,EAAE,GAAG,SAAS,CAAC,CAAC;IACvD,QAAQ,CAAC,EAAE,QAAQ,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAClC;AAED,MAAM,WAAW,iBAAiB;IAChC,iDAAiD;IACjD,OAAO,EAAE,MAAM,CAAC;IAChB,uFAAuF;IACvF,WAAW,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACpC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,yDAAyD;IACzD,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,qEAAqE;IACrE,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B;;;;;;OAMG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,qEAAqE;IACrE,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,2DAA2D;IAC3D,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,KAAK,iBAAiB,GAAG,sBAAsB,GAAG,iBAAiB,GAAG,gBAAgB,GAAG,YAAY,GAAG,cAAc,CAAC;AACvH,KAAK,kBAAkB,GAAG,wBAAwB,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,gCAAgC,GAAG,kCAAkC,GAAG,6BAA6B,GAAG,+BAA+B,CAAC;AACnQ,KAAK,mBAAmB,GAAG,QAAQ,GAAG,UAAU,GAAG,MAAM,GAAG,YAAY,GAAG,OAAO,GAAG,SAAS,CAAC;AAkM/F;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,cAAc,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAajF;AAgHD,qBAAa,UAAU;IACrB,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,QAAQ,CAAC,WAAW,CAAyB;IACrD,OAAO,CAAC,QAAQ,CAAC,KAAK,CAAS;IAC/B,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAO7B,OAAO,CAAC,uBAAuB,CAAC,CAAS;IACzC,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAS;IAC1C,OAAO,CAAC,QAAQ,CAAC,mBAAmB,CAAS;IAC7C,OAAO,CAAC,QAAQ,CAAC,oBAAoB,CAAS;IAC9C,OAAO,CAAC,QAAQ,CAAC,kBAAkB,CAAS;IAC5C,OAAO,CAAC,QAAQ,CAAC,SAAS,CAAS;gBAEvB,IAAI,EAAE,iBAAiB;IAYnC;;;;;;OAMG;IACG,mBAAmB,IAAI,OAAO,CAAC,MAAM,CAAC;YAY9B,kBAAkB;IAsD1B,eAAe,IAAI,OAAO,CAAC,OAAO,CAAC;IAWnC,cAAc,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC;IAWjD,cAAc,CAClB,IAAI,EAAE,kBAAkB,EACxB,KAAK,EAAE,mBAAmB,EAC1B,MAAM,GAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,SAAS,CAAM,GAC9C,OAAO,CAAC,OAAO,CAAC;IAwEnB;;;;;;OAMG;IACG,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,QAAQ,GAAG,YAAY,GAAG,SAAS,CAAC;IA0BlF,GAAG,CAAC,CAAC,GAAG,OAAO,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,GAAE,OAAY,GAAG,OAAO,CAAC,CAAC,CAAC;IAiIpE;;;;;OAKG;IACH,OAAO,CAAC,oBAAoB;IAY5B,OAAO,CAAC,uBAAuB;IAM/B,OAAO,CAAC,oBAAoB;IAI5B,OAAO,CAAC,2BAA2B;IAI7B,YAAY,CAChB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,EAC1B,IAAI,GAAE,iBAA0C,EAChD,gBAAgB,CAAC,EAAE,MAAM,GACxB,OAAO,CAAC,OAAO,CAAC;IAsEnB;;;;;;;;;OASG;IACG,kBAAkB,CACtB,IAAI,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7B,cAAc,CAAC,EAAE,MAAM,EACvB,SAAS,CAAC,EAAE,MAAM,EAClB,YAAY,CAAC,EAAE,WAAW,GACzB,OAAO,CAAC,OAAO,CAAC;YAiDL,sBAAsB;CAoDrC;AAUD;;;;;GAKG;AACH,wBAAsB,mBAAmB,CAAC,MAAM,EAAE,cAAc,CAAC,UAAU,CAAC,GAAG,OAAO,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAmG9G;AAgBD,wBAAsB,sBAAsB,CAC1C,MAAM,EAAE,cAAc,CAAC,UAAU,CAAC,EAClC,kBAAkB,CAAC,EAAE,CAAC,OAAO,EAAE,MAAM,KAAK,OAAO,EACjD,WAAW,GAAE,QAAQ,GAAG,WAAsB,GAC7C,OAAO,CAAC;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,KAAK,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAA;CAAE,CAAC,CA6DrE"}
@@ -450,6 +450,87 @@ export class BiosClient {
450
450
  this.introspectedWorkspaceId = id.trim();
451
451
  return this.introspectedWorkspaceId;
452
452
  }
453
+ async inferenceModelRead(path) {
454
+ const platformKey = this.authHeaders["X-API-Key"];
455
+ const auth = this.inferenceAuthHeader || this.serverlessAuthHeader || this.authHeaders.Authorization
456
+ || (platformKey ? `Bearer ${platformKey}` : "");
457
+ const headers = { "User-Agent": this.userAgent, Accept: "application/json" };
458
+ if (auth)
459
+ headers.Authorization = auth;
460
+ if (this.workspaceId)
461
+ headers["X-Workspace-ID"] = this.workspaceId;
462
+ if (this.orgId)
463
+ headers["X-Org-ID"] = this.orgId;
464
+ let response;
465
+ try {
466
+ response = await fetch(`${this.inferenceBaseUrl}${path}`, { method: "GET", headers });
467
+ }
468
+ catch {
469
+ throw new Error(JSON.stringify({
470
+ code: "MODELS_UNAVAILABLE", message: "Inference model discovery could not reach the API.",
471
+ instruction: "Retry shortly; do not treat this as an empty model roster.",
472
+ }));
473
+ }
474
+ const text = await response.text();
475
+ if (response.ok) {
476
+ for (const raw of [this.inferenceAuthHeader, this.serverlessAuthHeader, this.authHeaders.Authorization, platformKey]) {
477
+ const credential = raw?.replace(/^Bearer\s+/i, "").trim();
478
+ if (credential && credential.length >= 12 && text.includes(credential)) {
479
+ throw new Error(JSON.stringify({
480
+ code: "MODEL_RESPONSE_SUPPRESSED",
481
+ message: "The model catalog response contained authentication material and was withheld.",
482
+ instruction: "Contact support; do not quote or reuse this response.",
483
+ }));
484
+ }
485
+ }
486
+ }
487
+ if (!response.ok) {
488
+ let code = `HTTP_${response.status}`;
489
+ try {
490
+ const parsed = JSON.parse(text);
491
+ const candidate = parsed.error?.code ?? parsed.code;
492
+ if (typeof candidate === "string" && /^[A-Za-z_][A-Za-z0-9_-]{0,63}$/.test(candidate))
493
+ code = candidate;
494
+ }
495
+ catch { }
496
+ throw new Error(JSON.stringify({
497
+ code, status: response.status, message: "Inference model discovery failed.",
498
+ instruction: response.status === 404 && path.startsWith("/v1/models/")
499
+ ? "Call list_inference_models for ids this credential can invoke in its bound workspace; do not guess another workspace's id."
500
+ : statusGuidance(response.status),
501
+ }));
502
+ }
503
+ try {
504
+ return JSON.parse(text);
505
+ }
506
+ catch {
507
+ throw new Error(JSON.stringify({
508
+ code: "INVALID_MODELS_RESPONSE", message: "Inference model discovery returned invalid JSON.",
509
+ instruction: "Retry shortly; do not treat this as an empty model roster.",
510
+ }));
511
+ }
512
+ }
513
+ async inferenceModels() {
514
+ const response = await this.inferenceModelRead("/v1/models");
515
+ const roster = response && typeof response === "object" ? response : null;
516
+ const rows = roster?.data;
517
+ if (roster?.object !== "list" || !Array.isArray(rows) ||
518
+ !rows.every(row => row && typeof row.id === "string" && row.id.trim())) {
519
+ throw new Error(JSON.stringify({ code: "INVALID_MODELS_RESPONSE", message: "The inference model roster had no usable data array.", instruction: "Retry shortly; an unavailable source is not an empty roster." }));
520
+ }
521
+ return response;
522
+ }
523
+ async inferenceModel(modelId) {
524
+ const id = typeof modelId === "string" ? modelId.trim() : "";
525
+ if (!id)
526
+ throw new Error("A model id is required; list_inference_models gives the ids this credential can invoke.");
527
+ const response = await this.inferenceModelRead(`/v1/models/${encodeURIComponent(id)}`);
528
+ if (!response || typeof response !== "object" || response.object !== "model" ||
529
+ typeof response.id !== "string") {
530
+ throw new Error(JSON.stringify({ code: "INVALID_MODEL_RESPONSE", message: "Inference model retrieval returned no model id.", instruction: "Retry or list the models this credential can invoke." }));
531
+ }
532
+ return response;
533
+ }
453
534
  async serverlessRead(path, shape, params = {}) {
454
535
  const credentials = [this.authHeaders["X-API-Key"], this.authHeaders.Authorization, this.inferenceAuthHeader, this.serverlessAuthHeader]
455
536
  .map(raw => raw?.replace(/^Bearer\s+/i, "").trim() ?? "")
@@ -688,8 +769,8 @@ export class BiosClient {
688
769
  return (parsed ?? {});
689
770
  }
690
771
  /**
691
- * Select the Authorization header for /v1/chat/completions. A dedicated
692
- * deployment key (inferenceAuthHeader) wins when present; otherwise a
772
+ * Select the Authorization header for the unified /v1 inference verbs. A
773
+ * dedicated deployment key (inferenceAuthHeader) wins when present; otherwise a
693
774
  * serverless-scoped platform key (serverlessAuthHeader) or a hosted OAuth
694
775
  * grant authenticates model-ID-routed calls on the unified endpoint.
695
776
  */
@@ -697,16 +778,28 @@ export class BiosClient {
697
778
  const oauth = this.authHeaders.Authorization?.startsWith("Bearer mcp_at_") ? this.authHeaders.Authorization : "";
698
779
  const auth = this.inferenceAuthHeader || this.serverlessAuthHeader || oauth;
699
780
  if (!auth) {
700
- throw new Error("chat_with_inference requires a key: set RUNBIOS_INFERENCE_KEY for a dedicated deployment, "
701
- + "or provide a workspace platform key (RUNBIOS_API_KEY with serverless scope) to call catalog models by id");
781
+ throw new Error("An inference credential is required: set RUNBIOS_INFERENCE_KEY for a dedicated deployment, "
782
+ + "or provide a workspace platform key (RUNBIOS_API_KEY with serverless scope) to call compatible catalog models by id");
702
783
  }
703
784
  return auth;
704
785
  }
705
- async inferenceApi(body, idempotencyKey, requestId, callerSignal) {
706
- // Route streaming requests through the SSE aggregator so reasoning +
707
- // content deltas surface exactly as produced, collapsed to one result.
786
+ configuredInferenceKeys() {
787
+ return [this.inferenceAuthHeader, this.serverlessAuthHeader, this.authHeaders.Authorization, this.authHeaders["X-API-Key"]]
788
+ .map(raw => raw?.replace(/^Bearer\s+/i, "").trim() ?? "")
789
+ .filter(key => key.length >= 12);
790
+ }
791
+ redactInferenceError(message) {
792
+ return this.configuredInferenceKeys().reduce((text, key) => text.replaceAll(key, "[redacted]"), redactBuildIdentity(message));
793
+ }
794
+ containsInferenceCredential(content) {
795
+ return this.configuredInferenceKeys().some(key => content.includes(key));
796
+ }
797
+ async inferenceApi(body, idempotencyKey, requestId, callerSignal, path = "/v1/chat/completions", anthropicVersion) {
798
+ // Aggregate chat deltas; preserve other task-specific SSE events intact.
708
799
  if (body.stream === true) {
709
- return this.streamInferenceApi(body, idempotencyKey, requestId, callerSignal);
800
+ return path === "/v1/chat/completions"
801
+ ? this.streamInferenceApi(body, idempotencyKey, requestId, callerSignal)
802
+ : this.streamTaskInferenceApi(path, body, idempotencyKey, requestId, callerSignal, anthropicVersion);
710
803
  }
711
804
  const headers = {
712
805
  Authorization: this.resolveInferenceAuth(),
@@ -717,6 +810,8 @@ export class BiosClient {
717
810
  };
718
811
  if (idempotencyKey)
719
812
  headers["Idempotency-Key"] = idempotencyKey;
813
+ if (path === "/v1/messages")
814
+ headers["Anthropic-Version"] = anthropicVersion || "2023-06-01";
720
815
  // Inference POSTs are intentionally dispatched once. An MCP caller that
721
816
  // retries can reuse idempotency_key; this client never replays implicitly.
722
817
  const controller = new AbortController();
@@ -729,7 +824,7 @@ export class BiosClient {
729
824
  let response;
730
825
  let text;
731
826
  try {
732
- response = await fetch(`${this.inferenceBaseUrl}/v1/chat/completions`, {
827
+ response = await fetch(`${this.inferenceBaseUrl}${path}`, {
733
828
  method: "POST",
734
829
  headers,
735
830
  body: JSON.stringify(body),
@@ -746,14 +841,17 @@ export class BiosClient {
746
841
  // MCP client sees them as the tool error.
747
842
  try {
748
843
  const parsed = JSON.parse(text);
749
- throw new Error(redactBuildIdentity(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
844
+ throw new Error(this.redactInferenceError(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
750
845
  }
751
846
  catch (error) {
752
847
  if (error instanceof SyntaxError)
753
- throw new Error(redactBuildIdentity(text || `Inference API error ${response.status}`));
848
+ throw new Error(this.redactInferenceError(text || `Inference API error ${response.status}`));
754
849
  throw error;
755
850
  }
756
851
  }
852
+ if (this.containsInferenceCredential(text)) {
853
+ throw new Error("Inference result contained authentication material and was withheld.");
854
+ }
757
855
  let parsed;
758
856
  try {
759
857
  parsed = JSON.parse(text);
@@ -806,17 +904,76 @@ export class BiosClient {
806
904
  const text = await response.text().catch(() => "");
807
905
  try {
808
906
  const parsed = JSON.parse(text);
809
- throw new Error(redactBuildIdentity(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
907
+ throw new Error(this.redactInferenceError(parsed?.error?.message || parsed?.message || `Inference API error ${response.status}`));
810
908
  }
811
909
  catch (error) {
812
910
  if (error instanceof SyntaxError)
813
- throw new Error(redactBuildIdentity(text || `Inference API error ${response.status}`));
911
+ throw new Error(this.redactInferenceError(text || `Inference API error ${response.status}`));
814
912
  throw error;
815
913
  }
816
914
  }
817
915
  if (!response.body)
818
916
  throw new Error("Inference stream returned no body");
819
- return await aggregateChatStream(response.body);
917
+ const result = await aggregateChatStream(response.body);
918
+ if (this.containsInferenceCredential(JSON.stringify(result))) {
919
+ throw new Error("Inference result contained authentication material and was withheld.");
920
+ }
921
+ return result;
922
+ }
923
+ catch (error) {
924
+ if (error instanceof Error)
925
+ throw new Error(this.redactInferenceError(error.message));
926
+ throw error;
927
+ }
928
+ finally {
929
+ clearTimeout(timeout);
930
+ callerSignal?.removeEventListener("abort", relayAbort);
931
+ }
932
+ }
933
+ async streamTaskInferenceApi(path, body, idempotencyKey, requestId, callerSignal, anthropicVersion) {
934
+ const headers = {
935
+ Authorization: this.resolveInferenceAuth(),
936
+ Accept: "text/event-stream",
937
+ "Content-Type": "application/json",
938
+ "X-Request-ID": requestId || randomUUID(),
939
+ "User-Agent": this.userAgent,
940
+ };
941
+ if (idempotencyKey)
942
+ headers["Idempotency-Key"] = idempotencyKey;
943
+ if (path === "/v1/messages")
944
+ headers["Anthropic-Version"] = anthropicVersion || "2023-06-01";
945
+ const controller = new AbortController();
946
+ const relayAbort = () => controller.abort(callerSignal?.reason);
947
+ if (callerSignal?.aborted)
948
+ relayAbort();
949
+ else
950
+ callerSignal?.addEventListener("abort", relayAbort, { once: true });
951
+ const timeout = setTimeout(() => controller.abort(new Error(`Inference timed out after ${this.inferenceTimeoutMs}ms`)), this.inferenceTimeoutMs);
952
+ try {
953
+ const response = await fetch(`${this.inferenceBaseUrl}${path}`, {
954
+ method: "POST", headers, body: JSON.stringify(body), signal: controller.signal,
955
+ });
956
+ if (!response.ok) {
957
+ const text = await response.text().catch(() => "");
958
+ let message = text || `Inference API error ${response.status}`;
959
+ try {
960
+ const parsed = JSON.parse(text);
961
+ message = parsed?.error?.message || parsed?.message || message;
962
+ }
963
+ catch { }
964
+ throw new Error(this.redactInferenceError(message));
965
+ }
966
+ if (!response.headers.get("content-type")?.toLowerCase().includes("text/event-stream")) {
967
+ throw new Error("Inference endpoint did not return an event stream");
968
+ }
969
+ if (!response.body)
970
+ throw new Error("Inference stream returned no body");
971
+ return await collectInferenceEvents(response.body, data => this.containsInferenceCredential(data), path === "/v1/messages" ? "anthropic" : "openai");
972
+ }
973
+ catch (error) {
974
+ if (error instanceof Error)
975
+ throw new Error(this.redactInferenceError(error.message));
976
+ throw error;
820
977
  }
821
978
  finally {
822
979
  clearTimeout(timeout);
@@ -842,10 +999,15 @@ export async function aggregateChatStream(stream) {
842
999
  let model;
843
1000
  let created;
844
1001
  let usage;
1002
+ let completed = false;
845
1003
  const toolCalls = [];
846
1004
  const consume = (payload) => {
847
1005
  const data = payload.trim();
848
- if (!data || data === "[DONE]")
1006
+ if (data === "[DONE]") {
1007
+ completed = true;
1008
+ return;
1009
+ }
1010
+ if (!data)
849
1011
  return;
850
1012
  let event;
851
1013
  try {
@@ -897,7 +1059,7 @@ export async function aggregateChatStream(stream) {
897
1059
  }
898
1060
  };
899
1061
  try {
900
- for (;;) {
1062
+ while (!completed) {
901
1063
  const { done, value } = await reader.read();
902
1064
  if (done)
903
1065
  break;
@@ -909,16 +1071,22 @@ export async function aggregateChatStream(stream) {
909
1071
  const frame = buffer.slice(0, match.index);
910
1072
  buffer = buffer.slice(match.index + match[0].length);
911
1073
  emitFrame(frame, consume);
1074
+ if (completed)
1075
+ break;
912
1076
  }
913
1077
  }
914
- buffer += decoder.decode();
915
- if (buffer.trim())
916
- emitFrame(buffer, consume);
1078
+ if (!completed) {
1079
+ buffer += decoder.decode();
1080
+ if (buffer.trim())
1081
+ emitFrame(buffer, consume);
1082
+ }
917
1083
  }
918
1084
  finally {
919
1085
  await reader.cancel().catch(() => undefined);
920
1086
  reader.releaseLock();
921
1087
  }
1088
+ if (!completed)
1089
+ throw new Error("Inference stream ended before completion marker [DONE]; no finished result was returned.");
922
1090
  const message = { role, content: content || null };
923
1091
  if (reasoning)
924
1092
  message.reasoning_content = reasoning;
@@ -959,4 +1127,83 @@ function emitFrame(frame, consume) {
959
1127
  if (data.length > 0)
960
1128
  consume(data.join("\n"));
961
1129
  }
1130
+ export async function collectInferenceEvents(stream, containsCredential, termination = "openai") {
1131
+ const reader = stream.getReader();
1132
+ const decoder = new TextDecoder();
1133
+ const events = [];
1134
+ let buffer = "";
1135
+ let size = 0;
1136
+ let completed = false;
1137
+ const consume = (payload) => {
1138
+ const data = payload.trim();
1139
+ if (data === "[DONE]") {
1140
+ if (termination === "anthropic")
1141
+ throw new Error("Anthropic stream ended without message_stop");
1142
+ completed = true;
1143
+ return;
1144
+ }
1145
+ if (!data)
1146
+ return;
1147
+ if (containsCredential?.(data))
1148
+ throw new Error("Inference result contained authentication material and was withheld.");
1149
+ if (size + data.length > 2_000_000 || events.length >= 4096) {
1150
+ throw new Error("Inference stream exceeded the MCP result limit; reduce max_tokens and retry explicitly.");
1151
+ }
1152
+ let event;
1153
+ try {
1154
+ event = JSON.parse(data);
1155
+ }
1156
+ catch {
1157
+ throw new Error("Inference stream returned invalid JSON");
1158
+ }
1159
+ if (!event || typeof event !== "object" || Array.isArray(event)) {
1160
+ throw new Error("Inference stream returned an invalid event");
1161
+ }
1162
+ const record = event;
1163
+ if (record.error) {
1164
+ const err = record.error;
1165
+ const candidate = err && typeof err === "object" ? err.code : undefined;
1166
+ const code = typeof candidate === "string" && /^[A-Za-z_][A-Za-z0-9_-]{0,63}$/.test(candidate) ? candidate : "stream_error";
1167
+ throw new Error(`Inference stream failed (${code}); no completed result was returned.`);
1168
+ }
1169
+ events.push(record);
1170
+ size += data.length;
1171
+ if (termination === "anthropic" && record.type === "message_stop")
1172
+ completed = true;
1173
+ };
1174
+ try {
1175
+ while (!completed) {
1176
+ const { done, value } = await reader.read();
1177
+ if (done)
1178
+ break;
1179
+ buffer += decoder.decode(value, { stream: true });
1180
+ if (buffer.length > 2_000_000)
1181
+ throw new Error("Inference stream exceeded the MCP result limit; reduce max_tokens.");
1182
+ for (;;) {
1183
+ const match = /\r\n\r\n|\n\n|\r\r/.exec(buffer);
1184
+ if (!match || match.index === undefined)
1185
+ break;
1186
+ const frame = buffer.slice(0, match.index);
1187
+ buffer = buffer.slice(match.index + match[0].length);
1188
+ emitFrame(frame, consume);
1189
+ if (completed)
1190
+ break;
1191
+ }
1192
+ }
1193
+ if (!completed) {
1194
+ buffer += decoder.decode();
1195
+ if (buffer.trim())
1196
+ emitFrame(buffer, consume);
1197
+ }
1198
+ }
1199
+ finally {
1200
+ await reader.cancel().catch(() => undefined);
1201
+ reader.releaseLock();
1202
+ }
1203
+ if (events.length === 0)
1204
+ throw new Error("Inference stream finished without a result");
1205
+ if (!completed)
1206
+ throw new Error("Inference stream ended before completion; no finished result was returned.");
1207
+ return { object: "inference.event_stream", events };
1208
+ }
962
1209
  //# sourceMappingURL=api-client.js.map