runbios-mcp 0.2.1 → 0.2.2-dev.174

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -572,29 +572,57 @@ launchGates = resolveLaunchGates()) {
572
572
  version: VERSION,
573
573
  }, {
574
574
  instructions: (launchGates.training
575
- ? "Run BiOS serves models on its own GPU control plane. Call get_platform_guide first.\n"
576
- : "Run BiOS trains and serves models on its own GPU control plane. Call get_platform_guide first.\n")
575
+ ? "Run BiOS serves models through TWO distinct inference products: serverless inference and dedicated deployments. Call get_platform_guide first and choose the product before calling anything.\n"
576
+ : "Run BiOS provides serverless inference, dedicated deployments, datasets, and training. Call get_platform_guide first and choose the workflow before calling anything.\n")
577
577
  + prelaunchNote
578
578
  + "Rules that no tool schema can express:\n"
579
+ + "1. Identity: every tool acts only in the API key's bound organization and workspace. Call introspect_api_key first; never invent or substitute an org/workspace/account.\n"
580
+ + "2. Serverless is the default for immediate inference: no GPU booking, no deployment object, per-token billing. Use a published model id and chat_with_inference. Do NOT call create_inference for serverless.\n"
581
+ + "3. Dedicated inference reserves GPUs and bills per second. Use get_inference_gpu_options -> preflight_inference -> explicit spend consent -> create_inference -> poll get_inference_status until running -> chat_with_inference.\n"
579
582
  + (launchGates.training
580
- ? "1. Models: only ids in the Run BiOS catalog can be served (list_models / search_models). "
581
- : "1. Models: only ids in the Run BiOS catalog can be trained or served (list_models / search_models). ")
582
- + "This is not a Hugging Face lookup; an id outside the catalog is rejected, never mirrored on demand.\n"
583
- + "2. Serving is platform-derived: serving mode (full/adapter/merged), the OpenAI task "
584
- + "(chat, or completion for a base model with no chat template, or embedding/rerank), and image support are "
585
- + "derived from the model and checkpoint lineage. They are not tool inputs — read them from the "
586
- + "preflight/create/status response.\n"
587
- + `3. Context window: the default is min(model native max, ${CONTEXT_DEFAULT_CEILING}) tokens, falling back to `
588
- + `${CONTEXT_UNKNOWN_FALLBACK} when the native window is unknown, and the hard ceiling is the model's own native max. `
589
- + "Read model.native_max_context from preflight_inference before proposing a value.\n"
590
- + "4. GPUs and money: never substitute a GPU or exceed an approved price cap without asking. A capacity "
591
- + "rejection arrives as compact JSON with bookable alternatives; a 503 is a transient outage, never an out-of-stock verdict.\n"
592
- + "5. Report an endpoint as live only when the status says running.",
583
+ ? "4. Models: only ids in the Run BiOS catalog can be served (list_models / search_models). "
584
+ : "4. Models: only ids in the Run BiOS catalog can be trained or served (list_models / search_models). ")
585
+ + "This is not a live Hugging Face lookup; an id outside the catalog is rejected, never mirrored on demand.\n"
586
+ + "5. Dedicated serving is platform-derived: serving mode (full/adapter/merged), OpenAI task, and image support come from model/checkpoint lineage. Read them from preflight/create/status; do not invent them.\n"
587
+ + `6. Context window: the default is min(model native max, ${CONTEXT_DEFAULT_CEILING}) tokens, falling back to `
588
+ + `${CONTEXT_UNKNOWN_FALLBACK} when unknown. Read model.native_max_context from preflight_inference.\n`
589
+ + "7. GPUs and money: never substitute a GPU, create a paid resource, or exceed a price cap without explicit user approval. A 503 is transient, never proof of no capacity.\n"
590
+ + "8. Report a dedicated endpoint as live only when get_inference_status says running; then call it with chat_with_inference. Serverless needs no such poll.",
593
591
  });
594
592
  // The registrations below are moved verbatim from the original single-file
595
593
  // entry and intentionally keep their original indentation. `server.tool(...)`
596
594
  // is a thin delegate that adds the onToolCall hook around each handler.
597
595
  const registeredTools = new Set();
596
+ // MODEL_CATALOG_PATH is the ONE model-discovery endpoint the platform serves:
597
+ // GET /api/public/model-catalog (model-service/cmd/main.go registers it, with
598
+ // /stats and /{author}/{name} beside it).
599
+ //
600
+ // list_models and search_models both used to call `/api/public/model-search`,
601
+ // which no service has ever registered. Measured on production 2026-09-19 it
602
+ // answered HTTP 463 with an empty body from the load balancer, so BOTH tools
603
+ // failed for every caller — and they are the two the descriptions tell an agent
604
+ // to call FIRST, before create_training_job or create_inference. Any agent
605
+ // following the documented flow hit a dead end on step one.
606
+ //
607
+ // One constant, not two string literals: this endpoint moved once already
608
+ // (catalog_detail_route_test.go records the last time a prefix and its
609
+ // registrations drifted apart), and two copies is how one of them gets missed.
610
+ const MODEL_CATALOG_PATH = "/api/public/model-catalog";
611
+ // billionsParam coerces a human parameter count ("7B", "70b", " 13B ") into the
612
+ // bare number the catalog filter parses.
613
+ //
614
+ // The tool advertises "e.g. '7B', '13B', '70B'" and the handler runs ParseFloat,
615
+ // which fails on a trailing B and yields 0 — and 0 means "no filter" there. So
616
+ // every documented spelling of max_params was accepted and then silently
617
+ // ignored, which is worse than rejecting it: the caller gets a plausible,
618
+ // unfiltered answer. A value with no digits is left alone for the server to
619
+ // judge rather than silently dropped here.
620
+ function billionsParam(v) {
621
+ if (v === undefined)
622
+ return undefined;
623
+ const m = /^\s*([0-9]*\.?[0-9]+)\s*[bB]?\s*$/.exec(v);
624
+ return m ? m[1] : v;
625
+ }
598
626
  const datasetMixingSchema = z.object({
599
627
  mode: z.enum(["sequential", "shuffle", "interleave", "phased"]),
600
628
  seed: z.number().int().min(Number.MIN_SAFE_INTEGER).max(Number.MAX_SAFE_INTEGER).optional(),
@@ -657,6 +685,11 @@ launchGates = resolveLaunchGates()) {
657
685
  "training_methods",
658
686
  "gpu_selection",
659
687
  "inference",
688
+ "serverless_inference",
689
+ "dedicated_deployment",
690
+ "trained_model_deployment",
691
+ "identity_and_scope",
692
+ "end_to_end",
660
693
  "hyperparameters",
661
694
  "monitoring",
662
695
  "cost_optimization",
@@ -682,15 +715,26 @@ ${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "are" : "is"} not av
682
715
  What still works: every read and lifecycle tool for existing datasets and training jobs (list_datasets, preview_dataset, delete_dataset, list_training_jobs, get_training_status, get_training_metrics, get_training_logs, get_training_evals, stop_training_job, resume_training_job, get_training_checkpoints, get_checkpoint_download_manifest, delete_training_checkpoint), and the full inference surface (preflight_inference, create_inference, get_inference_status, chat_with_inference).`;
683
716
  const gatedOverview = `# Run BiOS Platform
684
717
 
685
- Run BiOS serves models on its own GPU control plane: a curated catalog of verified models — mirrored in its own storage, never fetched live from Hugging Face — deployed to dedicated OpenAI-compatible endpoints.
686
-
687
- ## Core workflow (serving, fully live)
688
- 1. **Find a model** — search_models / list_models read the hosted catalog; these are the only models that can be deployed
689
- 2. **Check GPU fit and price** — get_inference_gpu_options joins model-fit counts to live stock and total hourly prices
690
- 3. **Preflight** — preflight_inference validates the exact request with no wallet, queue, database, or GPU side effects
691
- 4. **Create the deployment** — create_inference books capacity book-before-reveal; a definitive miss returns CAPACITY_UNAVAILABLE and nothing exists
692
- 5. **Run** — poll get_inference_status until status is running; call the endpoint with chat_with_inference
693
- 6. **Control spend** — stop_inference / resume_inference / restart_inference / delete_inference
718
+ Run BiOS has TWO inference products. Choosing the right one is step zero; they are not two names for the same operation.
719
+
720
+ ## Serverless inference — immediate, no deployment
721
+ Use this by default for published Run BiOS model ids. There is no GPU selection, booking, deployment object, provisioning wait, stop/delete lifecycle, or hourly price cap. It is routed by the platform and billed per token.
722
+ 1. Call introspect_api_key to confirm the bound workspace and serverless scope.
723
+ 2. Read the published ids and capabilities from the serverless catalog; do not substitute a dedicated catalog repo id for a serverless slug.
724
+ 3. Call chat_with_inference with that model id, messages, and only capabilities the model advertises. Streaming is optional.
725
+ 4. The completion is the result. Do NOT call preflight_inference or create_inference for serverless.
726
+
727
+ ## Dedicated deployment — reserved GPU, durable endpoint
728
+ Choose this for dedicated capacity, custom/trained checkpoints, or a durable endpoint under your control. It creates a paid resource and bills GPU time per second.
729
+ 1. **Find a source** — search_models/list_models for a catalog base model, or get_training_checkpoints for a verified trained checkpoint.
730
+ 2. **Check identity and money** — introspect_api_key and get_wallet_balance. Never invent another org/workspace.
731
+ 3. **Check GPU fit and price** — get_inference_gpu_options.
732
+ 4. **Preflight** — preflight_inference validates the exact request with no wallet, queue, database, or GPU side effects. Read canonical_request, alternatives, queue_required and billing from the response.
733
+ 5. **Ask for approval** — before create_inference, state the chosen GPU, count, hourly cap, storage and queue consent. Never substitute an alternative silently.
734
+ 6. **Create** — create_inference; retain the returned id and one-time key securely. A booking handle must be polled with get_inference_booking.
735
+ 7. **Wait for reality** — poll get_inference_status through queue/provisioning/loading; never report the endpoint live until status=running.
736
+ 8. **Run** — call chat_with_inference using the dedicated inference key or deployment identity.
737
+ 9. **Observe/control/clean up** — metrics, notifications, stop/resume/restart, then delete_inference when the user explicitly asks to remove the paid deployment.
694
738
 
695
739
  ## Launching soon — not offered in this build
696
740
  ${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "launch" : "launches"} soon: the creation tools are not registered, so no tool call — and no workaround — can start one today. Existing datasets and training jobs are unaffected: every read and lifecycle tool below still works.
@@ -787,17 +831,31 @@ END TO END: capture what a model was asked and answered, have people or judges s
787
831
  const guides = {
788
832
  overview: `# Run BiOS Fine-Tuning Platform
789
833
 
790
- Run BiOS is a cloud fine-tuning platform for large language models. It hosts a curated catalog of verified base models — mirrored in its own storage, never fetched live from Hugging Face — trained natively by the BIOS training engine (supervised fine-tuning and continued pre-training; vision-language models train through SFT).
834
+ Run BiOS provides serverless inference, dedicated GPU deployments, datasets, and model training. These are separate products with different billing and lifecycle semantics.
835
+
836
+ ## Choose the workflow before calling tools
837
+ - **Need an answer now from a published serverless model?** Use serverless inference: model id + chat_with_inference. No GPU booking or deployment object; billed per token.
838
+ - **Need dedicated capacity or a durable endpoint?** Use dedicated deployment: GPU options -> preflight -> explicit approval -> create -> wait for running -> chat -> stop/delete. Billed per second of GPU time.
839
+ - **Need custom weights?** Prepare a dataset -> preflight and run training -> monitor -> select a verified checkpoint -> deploy that checkpoint through the dedicated workflow.
840
+ - **Need continuous improvement from real conversations?** Use the Conscious Loop workflow; capture is opt-in and automated model calls spend through the workspace's own serverless account.
791
841
 
792
- ## Core Workflow
793
- 1. **Upload a dataset** — JSONL, Parquet, or CSV. The platform auto-validates format, detects columns, and maps fields.
842
+ ## Training Workflow
843
+ 1. **Upload/import a dataset** — use upload_dataset for a local JSONL file or import_huggingface_dataset for a Hub source. Poll get_dataset_status until ready, then preview_dataset before training. A created row is not proof that processing succeeded.
794
844
  2. **Choose a base model** — Search the hosted Run BiOS catalog (search_models / list_models); every listed model is verified end-to-end and these are the only models that can be trained or deployed
795
845
  3. **Pick a training method** — SFT for instruction tuning (including vision-language models), CPT for continued pre-training on raw domain text
796
846
  4. **Select adapter** — LoRA (fast, efficient), QLoRA (lower VRAM), or Full fine-tune (max quality)
797
847
  5. **Configure hyperparameters** — learning rate, batch size, epochs, LoRA rank, etc. Smart defaults provided.
798
848
  6. **Choose a GPU** — A100 80GB, H100, A6000 etc. Platform recommends based on model size.
799
- 7. **Launch and monitor** — Real-time loss curves, eval metrics, training logs, checkpoint saving
800
- 8. **Download or deploy** — Get your fine-tuned model weights or merged checkpoints
849
+ 7. **Launch and monitor** — create_training_job only after preflight and explicit spend approval. Poll get_training_status; read metrics, evals and logs instead of inferring progress from elapsed time. Stop/resume only when available actions permit it.
850
+ 8. **Select a checkpoint** — get_training_checkpoints; use a verified/succeeded checkpoint. A running job or incomplete checkpoint is not deployable.
851
+ 9. **Deploy the trained model** — preflight_inference with source_type=checkpoint, source_job_id and source_checkpoint_id, approve GPU/cost, then create_inference. Serving mode and base-model lineage are derived; do not guess them.
852
+ 10. **Wait, infer, observe** — poll get_inference_status until running, call chat_with_inference, inspect metrics/notifications, then stop or delete when explicitly requested.
853
+ 11. **Clean up safely** — delete_training_checkpoint or delete_dataset only after confirming it is no longer needed. Deleting source data/checkpoints is irreversible and is never implied by stopping training or deleting a deployment.
854
+
855
+ ## Identity, account and scope
856
+ - Every MCP call uses the API key's bound organization and workspace. Start with introspect_api_key and repeat the resolved identity before any paid or destructive action.
857
+ - Never ask an agent to choose an arbitrary account id; tool responses are scoped server-side. A different workspace requires a different authorized credential.
858
+ - get_wallet_balance describes the same billing owner that training, dedicated deployment, and automated Loop calls spend from.
801
859
 
802
860
  ## What the PLATFORM decides for you (do not try to set these)
803
861
  Serving settings are derived from the model itself and are immutable. They are
@@ -1052,9 +1110,78 @@ When to use: Before SFT, when your domain has specialized vocabulary/knowledge.
1052
1110
  - Memory depends on the full configuration and input shapes, not parameter count alone. Do not invent a smaller minimum or claim that quantization always halves total memory.
1053
1111
  - Full fine-tuning is not a guarantee of higher accuracy than adapters. Use representative validation to compare outcomes.
1054
1112
  - Preflight admission is not proof of universal freedom from hardware faults or OOM. Report actual tested configurations and measured headroom.`,
1113
+ serverless_inference: `# Serverless Inference
1114
+
1115
+ Serverless is the immediate, per-token inference product. It does NOT create or manage a deployment and does not reserve a GPU.
1116
+
1117
+ ## Correct flow
1118
+ 1. Call introspect_api_key. Confirm the key's bound workspace and serverless permission.
1119
+ 2. Select a published serverless model id and read its advertised capabilities. A dedicated catalog repository id and a serverless model slug are not interchangeable.
1120
+ 3. Call chat_with_inference with the model id and messages. Optional surfaces include stream, tools/tool_choice, response_format and reasoning_effort only when the chosen model advertises them.
1121
+ 4. Read the returned completion. There is no provisioning status to poll and nothing to stop/delete.
1122
+
1123
+ ## Do not do this
1124
+ - Do not call get_inference_gpu_options, preflight_inference or create_inference for serverless.
1125
+ - Do not report an unsupported capability as an outage; an honest 400 means choose a model that advertises it.
1126
+ - Do not retry an empty completion blindly: finish_reason=length means max_tokens was exhausted, not that the endpoint lost data.
1127
+
1128
+ ## Billing
1129
+ Serverless is billed per token to the API key's bound workspace. It is distinct from dedicated GPU billing and from the Loop agent, even though the Loop agent spends through that workspace's serverless account.`,
1130
+ dedicated_deployment: `# Dedicated Model Deployment
1131
+
1132
+ A dedicated deployment reserves GPU capacity and creates a durable endpoint. Use it for dedicated capacity, custom/trained checkpoints, or lifecycle control. It bills GPU time per second and is NOT required for serverless inference.
1133
+
1134
+ ## Base-model flow
1135
+ 1. introspect_api_key and get_wallet_balance.
1136
+ 2. search_models/list_models, then get_inference_gpu_options.
1137
+ 3. preflight_inference: read canonical_request, selected_gpu, alternatives, queue_required, native context and billing.
1138
+ 4. State the exact GPU/count, storage, maximum hourly price and queue consent; wait for explicit user approval.
1139
+ 5. create_inference using canonical_request. Keep the deployment id and one-time inference key secure. If a booking handle is returned, poll get_inference_booking.
1140
+ 6. Poll get_inference_status through queued/provisioning/loading. Only status=running proves the endpoint is live.
1141
+ 7. chat_with_inference with the dedicated key/deployment identity.
1142
+ 8. get_inference_metrics and get_inference_notifications; stop/resume/restart only when requested.
1143
+ 9. delete_inference is irreversible and only follows explicit confirmation that the paid deployment should be removed.
1144
+
1145
+ Capacity errors include bookable alternatives. Never silently substitute a GPU or raise a price cap. A 503 is transient and is not proof that capacity is unavailable.`,
1146
+ trained_model_deployment: launchGates.training
1147
+ ? `# Deploying a Trained Model\n\nTraining creation is not available in this deployment. Existing verified checkpoints can still be read with get_training_checkpoints and deployed through the dedicated workflow when the key has access. Do not claim a running job or incomplete checkpoint is deployable.`
1148
+ : `# Train, Deploy, and Infer a Custom Model
1149
+
1150
+ 1. Prepare data with upload_dataset or import_huggingface_dataset. Poll get_dataset_status until ready and preview_dataset to verify the actual rows/format.
1151
+ 2. Choose a catalog base model; call get_training_capabilities, get_model_config, get_recommended_gpu/recommend_training_config, then preflight_training_job.
1152
+ 3. State method, adapter, GPU/count, dataset sampling, maximum hourly price and estimated bound; obtain explicit spend approval.
1153
+ 4. create_training_job. Poll get_training_status and read metrics, evals and logs. Do not translate elapsed time into progress or success.
1154
+ 5. After success, call get_training_checkpoints and select a verified checkpoint.
1155
+ 6. Call preflight_inference with source_type=checkpoint, source_job_id and source_checkpoint_id. Serving mode/base lineage are derived from the checkpoint; do not invent them.
1156
+ 7. Obtain deployment spend approval, create_inference, poll until running, then chat_with_inference.
1157
+ 8. Observe metrics/notifications and clean up only on explicit request. Deleting the deployment does not delete the checkpoint or dataset; each destructive operation is separate and irreversible.`,
1158
+ identity_and_scope: `# Identity, Account, and Scope
1159
+
1160
+ Every MCP tool acts inside the API key's bound organization and workspace.
1161
+ 1. Call introspect_api_key first. Read user/workspace/org, scopes, allowed tools and feature grants from the result.
1162
+ 2. Call get_wallet_balance before any training, dedicated deployment or automated Loop action that spends.
1163
+ 3. Never invent, guess or substitute another org/workspace/account id. A different workspace requires a different authorized credential.
1164
+ 4. Serverless calls, dedicated deployments, training jobs, datasets and Loop artifacts remain scoped to that resolved workspace. Empty lists may mean this workspace has no resources; they do not prove the platform has none globally.
1165
+ 5. Repeat the resolved workspace and cost owner before asking for spend or destructive consent, without exposing the credential itself.`,
1166
+ end_to_end: launchGates.training || launchGates.datasets
1167
+ ? `# End-to-End Availability\n\nServerless inference and dedicated deployment are fully live. Start with identity_and_scope, then choose serverless_inference or dedicated_deployment. Dataset/training creation is gated in this deployment; do not invent a workaround. Existing datasets, jobs and checkpoints remain readable and manageable through their lifecycle tools.`
1168
+ : `# End-to-End Platform Flow
1169
+
1170
+ 1. Identity: introspect_api_key and get_wallet_balance.
1171
+ 2. Data: upload/import -> get_dataset_status until ready -> preview_dataset.
1172
+ 3. Plan training: catalog model -> capabilities/config -> GPU recommendation -> preflight_training_job.
1173
+ 4. Consent and train: explicit spend approval -> create_training_job -> status/metrics/evals/logs -> verified checkpoint.
1174
+ 5. Deploy checkpoint: preflight_inference(source_type=checkpoint) -> explicit deployment approval -> create_inference -> status=running.
1175
+ 6. Infer: chat_with_inference -> metrics/notifications.
1176
+ 7. Improve: optional Conscious Loop capture/review/evaluation/training rule; capture and spending are separate opt-ins.
1177
+ 8. Clean up: stop/delete deployment, checkpoint or dataset only as separate explicit irreversible actions.
1178
+
1179
+ At every step use returned ids and available_actions. Never infer the next action from a status label alone, never report success before the authoritative terminal state, and never substitute a GPU/model/price without approval.`,
1055
1180
  inference: `# Model Inference Guide
1056
1181
 
1057
- Deployments serve either a verified fine-tuning checkpoint or a base model from the catalog through an OpenAI-compatible endpoint.
1182
+ Run BiOS has two inference products. Use serverless_inference for immediate per-token calls with no deployment, and dedicated_deployment for reserved GPU capacity or trained checkpoints.
1183
+
1184
+ Dedicated deployments serve either a verified fine-tuning checkpoint or a base model from the catalog through an OpenAI-compatible endpoint.
1058
1185
 
1059
1186
  ## Settings the platform derives (not tool inputs)
1060
1187
  serving_mode, model_task and supports_images are SERVER-DERIVED and immutable,
@@ -1341,8 +1468,8 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1341
1468
  // ("author/name"), so folding the provider into q turns it into a real
1342
1469
  // author filter instead of a silently ignored parameter.
1343
1470
  const q = [provider, query].filter(Boolean).join(" ").trim() || undefined;
1344
- const data = await client.api("/api/public/model-search", {
1345
- params: { q, max_params },
1471
+ const data = await client.api(MODEL_CATALOG_PATH, {
1472
+ params: { q, max_params: billionsParam(max_params) },
1346
1473
  });
1347
1474
  return json(data);
1348
1475
  });
@@ -1357,7 +1484,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1357
1484
  limit: z.number().int().min(1).max(60).optional().describe("Page size (default 24, max 60)."),
1358
1485
  offset: z.number().int().min(0).optional().describe("Pagination offset (default 0)."),
1359
1486
  }, async ({ type, sort, limit, offset }) => {
1360
- const data = await client.api("/api/public/model-search", {
1487
+ const data = await client.api(MODEL_CATALOG_PATH, {
1361
1488
  params: {
1362
1489
  type: type && type !== "all" ? type : undefined,
1363
1490
  sort,
@@ -1960,6 +2087,12 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1960
2087
  formData.append("file", blob, basename(absPath));
1961
2088
  formData.append("name", fileName);
1962
2089
  formData.append("size", fileStat.size.toString());
2090
+ // Dataset-service's multipart contract requires workspace_id in the BODY.
2091
+ // JSON routes can rely on the gateway-injected header, which is why every
2092
+ // read worked while upload answered `workspace_id: Field required` for a
2093
+ // perfectly valid API key. Hosted MCP has the OAuth grant's workspace;
2094
+ // local stdio resolves the key's binding once through introspection.
2095
+ formData.append("workspace_id", await client.resolvedWorkspaceId());
1963
2096
  const data = await client.api("/api/datasets/upload", {
1964
2097
  method: "POST",
1965
2098
  formData,