runbios-mcp 0.2.1-dev.97 → 0.2.1-rc.118

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -594,9 +594,47 @@ launchGates = resolveLaunchGates()) {
594
594
  // The registrations below are moved verbatim from the original single-file
595
595
  // entry and intentionally keep their original indentation. `server.tool(...)`
596
596
  // is a thin delegate that adds the onToolCall hook around each handler.
597
+ const registeredTools = new Set();
598
+ const datasetMixingSchema = z.object({
599
+ mode: z.enum(["sequential", "shuffle", "interleave", "phased"]),
600
+ seed: z.number().int().min(Number.MIN_SAFE_INTEGER).max(Number.MAX_SAFE_INTEGER).optional(),
601
+ weights: z.array(z.number().nonnegative()).min(1).max(64).optional(),
602
+ phases: z.array(z.object({
603
+ name: z.string().max(255).optional(),
604
+ portion: z.number().positive().max(1),
605
+ weights: z.array(z.number().nonnegative()).min(1).max(64).optional(),
606
+ shuffle: z.boolean().optional(),
607
+ })).min(1).optional(),
608
+ });
609
+ const trainingConfigurationSchema = {
610
+ dataset_sample_limits: z.record(z.string(), z.number().int().min(0)).optional().describe("Per-dataset first-N row caps; keys must be selected dataset IDs. Omit or use 0 for the full source."),
611
+ epochs: z.number().positive().max(1000).optional().describe("Number of training epochs. Governs the full run when max_steps is omitted (the default)."),
612
+ max_steps: z.number().int().min(1).max(10_000_000).optional().describe("Optional hard step cap that overrides epochs. Omit for no cap and all configured epochs; setting less than one epoch intentionally trains only part of the data."),
613
+ learning_rate: z.number().positive().max(1).optional().describe("Learning rate."),
614
+ batch_size: z.number().int().min(1).max(65536).optional().describe("Per-device train batch size."),
615
+ gradient_accumulation_steps: z.number().int().min(1).max(65536).optional(),
616
+ max_seq_length: trainingSeqLengthSchema(),
617
+ lora_rank: z.number().int().min(1).max(4096).optional(),
618
+ lora_alpha: z.number().positive().max(1_000_000).optional(),
619
+ warmup_ratio: z.number().min(0).max(1).optional(),
620
+ weight_decay: z.number().min(0).max(10).optional(),
621
+ scheduler: z
622
+ .enum(["linear", "cosine", "cosine_with_restarts", "polynomial", "constant", "constant_with_warmup"])
623
+ .optional()
624
+ .describe("Learning-rate scheduler."),
625
+ job_name: z.string().max(255).optional(),
626
+ eval_split_ratio: z.number().min(0).max(0.99).optional(),
627
+ eval_steps: z.number().int().positive().optional().describe("Evaluate every N optimizer steps when a validation split exists."),
628
+ save_steps: z.number().int().positive().optional().describe("Checkpoint interval in optimizer steps."),
629
+ early_stopping_patience: z.number().int().min(0).max(20).optional()
630
+ .describe("Stop after N evaluations without a better validation loss (0 = off; default 3 when eval_split_ratio is set). The best checkpoint is always saved."),
631
+ overlong_policy: z.enum(["truncate", "drop"]).optional()
632
+ .describe("SFT only. 'truncate' trims oldest context while preserving a trainable answer; 'drop' skips overlong samples. CPT packs raw text and does not use this policy."),
633
+ };
597
634
  const server = {
598
635
  tool(name, description, paramsSchema, cb) {
599
636
  mcp.tool(name, description, paramsSchema, wrapToolHandler(name, cb, hooks?.onToolCall));
637
+ registeredTools.add(name);
600
638
  },
601
639
  };
602
640
  /* ══════════════════════════════════════════════════════════════════════════ */
@@ -628,12 +666,10 @@ launchGates = resolveLaunchGates()) {
628
666
  a stand-in that says what launches soon and names only tools that ARE
629
667
  registered (reads/lifecycle stay live). At launch the gates open and the
630
668
  real topics return verbatim. */
631
- // 42 tools with every creation tool registered; the gates hide three
632
- // (create_training_job, upload_dataset, import_huggingface_dataset). The
633
- // coming-soon gate test pins this claim to the wire count, so drift fails.
634
- const liveToolCount = 42
635
- - (launchGates.training ? 1 : 0)
636
- - (launchGates.datasets ? 2 : 0);
669
+ // Count the actual registrations, including the conditional creation tools.
670
+ // This is read when the guide is called, after registration has completed.
671
+ // The coming-soon gate test also pins this claim to the wire count.
672
+ const liveToolCount = registeredTools.size;
637
673
  const comingSoonNote = `# Coming soon
638
674
 
639
675
  ${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "are" : "is"} not available in this build yet — the creation tools are not registered, so no tool call (and no workaround) can start one today. Do not retry or search for one.
@@ -690,9 +726,9 @@ get_inference_status instead of asserting them.
690
726
  4. search_models / list_models - Find a model from the hosted catalog
691
727
  5. get_model_config - Recommended settings for a chosen model
692
728
  6. list_supported_architectures - Which architectures can be served
693
- 7. get_gpu_pricing / get_recommended_gpu - GPU options and pricing
694
- 8. list_integrations - Connected Hugging Face accounts
695
- 9. list_datasets / preview_dataset / delete_dataset - Read and clean up existing datasets
729
+ 7. get_gpu_pricing / get_recommended_gpu / recommend_training_config - GPU options, pricing, and the trainer's own batch/memory/time recommendation
730
+ 8. list_integrations / search_huggingface_datasets / preview_huggingface_dataset - Discover Hub sources and inspect subsets/splits
731
+ 9. list_datasets / get_dataset_status / preview_dataset / delete_dataset - Read and clean up existing datasets
696
732
  10. get_training_capabilities - Read the authoritative training contract
697
733
  11. preflight_training_job - Validate a proposed job payload (no side effects)
698
734
  12. list_training_jobs / get_training_status - Monitor existing jobs
@@ -711,7 +747,26 @@ get_inference_status instead of asserting them.
711
747
  8. update_inference_policy - Change explicit queue consent or the maximum accepted hourly price
712
748
  9. stop_inference / resume_inference / restart_inference - Control serving lifecycle
713
749
  10. chat_with_inference - Call a serverless catalog model by id or a dedicated deployment via the unified /v1 endpoint (optional streaming)
714
- 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown`;
750
+ 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown
751
+
752
+ ## Conscious Loop MCP Tools
753
+ The loop turns what a deployed model actually did into what it learns next.
754
+ NOTHING IS RECORDED until loop_set_capture turns a source on — capture is a
755
+ consent decision, not a default, and these tools cannot bypass it.
756
+ 1. loop_set_capture - Turn recording on or off for a source and set how long conversations are kept. Start here; every other loop tool is inert until this is done.
757
+ 2. loop_capture_trace / loop_import_rows - Record one exchange as it happens (what the model was asked, what it answered, any tool calls), whoever served the model — Run BiOS, another provider, or your own servers. loop_import_rows brings in data that already exists instead: an export, a spreadsheet of past answers, preference pairs. Imported rows arrive as conversations NEEDING REVIEW, never as finished training data.
758
+ 3. loop_list_traces / loop_get_trace - Read what has been recorded. Credentials and personal details are stripped before storage, never after.
759
+ 4. loop_add_signal - Record whether an answer was right. Append-only. Verdict 'edited' with a correction is the most valuable feedback there is.
760
+ 5. loop_add_candidate / loop_list_candidates - Submit ALTERNATIVE answers to a recorded prompt. Sampling the model N times and scoring the samples produces DPO pairs without a human writing each one; a stronger model's answers do distillation the same way.
761
+ 6. loop_label_trace / loop_remove_label / loop_list_labels - Carve the corpus into slices. A bare tag with a parent makes a master group; NAMED DIMENSIONS (category, task, language, anything you like) are what let ONE captured corpus become a DIFFERENT dataset for every task, because filters on different dimensions AND together and nothing has to be re-labelled.
762
+ 7. loop_create_judge / loop_list_judges - A RUBRIC a model applies, for the question no rule can answer: was this actually helpful. The rubric doubles as a GRPO reward function, which makes prompts with no checkable answer trainable.
763
+ 8. loop_start_judge_run then loop_take_judge_work then loop_post_judge_verdicts - Run a rubric over a slice. YOU call the model: take the rendered prompt, send it, post the scores back.
764
+ 9. loop_create_grader / loop_list_graders - Deterministic rules that score answers with no person reading them, weighted so the criteria that matter count more, and aimable at one label so a rule about shipping does not mark down every billing answer.
765
+ 10. loop_grade_trace - Apply those rules to an answer AND its samples in one pass. This is what closes the loop automatically.
766
+ 11. loop_build_dataset - Curate the feedback into an sft, dpo, grpo or kto training set. Always reports why rows were left out. kto takes a bare thumbs-down, which the other three cannot use at all.
767
+ 12. loop_list_datasets - The sets built so far, each with how many rows it kept and why the rest were dropped.
768
+ 13. loop_preview_dataset - Read the actual rows before anyone trains on them; each carries the address of the conversation it came from.
769
+ 14. loop_stats - How much is recorded, how much is reviewed, and what each method could use right now.`;
715
770
  const guides = {
716
771
  overview: `# Run BiOS Fine-Tuning Platform
717
772
 
@@ -756,7 +811,7 @@ get_inference_status instead of asserting them.
756
811
  - JWT token: Use \`Authorization: Bearer <token>\` header with \`X-Org-ID\` and \`X-Workspace-ID\` headers
757
812
  - Call introspect_api_key to see your permissions, scopes, and which tools you can use
758
813
 
759
- ## Available MCP Tools (42 in this build, listed in recommended order)
814
+ ## Available MCP Tools (${liveToolCount} in this build, listed in recommended order)
760
815
  1. get_platform_guide - You're reading this! Context for all capabilities
761
816
  2. introspect_api_key - Discover your API key's permissions, scopes, and bound workspace
762
817
  3. get_wallet_balance - Check if you have enough credits before training
@@ -765,10 +820,11 @@ get_inference_status instead of asserting them.
765
820
  6. list_supported_architectures - Check which architectures can be trained or served
766
821
  7. get_training_capabilities - Read the authoritative training contract; call this before building automated payloads
767
822
  8. get_gpu_pricing - See GPU options and pricing
768
- 9. get_recommended_gpu - Get the best GPU for your model + adapter combo
769
- 10. list_integrations - See the Hugging Face accounts connected to your workspace
823
+ 9. get_recommended_gpu - Read default-config sizing; use full preflight for the actual run
824
+ 9a. recommend_training_config - Ask the trainer's sizing model for batch, accumulation, LR, predicted peak memory, minimum GPU count and wall clock on the GPU you picked; check your own batch size against it
825
+ 10. list_integrations / search_huggingface_datasets / preview_huggingface_dataset - Choose the Hub account, dataset, subset and split
770
826
  11. list_datasets / upload_dataset / import_huggingface_dataset - Prepare training data
771
- 12. preview_dataset - Verify dataset format before training
827
+ 12. get_dataset_status / preview_dataset - Verify readiness and dataset format before training
772
828
  13. delete_dataset - Remove a dataset you no longer need
773
829
  14. preflight_training_job - Validate the full job before creating it
774
830
  15. create_training_job - Launch the fine-tuning job
@@ -792,7 +848,26 @@ get_inference_status instead of asserting them.
792
848
  8. update_inference_policy - Change explicit queue consent or the maximum accepted hourly price
793
849
  9. stop_inference / resume_inference / restart_inference - Control serving lifecycle
794
850
  10. chat_with_inference - Call a serverless catalog model by id or a dedicated deployment via the unified /v1 endpoint (optional streaming)
795
- 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown`,
851
+ 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown
852
+
853
+ ## Conscious Loop MCP Tools
854
+ The loop turns what a deployed model actually did into what it learns next.
855
+ NOTHING IS RECORDED until loop_set_capture turns a source on — capture is a
856
+ consent decision, not a default, and these tools cannot bypass it.
857
+ 1. loop_set_capture - Turn recording on or off for a source and set how long conversations are kept. Start here; every other loop tool is inert until this is done.
858
+ 2. loop_capture_trace / loop_import_rows - Record one exchange as it happens (what the model was asked, what it answered, any tool calls), whoever served the model — Run BiOS, another provider, or your own servers. loop_import_rows brings in data that already exists instead: an export, a spreadsheet of past answers, preference pairs. Imported rows arrive as conversations NEEDING REVIEW, never as finished training data.
859
+ 3. loop_list_traces / loop_get_trace - Read what has been recorded. Credentials and personal details are stripped before storage, never after.
860
+ 4. loop_add_signal - Record whether an answer was right. Append-only. Verdict 'edited' with a correction is the most valuable feedback there is.
861
+ 5. loop_add_candidate / loop_list_candidates - Submit ALTERNATIVE answers to a recorded prompt. Sampling the model N times and scoring the samples produces DPO pairs without a human writing each one; a stronger model's answers do distillation the same way.
862
+ 6. loop_label_trace / loop_remove_label / loop_list_labels - Carve the corpus into slices. A bare tag with a parent makes a master group; NAMED DIMENSIONS (category, task, language, anything you like) are what let ONE captured corpus become a DIFFERENT dataset for every task, because filters on different dimensions AND together and nothing has to be re-labelled.
863
+ 7. loop_create_judge / loop_list_judges - A RUBRIC a model applies, for the question no rule can answer: was this actually helpful. The rubric doubles as a GRPO reward function, which makes prompts with no checkable answer trainable.
864
+ 8. loop_start_judge_run then loop_take_judge_work then loop_post_judge_verdicts - Run a rubric over a slice. YOU call the model: take the rendered prompt, send it, post the scores back.
865
+ 9. loop_create_grader / loop_list_graders - Deterministic rules that score answers with no person reading them, weighted so the criteria that matter count more, and aimable at one label so a rule about shipping does not mark down every billing answer.
866
+ 10. loop_grade_trace - Apply those rules to an answer AND its samples in one pass. This is what closes the loop automatically.
867
+ 11. loop_build_dataset - Curate the feedback into an sft, dpo, grpo or kto training set. Always reports why rows were left out. kto takes a bare thumbs-down, which the other three cannot use at all.
868
+ 12. loop_list_datasets - The sets built so far, each with how many rows it kept and why the rest were dropped.
869
+ 13. loop_preview_dataset - Read the actual rows before anyone trains on them; each carries the address of the conversation it came from.
870
+ 14. loop_stats - How much is recorded, how much is reviewed, and what each method could use right now.`,
796
871
  quick_start: `# Quick Start Guide
797
872
 
798
873
  ## Fastest path to a fine-tuned model:
@@ -808,12 +883,15 @@ For beginners: start with meta-llama/Llama-3.1-8B-Instruct (good balance of qual
808
883
  Option A: Upload your own file — call upload_dataset with the file path
809
884
  Option B: Import from HuggingFace — call import_huggingface_dataset with a repo ID
810
885
 
811
- Dataset must be JSONL: {"messages": [...]} per line for SFT, or {"text": "..."} per line for CPT. Other layouts are refused with conversion guidance.
886
+ Dataset upload accepts one file type: JSONL. Each line is {"messages": [...]} for SFT (with a final assistant response), or {"text": "..."} for CPT. A Hugging Face split may be stored upstream as Parquet or Arrow, but its ROWS must already have the same canonical shape. Other layouts are refused with conversion guidance before the dataset can become Ready.
812
887
 
813
- ### Step 4: Get recommended config
814
- Call get_model_config with your model name to see recommended GPU, adapter, and hyperparameters.
888
+ ### Step 4: Choose method, adapter, and the complete training configuration
889
+ Call get_model_config and get_training_capabilities. Choose SFT or CPT, adapter, sequence length, per-device batch, gradient accumulation, optimizer, precision, epochs/optional max_steps, validation and checkpoint settings BEFORE asking for a GPU. Empty max_steps means no cap: all configured epochs train. Early stopping defaults on when validation exists.
815
890
 
816
- ### Step 5: Create the training job
891
+ ### Step 5: Preflight the complete request and choose a GPU LAST
892
+ Call preflight_training_job with the model, dataset(s), method, adapter and complete config. Its GPU choices are sized from ALL of those inputs and joined to live inventory. Choose only a returned configuration; never guess a GPU count or use a choice below min_gpus.
893
+
894
+ ### Step 6: Create the training job
817
895
  Call create_training_job with:
818
896
  - model: the model name from step 2
819
897
  - dataset_id: the ID from step 3
@@ -821,7 +899,7 @@ Call create_training_job with:
821
899
  - adapter: "lora" (fastest, recommended for first jobs)
822
900
  - Leave other params as defaults unless you have specific needs
823
901
 
824
- ### Step 6: Monitor progress
902
+ ### Step 7: Monitor progress
825
903
  create_training_job books a GPU before returning (~40s). A 'booked' status means the GPU is secured (booked==secured) and training is starting; 'securing' means it is still booking — poll get_training_status until 'booked'/'running'. A booking-time capacity miss returns CAPACITY_UNAVAILABLE and no job is created.
826
904
  Call get_training_status periodically. Training typically takes 1-4 hours for small models.
827
905
  If a started job loses its pod it rests at 'interrupted' (billing already stopped, not 'failed') — call resume_training_job to continue from the last checkpoint.
@@ -891,11 +969,23 @@ Example JSONL:
891
969
  - ShareGPT conversations (from/value) -> role/content with human -> user, gpt -> assistant
892
970
  - Preference pairs (chosen/rejected) and prompt-only RL formats are not accepted for training.
893
971
 
972
+ ## Discover, preview and import
973
+ 1. Call list_integrations to choose a connected account, or use public Hub access.
974
+ 2. Call search_huggingface_datasets; use has_more and page for additional results.
975
+ 3. Call preview_huggingface_dataset. Choose from available_configs/available_splits and repeat the preview with that exact subset/split.
976
+ 4. Preserve integration_id, subset and split in import_huggingface_dataset. An explicit revision pins the import to that source; otherwise the service resolves and stores an immutable commit when importing.
977
+ 5. Poll get_dataset_status until ready. A returned dataset ID or temporarily unavailable preview does not mean validation succeeded. Preview the registered dataset before training, especially when importing an explicit older revision.
978
+
979
+ ## Composition and resume
980
+ - dataset_ids is an ordered array. Sequential mode keeps that order; shuffle randomizes it with a recorded seed (default 42).
981
+ - mixing supplies mode, seed, weights and optional phases. Weight positions match dataset_ids exactly. Use phased for an ordered curriculum and interleave for weighted draws.
982
+ - Preflight and create receive the same plan. Source pins, plan and checksum are stored; resume rebuilds data from metadata and verifies it rather than reusing a composed-data cache.
983
+
894
984
  ## Best Practices
895
- - Minimum 100 rows recommended (1,000+ for best results)
896
- - Maximum 500MB file size through this tool
897
- - Consistent formatting across all rows; remove duplicates and low-quality examples
898
- - Pick max_seq_length above your longest sample: samples longer than it are dropped and reported`,
985
+ - Keep a representative held-out validation set; quality matters more than a fixed row-count target.
986
+ - Maximum 500MB file size through the current MCP upload tool.
987
+ - Keep row structure consistent; remove unintended duplicates and low-quality examples.
988
+ - SFT defaults to truncating oldest context while retaining a trainable answer; unusable samples are dropped and reported. CPT packs raw text and has no per-sample truncate/drop option.`,
899
989
  training_methods: `# Training Methods
900
990
 
901
991
  ## SFT — Supervised Fine-Tuning ⭐ Most Common
@@ -909,49 +999,30 @@ Best for: Injecting domain knowledge into the base model before fine-tuning.
909
999
  Input: Raw text documents.
910
1000
  When to use: Before SFT, when your domain has specialized vocabulary/knowledge.
911
1001
 
912
- ## Recommended Pipeline
913
- 1. CPT (if domain-specific knowledge needed) → 2. SFT
914
- For most users: SFT alone is sufficient.`,
1002
+ ## Current limits
1003
+ - The current tools expose SFT and text-corpus CPT with full, LoRA and QLoRA; read get_training_capabilities for deployment-specific restrictions.
1004
+ - A VLM architecture does not imply support for image-only pretraining or multimodal preference objectives.
1005
+ - DPO, PPO and GRPO are not enabled by these training tools. They require separately verified training implementations; PPO/GRPO also require rollout inference and reward handling.
1006
+ - Training starts from catalog models. Do not promise a CPT-to-SFT pipeline using a customer's trained checkpoint as a new training source: that workflow is not exposed here.`,
915
1007
  gpu_selection: `# GPU Selection Guide
916
1008
 
917
- ## Available GPUs
918
-
919
- | GPU | VRAM | Best For | Per-Second Cost |
920
- |-----|------|----------|-----------------|
921
- | A6000 | 48 GB | Models up to 13B with LoRA | ~$0.0007/sec |
922
- | A100 80GB | 80 GB | Models up to 70B with QLoRA, up to 13B full | ~$0.0007/sec |
923
- | H100 | 80 GB | Large models, fastest training | ~$0.0011/sec |
924
-
925
- ## Model Size → GPU Recommendations
926
-
927
- ### LoRA Fine-Tuning (most common)
928
- - 1B-3B models: A6000 (48GB) — plenty of headroom
929
- - 7B-13B models: A100 80GB ⭐ sweet spot
930
- - 30B-70B models: A100 80GB or H100 (may need multi-GPU)
931
-
932
- ### QLoRA Fine-Tuning (4-bit quantized, uses less VRAM)
933
- - 7B-13B models: A6000 (48GB) — enough with quantization
934
- - 30B-70B models: A100 80GB
935
- - 70B+ models: H100 or multi-GPU A100
936
-
937
- ### Full Fine-Tuning (no adapter, maximum quality)
938
- - 1B-3B models: A100 80GB
939
- - 7B models: 2-4x A100 80GB
940
- - 13B+ models: 4-8x A100 or H100
941
-
942
- ## Cost Optimization Tips
943
- - Run BiOS bills per-SECOND, not per-hour — stop jobs early to save money
944
- - LoRA training is 5-10x cheaper than full fine-tune for comparable quality
945
- - QLoRA adds ~10% training time but halves VRAM needs
946
- - Start with fewer epochs (1-2) to validate, then train full if results look good
947
- - Use the get_recommended_gpu tool to get the optimal GPU for your specific setup
948
-
949
- ## Multi-GPU
950
- For models that don't fit in a single GPU's VRAM, the platform automatically shards across GPUs.
951
- Specify gpu_count in create_training_job (default: 1).
952
-
953
- ## Infrastructure and Price Caps
954
- The platform selects and manages the underlying infrastructure automatically. You choose only the GPU type, the GPU count, and a maximum total hourly price cap. Preflight returns ranked alternatives and the exact price that will be honored. A price above the approved cap is rejected rather than silently accepted.`,
1009
+ ## Configure first, choose the GPU last
1010
+ 1. Select the catalog model and revision, method, adapter and compatible datasets.
1011
+ 2. Set supported configuration fields from get_training_capabilities: sequence length, batch and accumulation, validation and run-length bounds. Review effective precision/optimizer settings; leave disabled engine-managed controls such as activation checkpointing to the engine.
1012
+ 3. The current preflight API requires a concrete gpu_type/gpu_count. Use get_recommended_gpu only to obtain a provisional candidate, then call preflight_training_job with that candidate and the complete configuration. Never treat default-config sizing as approval of the final run.
1013
+ 4. Review the service-calculated minimum, supported counts and current quote. Choose only a qualifying available type/count. Fit is not stock, and an unknown availability result is not permission to book.
1014
+ 5. Re-run preflight after changing any sizing input or GPU choice; create only after valid=true with the final selection. GPU-free preflight discovery is not currently supported by the deployed API.
1015
+
1016
+ ## Booking and recovery
1017
+ - Create rechecks capacity. Do not report a run as started while it is only securing; booked means capacity was accepted, not that the trainer has begun executing steps.
1018
+ - A capacity miss must show fresh qualifying alternatives and preserve the configuration. Ask before substituting hardware or exceeding a price cap.
1019
+ - Queueing is explicit consent and applies only when no qualifying alternative is available. Waiting is unbilled.
1020
+ - Reuse the idempotency key after timeouts instead of creating a second run.
1021
+
1022
+ ## Memory and quality
1023
+ - Memory depends on the full configuration and input shapes, not parameter count alone. Do not invent a smaller minimum or claim that quantization always halves total memory.
1024
+ - Full fine-tuning is not a guarantee of higher accuracy than adapters. Use representative validation to compare outcomes.
1025
+ - Preflight admission is not proof of universal freedom from hardware faults or OOM. Report actual tested configurations and measured headroom.`,
955
1026
  inference: `# Model Inference Guide
956
1027
 
957
1028
  Deployments serve either a verified fine-tuning checkpoint or a base model from the catalog through an OpenAI-compatible endpoint.
@@ -1399,6 +1470,50 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1399
1470
  return json(data);
1400
1471
  });
1401
1472
  /* ══════════════════════════════════════════════════════════════════════════ */
1473
+ /* TOOL: recommend_training_config */
1474
+ /* ══════════════════════════════════════════════════════════════════════════ */
1475
+ server.tool("recommend_training_config", "Ask the trainer's own sizing model what to run for a model on a GPU type: per-device batch size, gradient accumulation, learning rate, warmup, activation checkpointing, predicted peak GPU memory, the minimum GPU count that fits, a wall-clock estimate, and a verdict on the batch size you intend to use (with a same-global-batch alternative when it will not fit). Every number carries its basis (measured, derived, judgement). Side-effect free. When 'available' is false no advisor is deployed in this environment and no recommendation is returned -- do not treat that as 'anything fits'.", {
1476
+ model: z.string().describe("Full model name (e.g., 'Qwen/Qwen3.6-27B')"),
1477
+ gpu_type: z.string().describe("GPU type from get_recommended_gpu (e.g., 'B300_288GB')"),
1478
+ gpu_count: z.number().int().min(1).max(8).optional().describe("GPUs you intend to book (default 1)."),
1479
+ adapter: z.enum(TRAINING_ADAPTERS).optional().describe("Adapter type (default: 'lora')."),
1480
+ method: z.enum(["sft", "pt"]).optional().describe("Canonical training method (default: 'sft')."),
1481
+ max_length: z.number().int().min(128).max(262144).optional().describe("Training sequence length (default 2048)."),
1482
+ num_train_epochs: z.number().positive().optional().describe("Epochs, for the wall-clock estimate."),
1483
+ max_steps: z.number().int().positive().optional().describe("Step cap, if you set one."),
1484
+ per_device_train_batch_size: z.number().int().min(1).optional().describe("Your intended per-device batch; judged against the model."),
1485
+ gradient_accumulation_steps: z.number().int().min(1).optional(),
1486
+ dataset_ids: z.array(z.string()).max(20).optional().describe("Datasets the job will train on; their measured token statistics feed the sizing."),
1487
+ model_revision: z.string().optional(),
1488
+ }, async (params) => {
1489
+ const q = { model_id: params.model, gpu_type: params.gpu_type };
1490
+ if (params.gpu_count !== undefined)
1491
+ q.gpu_count = String(params.gpu_count);
1492
+ if (params.adapter)
1493
+ q.train_type = params.adapter;
1494
+ if (params.method)
1495
+ q.method = params.method;
1496
+ if (params.max_length !== undefined)
1497
+ q.max_length = String(params.max_length);
1498
+ if (params.num_train_epochs !== undefined)
1499
+ q.num_train_epochs = String(params.num_train_epochs);
1500
+ if (params.max_steps !== undefined)
1501
+ q.max_steps = String(params.max_steps);
1502
+ if (params.per_device_train_batch_size !== undefined)
1503
+ q.per_device_train_batch_size = String(params.per_device_train_batch_size);
1504
+ if (params.gradient_accumulation_steps !== undefined)
1505
+ q.gradient_accumulation_steps = String(params.gradient_accumulation_steps);
1506
+ if (params.dataset_ids?.length) {
1507
+ for (const id of params.dataset_ids)
1508
+ validateId(id, "dataset_ids");
1509
+ q.dataset_ids = params.dataset_ids.join(",");
1510
+ }
1511
+ if (params.model_revision)
1512
+ q.model_revision = params.model_revision;
1513
+ const data = await client.api("/api/training/recommend", { params: q });
1514
+ return json(data);
1515
+ });
1516
+ /* ══════════════════════════════════════════════════════════════════════════ */
1402
1517
  /* TOOL: get_inference_gpu_options */
1403
1518
  /* ══════════════════════════════════════════════════════════════════════════ */
1404
1519
  server.tool("get_inference_gpu_options", "Get inference model-fit GPU choices joined to the authoritative deployment inventory and pricing snapshot. PREFER model-addressed sizing: pass model=<model_id> (optionally revision / hf_integration_id) and the SERVER resolves the facts (vision-aware) and returns computed min_gpus, valid_counts, and bookable_counts — the exact minimums the create gate enforces. Returns valid tensor-parallel counts, exact total hourly prices, availability as available/out_of_stock/unknown, and ranked alternatives. Never interpret unknown inventory as available, never offer a count below min_gpus or outside bookable_counts, and never substitute a SKU without user approval.", {
@@ -1760,8 +1875,14 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1760
1875
  /* ══════════════════════════════════════════════════════════════════════════ */
1761
1876
  /* TOOL: list_datasets */
1762
1877
  /* ══════════════════════════════════════════════════════════════════════════ */
1763
- server.tool("list_datasets", "List all datasets in your workspace. Returns dataset ID, name, format (JSONL/Parquet/CSV), row count, file size, column names, and creation date for each dataset.", {}, async () => {
1764
- const data = await client.api("/api/datasets");
1878
+ server.tool("list_datasets", "Read one page of workspace datasets with metadata and readiness. Use offset/limit for more results; list membership does not mean the dataset is compatible with every training method.", {
1879
+ query: z.string().max(256).optional(),
1880
+ status: z.string().max(64).optional(),
1881
+ dataset_type: z.string().max(64).optional(),
1882
+ limit: z.number().int().min(1).max(200).optional(),
1883
+ offset: z.number().int().min(0).optional(),
1884
+ }, async ({ query, status, dataset_type, limit, offset }) => {
1885
+ const data = await client.api("/api/datasets", { params: { q: query, status, dataset_type, limit: String(limit ?? 50), offset: String(offset ?? 0) } });
1765
1886
  return json(data);
1766
1887
  });
1767
1888
  /* ══════════════════════════════════════════════════════════════════════════ */
@@ -1773,7 +1894,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1773
1894
  // server-side (403 DATASETS_COMING_SOON) as the authority. At launch the flag
1774
1895
  // opens and the registration returns verbatim.
1775
1896
  if (!launchGates.datasets) {
1776
- server.tool("upload_dataset", "Upload a dataset file for fine-tuning. Supports JSONL, Parquet, and CSV formats up to 500MB. The file is automatically validated: format detection, row counting, column identification, and schema validation. Returns the dataset ID needed for create_training_job.", {
1897
+ server.tool("upload_dataset", "Upload one canonical JSONL dataset (maximum 500MB): one JSON object per line, with a messages array and final assistant response for SFT, or one non-empty text field for CPT. JSON arrays, CSV, Parquet, TXT, Alpaca, ShareGPT, prompt/completion, question/answer and preference-pair layouts are refused with conversion guidance. The server validates the rows before the dataset becomes Ready or selectable for training. Returns the dataset ID needed for create_training_job.", {
1777
1898
  file_path: z.string().describe("Absolute path to the dataset file on disk"),
1778
1899
  name: z.string().optional().describe("Display name for the dataset (defaults to filename)"),
1779
1900
  }, async ({ file_path, name }) => {
@@ -1819,7 +1940,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1819
1940
  /* ══════════════════════════════════════════════════════════════════════════ */
1820
1941
  server.tool("preview_dataset", "Preview the first rows of a dataset to inspect its structure, column names, and content before training. Useful for verifying the dataset format is correct.", {
1821
1942
  dataset_id: z.string().describe("Dataset ID (e.g., 'ds_abc123')"),
1822
- rows: z.number().optional().describe("Number of rows to preview (default: 10, max: 50)"),
1943
+ rows: z.number().int().min(1).max(50).optional().describe("Number of rows to preview (default: 10, max: 50)"),
1823
1944
  }, async ({ dataset_id, rows }) => {
1824
1945
  validateId(dataset_id, "dataset_id");
1825
1946
  const data = await client.api(`/api/datasets/${dataset_id}/preview`, {
@@ -1838,8 +1959,41 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1838
1959
  return result(`Dataset ${dataset_id} deleted successfully.`);
1839
1960
  });
1840
1961
  /* ══════════════════════════════════════════════════════════════════════════ */
1841
- /* TOOL: list_integrations */
1962
+ /* TOOL: dataset discovery and status */
1842
1963
  /* ══════════════════════════════════════════════════════════════════════════ */
1964
+ server.tool("get_dataset_status", "Read dataset import/upload readiness, progress and validation failure. Do not select the dataset for training until status is ready; a dataset ID alone does not prove validation completed.", { dataset_id: z.string().min(1).max(128) }, async ({ dataset_id }) => {
1965
+ validateId(dataset_id, "dataset_id");
1966
+ return json(await client.api(`/api/datasets/${dataset_id}/status`));
1967
+ });
1968
+ server.tool("search_huggingface_datasets", "Search a bounded page of public Hub datasets, or browse with a selected connected account. Use list_integrations to choose the account; private/gated visibility depends on its permissions. Preserve integration_id for preview and import. Public search needs datasets:read; connected browsing needs integrations:read.", {
1969
+ query: z.string().max(256).optional(),
1970
+ integration_id: z.string().min(1).max(128).optional(),
1971
+ page: z.number().int().min(1).max(100).optional(),
1972
+ limit: z.number().int().min(1).max(50).optional(),
1973
+ sort: z.enum(["downloads", "likes"]).optional(),
1974
+ }, async ({ query, integration_id, page, limit, sort }) => {
1975
+ if (integration_id)
1976
+ validateId(integration_id, "integration_id");
1977
+ return json(await client.api(integration_id ? `/api/datasets/integrations/${integration_id}/browse` : "/api/datasets/hub-search", {
1978
+ params: {
1979
+ [integration_id ? "search" : "q"]: query,
1980
+ page: String(page ?? 1), limit: String(limit ?? 20), sort: sort ?? "downloads",
1981
+ },
1982
+ }));
1983
+ });
1984
+ server.tool("preview_huggingface_dataset", "Preview a Hub dataset before importing. Returns available configs/subsets and splits plus sample-format validation. If configuration selection is required, repeat with one returned subset and split. Use the same integration_id as search/import; authentication failures never fall back to public access. A preview outage is not proof of valid data.", {
1985
+ repo_id: z.string().min(1).max(256),
1986
+ integration_id: z.string().min(1).max(128).optional(),
1987
+ subset: z.string().max(256).optional(),
1988
+ split: z.string().min(1).max(128).optional(),
1989
+ rows: z.number().int().min(1).max(20).optional(),
1990
+ }, async ({ repo_id, integration_id, subset, split, rows }) => {
1991
+ if (integration_id)
1992
+ validateId(integration_id, "integration_id");
1993
+ return json(await client.api("/api/datasets/hub-preview", {
1994
+ params: { dataset_id: repo_id, integration_id, subset, split, limit: String(rows ?? 10), soft_errors: "true" },
1995
+ }));
1996
+ });
1843
1997
  server.tool("list_integrations", "List the integrations connected to your workspace, such as Hugging Face accounts. Returns each integration's ID, label, provider type, and status. Use an integration ID wherever a tool accepts integration_id: importing HuggingFace datasets. This is how you pick WHICH connected account an import should use when several are connected. This read is workspace-scoped: without workspace context it fails with WORKSPACE_CONTEXT_REQUIRED, which is a configuration fix (BIOS_WORKSPACE_ID or a workspace-bound API key), not something to retry.", {}, async () => {
1844
1998
  const data = await client.api("/api/datasets/integrations");
1845
1999
  return json(data);
@@ -1851,27 +2005,34 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1851
2005
  // gated, exactly like upload_dataset above; the service's 403
1852
2006
  // DATASETS_COMING_SOON stays the authority.
1853
2007
  if (!launchGates.datasets) {
1854
- server.tool("import_huggingface_dataset", "Import a dataset from HuggingFace Hub into your workspace. Works with public datasets and private datasets if your HuggingFace account is connected via Integrations. The dataset is downloaded, validated, and stored in your workspace. Returns the dataset ID for use with create_training_job.", {
2008
+ server.tool("import_huggingface_dataset", "Import the Hub dataset, subset and split selected through search/preview. Omit integration_id only for anonymous public access. The source is revision-pinned and either referenced or materialized in the background; poll get_dataset_status until ready before training. Canonical messages rows are required for SFT and text rows for CPT. This does not enable preference/RL training.", {
1855
2009
  repo_id: z
1856
- .string()
2010
+ .string().min(1).max(256)
1857
2011
  .describe("HuggingFace dataset repo whose rows already carry a messages array (SFT) or a text field (CPT)"),
1858
2012
  integration_id: z
1859
- .string()
1860
- .describe("HuggingFace integration ID from your connected integrations"),
1861
- split: z.string().optional().describe("Dataset split (default: 'train')"),
1862
- max_samples: z.number().optional().describe("Max rows to import (imports all if omitted)"),
1863
- name: z.string().optional().describe("Display name for the imported dataset"),
1864
- }, async ({ repo_id, integration_id, split, max_samples, name }) => {
1865
- const data = await client.api(`/api/datasets/integrations/${encodeURIComponent(integration_id)}/import`, {
1866
- method: "POST",
1867
- body: {
1868
- dataset_id: repo_id,
1869
- split: split ?? "train",
1870
- max_samples,
1871
- name: name ?? repo_id.split("/").pop(),
1872
- },
1873
- });
1874
- return json(data);
2013
+ .string().min(1).max(128).optional()
2014
+ .describe("Connected account ID; retain it from search/preview for private or gated datasets"),
2015
+ subset: z.string().max(256).optional().describe("Exact subset/config selected from the preview response"),
2016
+ split: z.string().min(1).max(128).optional().describe("Exact selected dataset split (default: 'train')"),
2017
+ revision: z.string().min(1).max(256).optional().describe("Requested source commit/tag/branch; the service records an immutable commit for training and resume"),
2018
+ max_samples: z.number().int().positive().optional().describe("Max rows to import (imports all if omitted)"),
2019
+ sample_strategy: z.enum(["first", "random"]).optional(),
2020
+ import_mode: z.enum(["auto", "reference", "materialize"]).optional(),
2021
+ name: z.string().max(255).optional().describe("Display name for the imported dataset"),
2022
+ }, async ({ repo_id, integration_id, subset, split, revision, max_samples, sample_strategy, import_mode, name }) => {
2023
+ const common = { name: name ?? repo_id.split("/").pop(), max_samples, sample_strategy, import_mode };
2024
+ if (integration_id) {
2025
+ validateId(integration_id, "integration_id");
2026
+ return json(await client.api(`/api/datasets/integrations/${integration_id}/import`, {
2027
+ method: "POST", body: { ...common, dataset_id: repo_id, subset, split: split ?? "train", revision },
2028
+ }));
2029
+ }
2030
+ const workspaceId = client.workspaceId || (await client.api("/api/api-keys/introspect"))?.workspace?.id;
2031
+ if (!workspaceId)
2032
+ throw new Error("Workspace context is required: configure RUNBIOS_WORKSPACE_ID or use a workspace-bound API key.");
2033
+ return json(await client.api("/api/datasets/register-hf", {
2034
+ method: "POST", body: { ...common, workspace_id: workspaceId, hf_dataset_id: repo_id, hf_subset: subset, hf_split: split ?? "train", hf_revision: revision },
2035
+ }));
1875
2036
  });
1876
2037
  }
1877
2038
  /* ══════════════════════════════════════════════════════════════════════════ */
@@ -1893,7 +2054,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1893
2054
  model: z.string().describe("Base model id from the Run BiOS catalog (e.g., 'meta-llama/Llama-3.1-8B-Instruct'). Must be hosted on Run BiOS — pick from list_models/search_models. Training always starts from a catalog base model; training FROM a fine-tuned model (adapter stacking) is not supported."),
1894
2055
  model_revision: z.string().max(256).optional().describe("Requested model branch, tag, or commit. Copy the exact 40-hex value returned in canonical_request into create."),
1895
2056
  dataset_id: z.string().optional(),
1896
- dataset_ids: z.array(z.string()).min(1).optional(),
2057
+ dataset_ids: z.array(z.string()).min(1).max(64).optional(),
1897
2058
  method: z.enum(TRAINING_METHODS),
1898
2059
  adapter: z.enum(TRAINING_ADAPTERS).optional(),
1899
2060
  gpu_type: z.string().optional(),
@@ -1915,8 +2076,9 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1915
2076
  num_checkpoints: z.number().int().min(1).max(100).optional(),
1916
2077
  integration_id: z.string().optional(),
1917
2078
  network_volume_id: z.string().optional(),
1918
- cache_dataset: z.boolean().optional(),
1919
2079
  dataset_mixing: z.enum(["shuffle", "sequential", "interleave"]).optional(),
2080
+ mixing: datasetMixingSchema.optional().describe("Composition plan. Dataset IDs define source order and weight positions; default seed is 42. The source pins, plan and checksum are retained for metadata-only resume."),
2081
+ ...trainingConfigurationSchema,
1920
2082
  config: z.record(z.string(), z.unknown()).optional(),
1921
2083
  }, async (params) => {
1922
2084
  await assertModelHosted(params.model, "preflight_training_job");
@@ -1933,7 +2095,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1933
2095
  // server-side (403 TRAINING_COMING_SOON before any billing or booking) as the
1934
2096
  // authority. At launch the flag opens and the registration returns verbatim.
1935
2097
  if (!launchGates.training) {
1936
- server.tool("create_training_job", "Create a fine-tuning job using the same contract as the web UI, following BOOK-BEFORE-REVEAL: the call blocks while the ranked GPU ladder is booked (~40s). A job id and the started email exist only once a real pod is secured, so status is 'booked' (booked==secured; training then provisions/downloads/runs on its own) or 'securing' (still booking at the deadline — poll get_training_status until booked/running). With explicit queue consent it may return 'queued'. A definitive booking-time miss returns the structured CAPACITY_UNAVAILABLE (409) with fresh available_gpus and NO job exists (nothing charged); never auto-substitute a GPU. Call preflight_training_job first, preserve the accepted ranked GPU choices, queue deadline, and maximum total hourly price, then reuse one idempotency key after timeouts.", {
2098
+ server.tool("create_training_job", "Create a training job using the same BOOK-BEFORE-REVEAL contract as the web UI. Call preflight_training_job first with the complete model/method/adapter/dataset/config request, then choose one returned GPU configuration: GPU is last because its minimum depends on every earlier choice. The call does not reveal a run as started until a real pod is secured; status is booked (secured) or securing (poll get_training_status). A booking-time miss returns CAPACITY_UNAVAILABLE with fresh qualifying available_gpus and no job/no charge: choose one and retry with the same setup. The rejected GPU type is never recommended back. Queue consent is valid only when available_gpus is empty; waiting is unbilled and expires automatically. Reuse one idempotency key after timeouts.", {
1937
2099
  model: z.string().describe("Base model id from the Run BiOS catalog (e.g., 'meta-llama/Llama-3.1-8B-Instruct'). Must be hosted on Run BiOS — pick from list_models/search_models. Training always starts from a catalog base model; training FROM a fine-tuned model (adapter stacking) is not supported."),
1938
2100
  model_revision: z.string().max(256).optional().describe("Exact model_revision from preflight canonical_request; a branch or tag is accepted but resolved again before any paid mutation."),
1939
2101
  idempotency_key: z
@@ -1944,10 +2106,10 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1944
2106
  .optional()
1945
2107
  .describe("Stable retry key. Reuse it after a timeout; omit to generate one for this tool call."),
1946
2108
  dataset_id: z.string().optional().describe("Compatibility field for one dataset ID."),
1947
- dataset_ids: z.array(z.string()).min(1).optional().describe("One or more dataset IDs to compose for training."),
2109
+ dataset_ids: z.array(z.string()).min(1).max(64).optional().describe("One or more dataset IDs to compose for training."),
1948
2110
  method: z.enum(TRAINING_METHODS).describe("Training method: sft, or pt/cpt for continued pre-training."),
1949
2111
  adapter: z.enum(TRAINING_ADAPTERS).optional().describe("Any adapter supported by the Run BiOS contract."),
1950
- gpu_type: z.string().optional().describe("GPU type (e.g., 'A100_80GB'). Use get_recommended_gpu to find the best option. Auto-selected if omitted."),
2112
+ gpu_type: z.string().optional().describe("GPU type returned by preflight_training_job after the complete training configuration is known. Do not guess or use a pre-configuration recommendation."),
1951
2113
  gpu_count: z.number().int().min(1).max(8).optional().describe("Number of GPUs for the primary choice."),
1952
2114
  gpu_priorities: z
1953
2115
  .array(z.object({
@@ -1964,29 +2126,11 @@ create (1-5 choices), and the queue itself stays opt-in.`,
1964
2126
  max_price_hour_cents: priceCapSchema(true),
1965
2127
  storage_gb: z.number().int().min(1).max(10000).optional(),
1966
2128
  num_checkpoints: z.number().int().min(1).max(100).optional(),
1967
- epochs: z.number().positive().max(1000).optional().describe("Number of training epochs."),
1968
- learning_rate: z.number().positive().max(1).optional().describe("Learning rate."),
1969
- batch_size: z.number().int().min(1).max(65536).optional().describe("Per-device train batch size."),
1970
- gradient_accumulation_steps: z.number().int().min(1).max(65536).optional(),
1971
- max_seq_length: trainingSeqLengthSchema(),
1972
- lora_rank: z.number().int().min(1).max(4096).optional(),
1973
- lora_alpha: z.number().positive().max(1_000_000).optional(),
1974
- warmup_ratio: z.number().min(0).max(1).optional(),
1975
- weight_decay: z.number().min(0).max(10).optional(),
1976
- scheduler: z
1977
- .enum(["linear", "cosine", "cosine_with_restarts", "polynomial", "constant", "constant_with_warmup"])
1978
- .optional()
1979
- .describe("Learning-rate scheduler."),
1980
- job_name: z.string().max(255).optional(),
1981
- eval_split_ratio: z.number().min(0).max(0.99).optional(),
1982
- early_stopping_patience: z.number().int().min(0).max(20).optional()
1983
- .describe("Stop after N evaluations without a better validation loss (0 = off; default 3 when eval_split_ratio is set). The best checkpoint is always saved."),
1984
- overlong_policy: z.enum(["truncate", "drop"]).optional()
1985
- .describe("Samples longer than max_seq_length: 'truncate' keeps the last tokens so the answer is trained (default); 'drop' skips them."),
1986
2129
  integration_id: z.string().optional(),
1987
2130
  network_volume_id: z.string().optional(),
1988
- cache_dataset: z.boolean().optional(),
1989
2131
  dataset_mixing: z.enum(["shuffle", "sequential", "interleave"]).optional(),
2132
+ mixing: datasetMixingSchema.optional().describe("Composition plan. Dataset IDs define source order and weight positions; default seed is 42. The source pins, plan and checksum are retained for metadata-only resume."),
2133
+ ...trainingConfigurationSchema,
1990
2134
  config: z.record(z.string(), z.unknown()).optional().describe("Additional Run BiOS training configuration fields."),
1991
2135
  }, async (params) => {
1992
2136
  await assertModelHosted(params.model, "create_training_job");
@@ -2028,7 +2172,7 @@ create (1-5 choices), and the queue itself stays opt-in.`,
2028
2172
  /* ══════════════════════════════════════════════════════════════════════════ */
2029
2173
  /* TOOL: get_training_metrics */
2030
2174
  /* ══════════════════════════════════════════════════════════════════════════ */
2031
- server.tool("get_training_metrics", "Get training metrics history for a job: training loss over time, eval loss, learning rate schedule, and evaluation results. Use this to check if training is progressing well (loss should decrease) or if the model is overfitting (eval loss increasing while train loss decreases).", {
2175
+ server.tool("get_training_metrics", "Get training metrics history for a job: training loss over time, eval loss, learning rate schedule, and evaluation results, plus resource_metrics (the latest worker snapshot: per-GPU memory and utilization, system RAM, CPU when the worker reports it, disk). Fields the worker did not report are null, never zero. Use this to check if training is progressing well (loss should decrease), if the model is overfitting (eval loss increasing while train loss decreases), or whether memory headroom allows a larger batch.", {
2032
2176
  job_id: z.string().describe("Training job ID"),
2033
2177
  }, async ({ job_id }) => {
2034
2178
  validateId(job_id, "job_id");
@@ -2234,6 +2378,303 @@ create (1-5 choices), and the queue itself stays opt-in.`,
2234
2378
  return json(data);
2235
2379
  });
2236
2380
  }
2381
+ /* ══════════════════════════════════════════════════════════════════════════ */
2382
+ /* TOOLS: Conscious Loop */
2383
+ /* */
2384
+ /* Capture what a model was asked and answered, record whether it was right, */
2385
+ /* and curate the result into training data. */
2386
+ /* */
2387
+ /* Every one of these is workspace-scoped and needs an API key carrying */
2388
+ /* loop:read or loop:write. Those scopes are absent from the Read Only */
2389
+ /* preset on purpose: what they return is raw prompts and completions, not */
2390
+ /* catalog metadata, so a key holds them only by an explicit grant. */
2391
+ /* */
2392
+ /* NOTHING IS RECORDED until loop_set_capture turns a source on. An agent */
2393
+ /* calling loop_capture_trace against a source nobody enabled gets */
2394
+ /* captured:false back, not an error and not a stored conversation. */
2395
+ /* ══════════════════════════════════════════════════════════════════════════ */
2396
+ server.tool("loop_set_capture", "Turn conversation capture on or off for one source, and choose how long recordings are kept. THIS IS THE CONSENT DECISION the rest of the Conscious Loop rests on: until it is made, nothing is recorded. The source is a deployment id for models served on Run BiOS, or any label you choose for traffic served elsewhere (another provider, your own servers, an agent framework). Recordings are deleted automatically once the retention window passes.", {
2397
+ source: z.string().describe("The source to record: a Run BiOS deployment id, or a label you choose for traffic served elsewhere (e.g. 'my-support-agent')"),
2398
+ enabled: z.boolean().describe("true starts recording new conversations from this source; false stops. Turning it off never deletes what is already stored."),
2399
+ retention_days: z.number().int().optional().describe("How many days recordings are kept before automatic deletion (1-3650, default 30)"),
2400
+ sample_rate: z.number().optional().describe("Record only a fraction of conversations, 0 to 1. Deterministic per conversation, so a multi-turn exchange is never split in half."),
2401
+ }, async ({ source, enabled, retention_days, sample_rate }) => {
2402
+ const data = await client.api(`/api/loop/configs/${encodeURIComponent(source)}`, {
2403
+ method: "PUT",
2404
+ body: { enabled, retention_days, sample_rate },
2405
+ });
2406
+ return json(data);
2407
+ });
2408
+ server.tool("loop_capture_trace", "Record one exchange — what the model was asked and what it answered — so it can be reviewed and later trained on. Works whoever served the model. Pass tool_calls and tools when the turn invoked a function: without them a tool-using exchange trains the model to reply in prose exactly where it should have called something. The returned trace_id is Run BiOS's own; if you pass request_id, replaying it returns the same trace instead of storing a duplicate. If the source has not been enabled with loop_set_capture this returns captured:false and stores nothing.", {
2409
+ source: z.string().describe("The source being recorded — must already be enabled with loop_set_capture"),
2410
+ model: z.string().describe("The model that produced the answer, e.g. 'gpt-4o' or a Run BiOS model id"),
2411
+ messages: z.array(z.object({
2412
+ role: z.string().describe("system | user | assistant | tool"),
2413
+ content: z.string().optional(),
2414
+ tool_call_id: z.string().optional().describe("On a tool result turn, the call it answers"),
2415
+ name: z.string().optional().describe("On a tool result turn, the tool's name"),
2416
+ })).describe("The conversation as it was sent to the model"),
2417
+ completion: z.string().optional().describe("The assistant's answer text. Omit when the answer was purely a tool call."),
2418
+ tool_calls: z.array(z.object({
2419
+ id: z.string().optional(),
2420
+ type: z.string().optional(),
2421
+ function: z.object({
2422
+ name: z.string(),
2423
+ arguments: z.string().describe("JSON-encoded string, as the chat-completions wire format uses"),
2424
+ }),
2425
+ })).optional().describe("What the answering turn invoked, if anything"),
2426
+ tools: z.any().optional().describe("The tool schema the model was offered. Training tool use without it teaches the model to invent function names."),
2427
+ conversation_id: z.string().optional().describe("Ties the turns of one conversation together"),
2428
+ request_id: z.string().optional().describe("Your own idempotency handle — replaying it will not create a second trace"),
2429
+ latency_ms: z.number().int().optional(),
2430
+ }, async ({ source, model, messages, completion, tool_calls, tools, conversation_id, request_id, latency_ms }) => {
2431
+ const data = await client.api("/api/loop/traces", {
2432
+ method: "POST",
2433
+ body: {
2434
+ deployment_id: source, model, messages,
2435
+ completion: completion ?? "",
2436
+ tool_calls, tools, conversation_id, request_id, latency_ms,
2437
+ },
2438
+ });
2439
+ return json(data);
2440
+ });
2441
+ server.tool("loop_import_rows", "Bring data the user ALREADY HAS into the loop — an export from another provider, a spreadsheet of past answers, a set of preference pairs. This does NOT create a training set: it creates conversations, in the same place captured ones live, subject to the same review, rules, judges and labels. A row becomes trainable when something says it is good, never because it arrived in a file. Each row is read for what it is: a prompt/chosen/rejected triple becomes a preference pair; an answer plus a yes-or-no becomes a thumbs verdict; a question and an answer waits for review; a question with no answer waits for an answer; a paragraph of prose is refused, because it is not a conversation. A verdict that arrives with the file is kept, but recorded as having come from the user's earlier process rather than from a reviewer here. Read the by_shape and refused_why counts back to the user: a file that was silently the wrong shape shows up there rather than merely importing fewer rows than expected.", {
2442
+ source: z.string().describe("Where this came from, e.g. 'zendesk-2026' or 'gpt4-history'. Required, and becomes the source every imported conversation is filed under — do not use a generic word like 'import', because that produces a corpus nobody can slice later."),
2443
+ rows: z.array(z.record(z.string(), z.any())).describe("The rows themselves, already parsed. At most 5000 per call: the call is synchronous. Pass each row in whatever shape it already has rather than reshaping it — the shape is how the server decides whether it carries a verdict."),
2444
+ model: z.string().optional().describe("Which model produced these answers, when the rows do not say per-row"),
2445
+ labels: z.array(z.string()).optional().describe("Applied to every row. Importing a dump is usually the moment somebody knows what all of it is, and labels are what make a slice of it selectable later."),
2446
+ attributes: z.record(z.string(), z.string()).optional().describe("Key/value facts applied to every row, e.g. {\"category\": \"billing\"}"),
2447
+ }, async ({ source, rows, model, labels, attributes }) => {
2448
+ const data = await client.api("/api/loop/import", {
2449
+ method: "POST",
2450
+ body: { source, rows, model, labels, attributes },
2451
+ });
2452
+ return json(data);
2453
+ });
2454
+ server.tool("loop_list_traces", "List recorded conversations for this workspace, newest first. Use signalled_only to see just the ones that already carry a verdict, which is what a training set can actually be built from.", {
2455
+ source: z.string().optional().describe("Only conversations from this source"),
2456
+ conversation_id: z.string().optional(),
2457
+ label: z.string().optional().describe("Only conversations carrying this bare tag. A tag with children matches them too."),
2458
+ attributes: z.record(z.string(), z.string()).optional().describe("Only conversations matching every one of these dimensions, e.g. {\"category\":\"billing\",\"language\":\"es\"}"),
2459
+ unlabelled: z.boolean().optional().describe("Only the conversations nobody has described yet — the pile worth looking at before deciding what the dimensions should be"),
2460
+ signalled_only: z.boolean().optional().describe("Only conversations that already have feedback on them"),
2461
+ limit: z.number().int().optional().describe("Default 50, maximum 200"),
2462
+ offset: z.number().int().optional(),
2463
+ }, async ({ source, conversation_id, label, attributes, unlabelled, signalled_only, limit, offset }) => {
2464
+ const attrs = Object.entries(attributes ?? {})
2465
+ .filter(([k, v]) => k && v)
2466
+ .map(([k, v]) => `${k}:${v}`);
2467
+ const data = await client.api("/api/loop/traces", {
2468
+ params: {
2469
+ deployment_id: source,
2470
+ conversation_id,
2471
+ label,
2472
+ attr: attrs.length ? attrs : undefined,
2473
+ unlabelled: unlabelled ? "true" : undefined,
2474
+ signalled: signalled_only ? "true" : undefined,
2475
+ limit: limit?.toString(),
2476
+ offset: offset?.toString(),
2477
+ },
2478
+ });
2479
+ return json(data);
2480
+ });
2481
+ server.tool("loop_get_trace", "Read one recorded conversation in full, with every verdict anyone has recorded on it. Anything that looked like a credential or a personal detail was removed before it was stored, and redaction_report says what.", { trace_id: z.string().describe("The trace id returned by loop_capture_trace or loop_list_traces") }, async ({ trace_id }) => {
2482
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}`);
2483
+ return json(data);
2484
+ });
2485
+ server.tool("loop_add_signal", "Record a verdict on a conversation: was the answer right? Feedback is APPEND-ONLY — a second verdict does not replace the first, because two reviewers disagreeing is information worth keeping.\n\nWHICH VERDICT TO USE, because they train different things:\n• accepted (a thumbs up) — the answer was good. It becomes an example to learn from.\n• rejected (a thumbs down) — it was wrong and you have nothing better. This REMOVES the answer from training rather than teaching anything; prefer edited whenever you know the better answer.\n• edited — the model was WRONG and here is what it should have said. The most valuable single action: it produces both halves of a preference pair, so the model learns your answer and learns to avoid its own.\n• gold — the REFERENCE answer for this question, recorded whatever the model happened to say. Unlike edited it asserts nothing about the model, so if it matches what was said no preference is invented — but it still trains SFT, and it stands in as the GRPO reference when no separate ground_truth was given. It is the most reusable feedback there is.\n• scored — a number, with ground_truth when the answer is checkable.", {
2486
+ trace_id: z.string(),
2487
+ verdict: z.enum(["accepted", "rejected", "edited", "scored", "gold"]).describe("accepted = thumbs up; rejected = thumbs down with nothing better to offer; edited = the model was wrong and you wrote the better answer; gold = the reference answer for this question regardless of what the model said; scored = a number or checkable fact"),
2488
+ source: z.enum(["human", "verifier", "judge", "behavioural"]).optional().describe("Who judged. A human outranks a verifier, a verifier outranks a judge, a judge outranks a behavioural hint. Default human."),
2489
+ correction: z.string().optional().describe("REQUIRED when verdict is 'edited' (the answer it should have given) or 'gold' (the reference answer for this question)"),
2490
+ score: z.number().optional().describe("REQUIRED when verdict is 'scored'"),
2491
+ ground_truth: z.string().optional().describe("The value or fact the answer can be checked against"),
2492
+ reason: z.string().optional().describe("Why, in your own words"),
2493
+ author: z.string().optional(),
2494
+ }, async ({ trace_id, verdict, source, correction, score, ground_truth, reason, author }) => {
2495
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/signals`, {
2496
+ method: "POST",
2497
+ body: { verdict, source: source ?? "human", correction, score, ground_truth, reason, author },
2498
+ });
2499
+ return json(data);
2500
+ });
2501
+ server.tool("loop_add_candidate", "Submit an ALTERNATIVE answer to a prompt that was already recorded, and score it. This is how preference training scales without a person writing every answer: sample the model several times for the same prompt, score each sample, and a DPO pair falls out automatically — the best-scoring sample becomes 'chosen' and the worst becomes 'rejected'. Point a stronger model at the prompt instead and the same mechanism does distillation. SCORE EVERY CANDIDATE: an unscored alternative cannot pair, because it says nothing about which answer is preferred. A human correction recorded with loop_add_signal always outranks any score you submit here.", {
2502
+ trace_id: z.string().describe("The recorded conversation this is an alternative answer to"),
2503
+ completion: z.string().optional().describe("The alternative answer's text"),
2504
+ tool_calls: z.array(z.object({
2505
+ id: z.string().optional(),
2506
+ type: z.string().optional(),
2507
+ function: z.object({ name: z.string(), arguments: z.string() }),
2508
+ })).optional().describe("When the alternative was itself a tool call"),
2509
+ model: z.string().optional().describe("Which model produced this alternative — the teacher, when distilling"),
2510
+ score: z.number().optional().describe("How good it is. Higher is better. Required for the candidate to take part in a pair."),
2511
+ score_source: z.enum(["human", "verifier", "judge", "behavioural"]).optional().describe("What produced the score. Required whenever a score is given."),
2512
+ reason: z.string().optional(),
2513
+ }, async ({ trace_id, completion, tool_calls, model, score, score_source, reason }) => {
2514
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/candidates`, {
2515
+ method: "POST",
2516
+ body: { completion: completion ?? "", tool_calls, model, score, score_source, reason },
2517
+ });
2518
+ return json(data);
2519
+ });
2520
+ server.tool("loop_list_candidates", "Read every alternative answer recorded for one prompt, with its score and what produced that score. Use this to see whether sampling has produced a spread worth training on — if every sample scored the same, the reward function cannot separate them and no pair will be built.", { trace_id: z.string() }, async ({ trace_id }) => {
2521
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/candidates`);
2522
+ return json(data);
2523
+ });
2524
+ server.tool("loop_label_trace", "Label a recorded conversation so it can be selected later. An average over everything is the least useful thing to train on — a model weak at refunds is fixed with refund examples, and you can only select those if the conversation was labelled when it arrived. A label may name a PARENT, which makes a master group: labelling something 'refunds' with parent 'billing' makes it selectable as either, and selecting 'billing' later gathers every child without anybody maintaining a list of them. Labels are lowercased and trimmed, so 'Refunds' and 'refunds' are one label.", {
2525
+ trace_id: z.string(),
2526
+ labels: z.array(z.string()).optional().describe("Bare tags, e.g. ['refunds','escalated']. Lowercase letters, digits, dot, dash or underscore."),
2527
+ attributes: z.record(z.string(), z.string()).optional().describe("NAMED DIMENSIONS, e.g. {\"category\":\"billing\",\"task\":\"refund-handling\",\"language\":\"es\"}. This is what lets one captured corpus become a different dataset for every task somebody trains for: filters on different dimensions AND together, and a dimension is matched EXACTLY within its key, so source=support never matches team=support. Setting one dimension leaves the others untouched, so correcting a mistake needs no delete first."),
2528
+ parent: z.string().optional().describe("The master group these belong to, e.g. 'billing'"),
2529
+ }, async ({ trace_id, labels, attributes, parent }) => {
2530
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/labels`, {
2531
+ method: "POST", body: { labels, attributes, parent },
2532
+ });
2533
+ return json(data);
2534
+ });
2535
+ server.tool("loop_remove_label", "Remove one label from a conversation. The counterpart to loop_label_trace, which until now had none — an agent could label something and then had no way to correct it. 'key' names the dimension; leave it out for a bare tag, which is what a plain name has always been.", {
2536
+ trace_id: z.string(),
2537
+ label: z.string().describe("The value to remove, e.g. 'refunds' or 'billing'"),
2538
+ key: z.string().optional().describe("The dimension it belongs to, e.g. 'category'. Omit for a bare tag."),
2539
+ }, async ({ trace_id, label, key }) => {
2540
+ const qs = key ? `?key=${encodeURIComponent(key)}` : "";
2541
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/labels/${encodeURIComponent(label)}${qs}`, { method: "DELETE" });
2542
+ return json(data);
2543
+ });
2544
+ server.tool("loop_list_labels", "Every label in this workspace with how many conversations carry it. Use it to see where the corpus actually is before choosing a slice to train on — and to discover the master groups, since a parent may have no conversations labelled with it directly.", {}, async () => json(await client.api("/api/loop/labels")));
2545
+ server.tool("loop_create_judge", "Write a RUBRIC a model applies to answers nobody has time to read. A grader is a rule — deterministic, cheap, and blind to anything it was not told to look for; a judge is instructions in your own words, and it is the only thing that can answer 'was this answer actually helpful'. Score several things SEPARATELY: 'helpful but wrong' is not expressible in one number, and an average that hides it is worse than no score. The rubric also becomes a REWARD FUNCTION for GRPO — a prompt with a rubric attached is trainable even though no string the answer has to equal exists, which is most real work.", {
2546
+ name: z.string().describe("What this judge is for, e.g. 'Billing tone'"),
2547
+ instructions: z.string().describe("What makes an answer good here, in your own words"),
2548
+ dimensions: z.array(z.object({
2549
+ key: z.string().describe("Lowercase name, e.g. 'accuracy'"),
2550
+ description: z.string().optional().describe("What that dimension means"),
2551
+ })).describe("What to score, separately. At most 12."),
2552
+ label: z.string().optional().describe("Only judge this slice. A parent label gathers its children."),
2553
+ deployment_id: z.string().optional().describe("Only judge conversations from this source"),
2554
+ sample: z.number().optional().describe("At most this many conversations per run"),
2555
+ only_unscored: z.boolean().optional().describe("Skip conversations a judge has already scored"),
2556
+ model: z.string().optional().describe("Which model you intend to judge with, recorded so two runs are comparable"),
2557
+ write_gold: z.boolean().optional().describe("Let the judge write the answer that SHOULD have been given. Off by default: training on a model-written reference is distillation, and that is a decision you make deliberately."),
2558
+ }, async ({ name, instructions, dimensions, label, deployment_id, sample, only_unscored, model, write_gold }) => {
2559
+ const data = await client.api("/api/loop/judges", {
2560
+ method: "POST",
2561
+ body: {
2562
+ name, instructions, dimensions, model, write_gold,
2563
+ selection: { label, deployment_id, sample, only_unscored },
2564
+ },
2565
+ });
2566
+ return json(data);
2567
+ });
2568
+ server.tool("loop_list_judges", "Every rubric this workspace has written, with what it scores and which slice it covers.", {}, async () => json(await client.api("/api/loop/judges")));
2569
+ server.tool("loop_start_judge_run", "Open a run of one judge: select the conversations NOW and freeze the rubric onto them. The selection is fixed at this moment on purpose — a run whose selection is a live query silently grows as conversations arrive, so 'we scored the refunds slice' becomes a claim about a set that no longer exists. Follow with loop_take_judge_work.", {
2570
+ judge_id: z.string(),
2571
+ limit: z.number().optional().describe("Cap this pass below the judge's own sample, for a cheap look first"),
2572
+ }, async ({ judge_id, limit }) => {
2573
+ const data = await client.api(`/api/loop/judges/${encodeURIComponent(judge_id)}/runs`, {
2574
+ method: "POST", body: limit ? { limit } : {},
2575
+ });
2576
+ return json(data);
2577
+ });
2578
+ server.tool("loop_take_judge_work", "Take the next conversations to score. Each item arrives with a 'prompt' ALREADY RENDERED — send that text to a model as-is and reply with the JSON it asks for. Do not assemble your own: two callers that build the prompt differently are not applying one rubric. YOU run the model here on purpose; the service that stores these conversations holds no credentials to any model, which is most of the reason it is safe to store them there. Nothing is marked taken, so asking again returns the same items.", { run_id: z.string(), limit: z.number().optional() }, async ({ run_id, limit }) => {
2579
+ const qs = limit ? `?limit=${limit}` : "";
2580
+ return json(await client.api(`/api/loop/runs/${encodeURIComponent(run_id)}/work${qs}`));
2581
+ });
2582
+ server.tool("loop_post_judge_verdicts", "Hand the scores back. Each verdict carries one score per dimension the rubric asked for, each between 0 and 1, plus a reason — a score nobody can argue with is one nobody can correct. Scoring something the rubric never asked for is REFUSED rather than averaged in, and reported in 'rejected'. If you could not score an item, send it with 'error' instead of dropping it: a run that quietly shrinks is one whose coverage nobody can state.", {
2583
+ run_id: z.string(),
2584
+ verdicts: z.array(z.object({
2585
+ trace_id: z.string(),
2586
+ scores: z.record(z.string(), z.number()).optional().describe("One score per dimension, 0 to 1"),
2587
+ overall: z.number().optional().describe("Your own combination. Left out, the plain mean is used."),
2588
+ reason: z.string().optional().describe("Why, in a sentence or two"),
2589
+ gold: z.string().optional().describe("The answer that should have been given. Stored only if the judge has write_gold."),
2590
+ error: z.string().optional().describe("Mark an item you could not score"),
2591
+ })),
2592
+ finish: z.boolean().optional().describe("Close the run even with items still pending"),
2593
+ }, async ({ run_id, verdicts, finish }) => {
2594
+ const data = await client.api(`/api/loop/runs/${encodeURIComponent(run_id)}/verdicts`, {
2595
+ method: "POST", body: { verdicts, finish: finish ?? false },
2596
+ });
2597
+ return json(data);
2598
+ });
2599
+ server.tool("loop_create_grader", "Write a rule that scores answers automatically, with no person reading them. Human review is the most trustworthy feedback and the least available; a grader is written once and applied to every answer afterwards. Every kind is DETERMINISTIC — no model call — which is why a verifier's score outranks a judge's when they disagree: it cannot be flattered and it cannot drift between runs. 'weight' is the multiplier: a criterion that matters twice as much gets twice the weight, and the combined score is the weighted mean over the rules that actually applied. CAUTION WHEN WRITING A SET: a rule made only of 'forbidden' phrases is satisfied VACUOUSLY by an answer that says nothing, so pair it with a required phrase or you are rewarding silence.", {
2600
+ name: z.string().describe("What this rule checks, in your words"),
2601
+ kind: z.enum(["exact_match", "contains", "regex", "json_valid", "numeric", "tool_called"]),
2602
+ expected: z.string().optional().describe("For exact_match and numeric: the value to compare against"),
2603
+ required: z.array(z.string()).optional().describe("For contains: every phrase that must appear"),
2604
+ forbidden: z.array(z.string()).optional().describe("For contains: phrases that must not appear"),
2605
+ pattern: z.string().optional().describe("For regex: the pattern the answer must match"),
2606
+ keys: z.array(z.string()).optional().describe("For json_valid: keys the parsed answer must carry"),
2607
+ tolerance: z.number().optional().describe("For numeric: how far from expected still counts"),
2608
+ function_name: z.string().optional().describe("For tool_called: the function that must have been invoked"),
2609
+ case_sensitive: z.boolean().optional().describe("Turn off the default case and whitespace normalisation"),
2610
+ label: z.string().optional().describe("Aim the rule at one slice. A master group covers its children. Outside that slice the rule is NOT APPLIED rather than failed, so it stays out of the score entirely. Without this, a rule like 'must name the order number' marks down every billing answer for not mentioning a shipment."),
2611
+ weight: z.number().optional().describe("The multiplier, greater than zero. Default 1."),
2612
+ source: z.string().optional().describe("Limit the rule to one capture source. Omit to apply it everywhere."),
2613
+ }, async (input) => {
2614
+ const data = await client.api("/api/loop/graders", {
2615
+ method: "POST",
2616
+ body: {
2617
+ name: input.name, kind: input.kind, weight: input.weight,
2618
+ deployment_id: input.source, label: input.label,
2619
+ config: {
2620
+ expected: input.expected, required: input.required, forbidden: input.forbidden,
2621
+ pattern: input.pattern, keys: input.keys, tolerance: input.tolerance,
2622
+ function: input.function_name, case_sensitive: input.case_sensitive,
2623
+ },
2624
+ },
2625
+ });
2626
+ return json(data);
2627
+ });
2628
+ server.tool("loop_list_graders", "Every scoring rule this workspace has written, with its weight and whether it is enabled.", {}, async () => json(await client.api("/api/loop/graders")));
2629
+ server.tool("loop_grade_trace", "Apply this workspace's rules to one recorded answer AND to every alternative sampled for it, in one pass. That pairing is the point: scoring only the original gives a verdict, while scoring the samples too gives a preference pair — so sampling the model and then calling this produces DPO training data with nobody reading anything. A run where NO rule could apply writes nothing at all: no verdict, no scores. Recording a zero that no rule produced would poison the training data with a judgement nobody reached, so you get applied:0 back instead. Read the per-rule 'detail' to see which criterion an answer failed; a single combined number tells you nothing about what to fix.", { trace_id: z.string() }, async ({ trace_id }) => {
2630
+ const data = await client.api(`/api/loop/traces/${encodeURIComponent(trace_id)}/grade`, {
2631
+ method: "POST", body: {},
2632
+ });
2633
+ return json(data);
2634
+ });
2635
+ server.tool("loop_build_dataset", "Turn the recorded feedback into a training set. The three methods need genuinely different feedback, so a set built for one CANNOT be reshaped into another: 'sft' uses answers marked right and answers that were rewritten; 'dpo' needs answers that were REWRITTEN, so a better and a worse version of the same reply exist; 'grpo' needs answers with a checkable ground truth OR a judge rubric; 'kto' needs only a yes or a no on an answer, INCLUDING a bare thumbs-down that trains nothing under the other three — which is why it usually has the most rows. The result always reports rejected_counts — why rows were left out — so a small set can be explained rather than guessed at. holdout_percent holds back the most RECENT work rather than a random slice, so an evaluation measures whether the model generalised instead of memorising the same week.", {
2636
+ name: z.string().describe("A name you will recognise later"),
2637
+ method: z.enum(["sft", "dpo", "grpo", "kto"]),
2638
+ attributes: z.record(z.string(), z.string()).optional().describe("Narrow to named dimensions, ANDed together: {\"category\":\"billing\",\"language\":\"es\"}. This is how ONE captured corpus becomes a DIFFERENT dataset for every task — nothing is re-labelled to do it."),
2639
+ unlabelled: z.boolean().optional().describe("Only the conversations nobody has described yet"),
2640
+ source: z.string().optional().describe("Only conversations from this source"),
2641
+ label: z.string().optional().describe("Only conversations carrying this label. A label with children selects them too, so 'billing' picks up 'refunds'."),
2642
+ from: z.string().optional().describe("RFC3339 timestamp, earliest conversation to consider"),
2643
+ to: z.string().optional().describe("RFC3339 timestamp, latest conversation to consider"),
2644
+ sample: z.number().int().optional().describe("Take this many at random from what the filters matched. Applied before the holdout split, so the held-back slice is still the newest of what was chosen."),
2645
+ holdout_percent: z.number().int().optional().describe("0-50. Holds back the newest slice for evaluation."),
2646
+ }, async ({ name, method, source, label, attributes, unlabelled, from, to, sample, holdout_percent }) => {
2647
+ const data = await client.api("/api/loop/datasets", {
2648
+ method: "POST",
2649
+ body: {
2650
+ name, method, deployment_id: source, label, attributes, unlabelled,
2651
+ from, to, sample, holdout_percent,
2652
+ },
2653
+ });
2654
+ return json(data);
2655
+ });
2656
+ server.tool("loop_list_datasets", "List the training sets built in this workspace, each with how many rows it kept, how many conversations were considered, and why the rest were left out.", {
2657
+ method: z.enum(["sft", "dpo", "grpo", "kto"]).optional(),
2658
+ limit: z.number().int().optional(),
2659
+ }, async ({ method, limit }) => {
2660
+ const data = await client.api("/api/loop/datasets", {
2661
+ params: { method, limit: limit?.toString() },
2662
+ });
2663
+ return json(data);
2664
+ });
2665
+ server.tool("loop_preview_dataset", "Read the actual training rows in a set before anyone spends money training on it. Each row carries the address of the conversation it was built from, so a suspicious row can be traced back to what produced it.", {
2666
+ dataset_id: z.string(),
2667
+ limit: z.number().int().optional().describe("Default 50, maximum 200"),
2668
+ }, async ({ dataset_id, limit }) => {
2669
+ const data = await client.api(`/api/loop/datasets/${encodeURIComponent(dataset_id)}/items`, {
2670
+ params: { limit: limit?.toString() },
2671
+ });
2672
+ return json(data);
2673
+ });
2674
+ server.tool("loop_stats", "How much has been recorded in this workspace, how much of it has been reviewed, and how many rows each training method could use right now. The 'ready' figures are upper bounds: repeated questions are folded into one row when a set is built, so the finished count can be lower.", {}, async () => {
2675
+ const data = await client.api("/api/loop/stats");
2676
+ return json(data);
2677
+ });
2237
2678
  return mcp;
2238
2679
  }
2239
2680
  //# sourceMappingURL=server.js.map