PyPI - arbor-ai - Versions diffs - 0.1.12__py3-none-any.whl → 0.1.13__py3-none-any.whl - Mend

arbor-ai 0.1.12py3-none-any.whl → 0.1.13py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (11) hide show

arbor/server/api/models/schemas.py CHANGED Viewed

@@ -199,10 +199,16 @@ class GRPOConfigRequest(BaseModel):
     bf16: Optional[bool] = None
     scale_rewards: Optional[bool] = None
     max_grad_norm: Optional[float] = None
+    report_to: Optional[str] = None
+    log_completions: Optional[bool] = None
+    logging_steps: Optional[int] = None
+    mask_truncated_completions: Optional[bool] = None
+    # Arbor specific
+    max_context_length: Optional[int] = None
     lora: Optional[bool] = None
-    update_interval: Optional[int] = None
     # To name the run
     suffix: Optional[str] = None
+    generation_batch_size: Optional[int] = None
 class GRPOConfigResponse(BaseModel):
@@ -216,8 +222,23 @@ class GRPOTerminateRequest(BaseModel):
 class GRPOTerminateResponse(BaseModel):
     status: str
     current_model: str
+    checkpoints: Optional[dict[str, str]] = None
+    last_checkpoint: Optional[str] = None
 class GRPOStepResponse(BaseModel):
     status: str
     current_model: str
+    checkpoints: dict[str, str]
+    last_checkpoint: Optional[str] = None
+class GRPOCheckpointRequest(BaseModel):
+    checkpoint_name: str
+class GRPOCheckpointResponse(BaseModel):
+    status: str
+    current_model: str
+    checkpoints: dict[str, str]
+    last_checkpoint: str

arbor/server/api/routes/grpo.py CHANGED Viewed

@@ -4,6 +4,8 @@ import subprocess
 from fastapi import APIRouter, BackgroundTasks, Request
 from arbor.server.api.models.schemas import (
+    GRPOCheckpointRequest,
+    GRPOCheckpointResponse,
     GRPOConfigRequest,
     GRPOConfigResponse,
     GRPORequest,
@@ -31,17 +33,24 @@ def run_grpo_step(
     inference_manager = request.app.state.inference_manager
     grpo_manager = request.app.state.grpo_manager
-    current_model = grpo_manager.grpo_step(grpo_request, inference_manager)
+    step_data = grpo_manager.grpo_step(grpo_request, inference_manager)
-    return GRPOStepResponse(status="success", current_model=current_model)
+    return GRPOStepResponse(status="success", **step_data)
 @router.post("/update_model", response_model=GRPOStepResponse)
 def update_model(request: Request):
     grpo_manager = request.app.state.grpo_manager
     inference_manager = request.app.state.inference_manager
-    current_model = grpo_manager.update_model(request, inference_manager)
-    return GRPOStepResponse(status="success", current_model=current_model)
+    update_model_data = grpo_manager.update_model(request, inference_manager)
+    return GRPOStepResponse(status="success", **update_model_data)
+@router.post("/checkpoint", response_model=GRPOCheckpointResponse)
+def checkpoint(request: Request, grpo_checkpoint_request: GRPOCheckpointRequest):
+    grpo_manager = request.app.state.grpo_manager
+    checkpoint_data = grpo_manager.checkpoint(grpo_checkpoint_request)
+    return GRPOCheckpointResponse(status="success", **checkpoint_data)
 @router.post("/terminate", response_model=GRPOTerminateResponse)
@@ -50,5 +59,5 @@ def terminate_grpo(request: Request):
     grpo_manager = request.app.state.grpo_manager
     inference_manager = request.app.state.inference_manager
-    final_model = grpo_manager.terminate(inference_manager)
-    return GRPOTerminateResponse(status="success", current_model=final_model)
+    terminate_data = grpo_manager.terminate(inference_manager)
+    return GRPOTerminateResponse(status="success", **terminate_data)

arbor/server/services/grpo_manager.py CHANGED Viewed

@@ -13,7 +13,11 @@ from datetime import datetime
 from pathlib import Path
 from typing import Optional
-from arbor.server.api.models.schemas import GRPOConfigRequest, GRPORequest
+from arbor.server.api.models.schemas import (
+    GRPOCheckpointRequest,
+    GRPOConfigRequest,
+    GRPORequest,
+)
 from arbor.server.core.config import Settings
 from arbor.server.services.comms.comms import ArborServerCommsHandler
 from arbor.server.services.inference_manager import InferenceManager
@@ -28,7 +32,10 @@ class GRPOManager:
         self.server_comms_handler = None
         self.status_thread = None
         self.model_saved_and_reload_requested = False
+        self.saving_checkpoint = False
+        self.checkpoints = {}
+        self.last_checkpoint = None
         self.data_count = 0
         self.last_inference_update = 0
         # Set up signal handler
@@ -86,12 +93,17 @@ class GRPOManager:
             "bf16",
             "scale_rewards",
             "max_grad_norm",
+            "report_to",
+            "log_completions",
+            "logging_steps",
+            "generation_batch_size",
+            "mask_truncated_completions",
         ]
         trl_train_kwargs = {
             key: train_kwargs[key] for key in trl_keys if key in train_kwargs
         }
-        arbor_keys = ["update_interval", "lora"]
+        arbor_keys = ["max_context_length", "lora"]
         arbor_train_kwargs = {
             key: train_kwargs[key] for key in arbor_keys if key in train_kwargs
         }
@@ -119,6 +131,8 @@ class GRPOManager:
         # Start the training process with ZMQ ports
         my_env = os.environ.copy()
         my_env["CUDA_VISIBLE_DEVICES"] = self.settings.arbor_config.training.gpu_ids
+        # WandB can block the training process for login, so we silence it
+        my_env["WANDB_SILENT"] = "true"
         num_processes = self.settings.arbor_config.training.gpu_ids.count(",") + 1
@@ -209,6 +223,12 @@ class GRPOManager:
         # Launch the inference server
         print("Launching inference server...")
+        # launch_kwargs = {
+        #     k: v for k, v in arbor_train_kwargs.items() if k in ["max_context_length"]
+        # }
+        inference_manager.launch_kwargs["max_context_length"] = arbor_train_kwargs.get(
+            "max_context_length", None
+        )
         inference_manager.launch(self.current_model)
     def _handle_status_updates(self, inference_manager: InferenceManager):
@@ -228,6 +248,12 @@ class GRPOManager:
                         self.model_saved_and_reload_requested = False
                         self.current_model = status["output_dir"]
                         print("Model update complete")
+                elif status["status"] == "checkpoint_saved":
+                    print("Received checkpoint saved status")
+                    self.checkpoints[status["checkpoint_name"]] = status["output_dir"]
+                    self.last_checkpoint = status["checkpoint_name"]
+                    self.saving_checkpoint = False
+                    print("Checkpoint saved")
                 elif status["status"] == "error":
                     print(f"Training error: {status.get('error', 'Unknown error')}")
                 elif status["status"] == "terminated":
@@ -249,6 +275,10 @@ class GRPOManager:
             )
             time.sleep(5)
+        while self.saving_checkpoint:
+            print("Saving checkpoint, pausing GRPO steps until checkpoint is saved...")
+            time.sleep(5)
         try:
             # Send the batch to the training process
             self.server_comms_handler.send_data(request.batch)
@@ -256,12 +286,11 @@ class GRPOManager:
         except Exception as e:
             print(f"Failed to send batch to training process: {e}")
-        # We tell the script to save the model. The script will let us know when it's done via the status update handler
-        # Then we'll actually run the update_model function in the inference manager and finally update the last_inference_update variable
-        # if self._should_update_model():
-        #     self.server_comms_handler.send_command({"command": "save_model"})
-        return self.current_model
+        return {
+            "current_model": self.current_model,
+            "checkpoints": self.checkpoints,
+            "last_checkpoint": self.last_checkpoint,
+        }
     def update_model(self, request, inference_manager: InferenceManager):
         if inference_manager._session:
@@ -286,18 +315,41 @@ class GRPOManager:
                 "Waiting for model to be saved and reloaded... This usually takes 20-30 seconds"
             )
             time.sleep(5)
-        return self.current_model
+        return {
+            "current_model": self.current_model,
+            "checkpoints": self.checkpoints,
+            "last_checkpoint": self.last_checkpoint,
+        }
+    def checkpoint(self, request: GRPOCheckpointRequest):
+        self.saving_checkpoint = True
+        self.server_comms_handler.send_command(
+            {"command": "save_checkpoint", "checkpoint_name": request.checkpoint_name}
+        )
+        while self.saving_checkpoint:
+            print("Waiting for checkpoint to be saved...")
+            time.sleep(5)
+        return {
+            "current_model": self.current_model,
+            "checkpoints": self.checkpoints,
+            "last_checkpoint": self.last_checkpoint,
+        }
     def terminate(self, inference_manager: InferenceManager):
         """Clean up resources and save the final model."""
+        termination_data = {
+            "current_model": self.current_model,
+            "checkpoints": self.checkpoints,
+            "last_checkpoint": self.last_checkpoint,
+        }
         try:
             # Stop the inference server
             if inference_manager.process is not None:
                 inference_manager.kill()
             # Send termination command through REQ socket
-            # self.server_comms_handler.send_broadcast({"message": "terminate"})
-            self.training_process.terminate()
+            self.server_comms_handler.send_broadcast({"message": "terminate"})
+            # self.training_process.terminate()
             print("Waiting for training process to finish")
             # Wait for training process to finish
@@ -336,17 +388,13 @@ class GRPOManager:
                     )
                 output_dir = self.train_kwargs["output_dir"]
                 self.train_kwargs = None
-                return output_dir
             else:
                 print("Training terminated, no output directory specified")
                 self.train_kwargs = None
-                return None
+        return termination_data
     def _should_update_model(self):
-        # return (
-        #     self.data_count - self.last_inference_update
-        #     >= self.train_kwargs["update_interval"]
-        # )
         return self.model_saved_and_reload_requested

arbor/server/services/inference_manager.py CHANGED Viewed

@@ -1,6 +1,7 @@
 import asyncio
 import json
 import os
+import random
 import signal
 import socket
 import subprocess
@@ -47,7 +48,12 @@ class InferenceManager:
     def is_server_restarting(self):
         return self.restarting
-    def launch(self, model: str, launch_kwargs: Optional[Dict[str, Any]] = None):
+    def launch(
+        self,
+        model: str,
+        launch_kwargs: Optional[Dict[str, Any]] = None,
+        max_retries: int = 3,
+    ):
         if self.is_server_running():
             print("Server is already launched.")
             return
@@ -59,77 +65,112 @@ class InferenceManager:
             if model.startswith(prefix):
                 model = model[len(prefix) :]
-        print(f"Grabbing a free port to launch an SGLang server for model {model}")
-        port = get_free_port()
-        timeout = launch_kwargs.get("timeout", 1800)
-        my_env = os.environ.copy()
-        my_env["CUDA_VISIBLE_DEVICES"] = self.settings.arbor_config.inference.gpu_ids
-        n_gpus = self.settings.arbor_config.inference.gpu_ids.count(",") + 1
-        # command = f"vllm serve {model} --port {port} --gpu-memory-utilization 0.9 --tensor-parallel-size {n_gpus} --max_model_len 8192 --enable_prefix_caching"
-        command = f"python -m sglang_router.launch_server --model-path {model} --dp-size {n_gpus} --port {port} --host 0.0.0.0 --disable-radix-cache"
-        print(f"Running command: {command}")
-        # We will manually stream & capture logs.
-        process = subprocess.Popen(
-            command.replace("\\\n", " ").replace("\\", " ").split(),
-            text=True,
-            stdout=subprocess.PIPE,  # We'll read from pipe
-            stderr=subprocess.STDOUT,  # Merge stderr into stdout
-            env=my_env,
-        )
-        # A threading.Event to control printing after the server is ready.
-        # This will store *all* lines (both before and after readiness).
-        print(f"SGLang server process started with PID {process.pid}.")
-        stop_printing_event = threading.Event()
-        logs_buffer = []
-        def _tail_process(proc, buffer, stop_event):
-            while True:
-                line = proc.stdout.readline()
-                if not line and proc.poll() is not None:
-                    # Process ended and no new line
-                    break
-                if line:
-                    buffer.append(line)
-                    # Print only if stop_event is not set
-                    if not stop_event.is_set():
-                        print(f"[SGLang LOG] {line}", end="")
-        # Start a background thread to read from the process continuously
-        thread = threading.Thread(
-            target=_tail_process,
-            args=(process, logs_buffer, stop_printing_event),
-            daemon=True,
-        )
-        thread.start()
-        # Wait until the server is ready (or times out)
-        base_url = f"http://localhost:{port}"
-        try:
-            wait_for_server(base_url, timeout=timeout)
-        except TimeoutError:
-            # If the server doesn't come up, we might want to kill it:
-            process.kill()
-            raise
-        # Once server is ready, we tell the thread to stop printing further lines.
-        stop_printing_event.set()
-        # A convenience getter so the caller can see all logs so far (and future).
-        def get_logs() -> str:
-            # Join them all into a single string, or you might return a list
-            return "".join(logs_buffer)
-        # Let the user know server is up
-        print(f"Server ready on random port {port}!")
+        retries = 0
+        while retries < max_retries:
+            try:
+                print(
+                    f"Attempt {retries + 1} of {max_retries} to launch server for model {model}"
+                )
+                print(
+                    f"Grabbing a free port to launch an SGLang server for model {model}"
+                )
+                port = get_free_port()
+                timeout = launch_kwargs.get("timeout", 1800)
+                my_env = os.environ.copy()
+                my_env["CUDA_VISIBLE_DEVICES"] = (
+                    self.settings.arbor_config.inference.gpu_ids
+                )
+                n_gpus = self.settings.arbor_config.inference.gpu_ids.count(",") + 1
+                # command = f"vllm serve {model} --port {port} --gpu-memory-utilization 0.9 --tensor-parallel-size {n_gpus} --max_model_len 8192 --enable_prefix_caching"
+                command = f"python -m sglang_router.launch_server --model-path {model} --dp-size {n_gpus} --port {port} --host 0.0.0.0 --disable-radix-cache"
+                print(f"Running command: {command}")
+                if launch_kwargs.get("max_context_length"):
+                    command += (
+                        f" --context-length {launch_kwargs['max_context_length']}"
+                    )
-        self.launch_kwargs["api_base"] = f"http://localhost:{port}/v1"
-        self.launch_kwargs["api_key"] = "local"
-        self.get_logs = get_logs
-        self.process = process
-        self.thread = thread
-        self.current_model = model
+                # We will manually stream & capture logs.
+                process = subprocess.Popen(
+                    command.replace("\\\n", " ").replace("\\", " ").split(),
+                    text=True,
+                    stdout=subprocess.PIPE,  # We'll read from pipe
+                    stderr=subprocess.STDOUT,  # Merge stderr into stdout
+                    env=my_env,
+                )
+                # A threading.Event to control printing after the server is ready.
+                # This will store *all* lines (both before and after readiness).
+                print(f"SGLang server process started with PID {process.pid}.")
+                stop_printing_event = threading.Event()
+                logs_buffer = []
+                def _tail_process(proc, buffer, stop_event):
+                    while True:
+                        line = proc.stdout.readline()
+                        if not line and proc.poll() is not None:
+                            # Process ended and no new line
+                            break
+                        if line:
+                            buffer.append(line)
+                            # Print only if stop_event is not set
+                            if not stop_event.is_set():
+                                print(f"[SGLang LOG] {line}", end="")
+                # Start a background thread to read from the process continuously
+                thread = threading.Thread(
+                    target=_tail_process,
+                    args=(process, logs_buffer, stop_printing_event),
+                    daemon=True,
+                )
+                thread.start()
+                # Wait until the server is ready (or times out)
+                base_url = f"http://localhost:{port}"
+                try:
+                    wait_for_server(base_url, timeout=timeout)
+                except TimeoutError:
+                    # If the server doesn't come up, we might want to kill it:
+                    process.kill()
+                    raise
+                # Once server is ready, we tell the thread to stop printing further lines.
+                stop_printing_event.set()
+                # A convenience getter so the caller can see all logs so far (and future).
+                def get_logs() -> str:
+                    # Join them all into a single string, or you might return a list
+                    return "".join(logs_buffer)
+                # Let the user know server is up
+                print(f"Server ready on random port {port}!")
+                self.launch_kwargs["api_base"] = f"http://localhost:{port}/v1"
+                self.launch_kwargs["api_key"] = "local"
+                self.get_logs = get_logs
+                self.process = process
+                self.thread = thread
+                self.current_model = model
+                # If we get here, the launch was successful
+                return
+            except Exception as e:
+                retries += 1
+                print(
+                    f"Failed to launch server (attempt {retries} of {max_retries}): {str(e)}"
+                )
+                # Clean up any failed processes
+                if "process" in locals():
+                    try:
+                        process.kill()
+                    except:
+                        pass
+                if retries == max_retries:
+                    raise Exception(
+                        f"Failed to launch server after {max_retries} attempts"
+                    ) from e
+                # Wait a bit before retrying
+                time.sleep(min(2**retries, 30))  # Exponential backoff, max 30 seconds
     def kill(self):
         from sglang.utils import terminate_process
@@ -184,7 +225,7 @@ class InferenceManager:
         print(f"Running inference for model {model}")
         # Monkeypatch:
         if model != self.current_model:
-            print(f"MONKEYPATCH: Model changed from {model} to {self.current_model}")
+            print(f"Model changed from {model} to {self.current_model}")
             model = self.current_model
             request_json["model"] = model
@@ -214,6 +255,12 @@ class InferenceManager:
                 await self._session.close()
                 self._session = None
             return None
+        except json.decoder.JSONDecodeError:
+            print(f"JSON Decode Error during inference: {content}")
+            return {
+                "error": "JSON Decode Error",
+                "content": content if content else "Content is null",
+            }
         except Exception as e:
             print(f"Error during inference: {e}")
             raise
@@ -241,6 +288,7 @@ class InferenceManager:
         tik = time.time()
         self.kill()
         print("Just killed server")
+        time.sleep(5)
         # Check that output directory exists and was created successfully
         print(f"Checking that output directory {output_dir} exists")
         if not os.path.exists(output_dir):

arbor/server/services/scripts/grpo_training.py CHANGED Viewed

@@ -14,7 +14,7 @@ from typing import Any, List, Optional, Union
 import torch
 import zmq
 from accelerate import Accelerator
-from accelerate.utils import gather
+from accelerate.utils import broadcast_object_list, gather, gather_object
 from datasets import Dataset, IterableDataset, load_dataset
 from peft import AutoPeftModelForCausalLM, LoraConfig, PeftConfig  # type: ignore
 from torch.utils.data import Dataset
@@ -23,6 +23,7 @@ from transformers import (
     PreTrainedTokenizerBase,
     Trainer,
     TrainerCallback,
+    is_wandb_available,
 )
 from trl import GRPOConfig, GRPOTrainer
 from trl.data_utils import maybe_apply_chat_template
@@ -32,6 +33,9 @@ from arbor.server.services.comms.comms import (
     ArborServerCommsHandler,
 )
+if is_wandb_available():
+    import wandb
 last_step_time = None
 last_queue_pop_time = None
@@ -65,8 +69,9 @@ class ArborGRPOTrainer(GRPOTrainer):
         ] = (None, None),
         peft_config: Optional["PeftConfig"] = None,
         comms_handler: Optional[ArborScriptCommsHandler] = None,
-        update_interval: Optional[int] = 5,
         lora: Optional[bool] = False,
+        # We do nothing with max_context_length right now
+        max_context_length: Optional[int] = None,
         **kwargs,
     ):
@@ -85,12 +90,12 @@ class ArborGRPOTrainer(GRPOTrainer):
         self.peft_config = peft_config
         self.scale_rewards = scale_rewards
         self.comms_handler = comms_handler
-        self.update_interval = update_interval
     def _generate_and_score_completions(
         self, batch: List[dict[str, Any]]
     ) -> dict[str, Union[torch.Tensor, Any]]:
         device = self.accelerator.device
+        mode = "train" if self.model.training else "eval"
         # Process prompts and completions
         prompt_completion_texts = []
@@ -106,12 +111,12 @@ class ArborGRPOTrainer(GRPOTrainer):
             )
         # Tokenize prompts
-        prompt_texts = [
+        prompts_text = [
             prompt_completion_text["prompt"]
             for prompt_completion_text in prompt_completion_texts
         ]
         prompt_inputs = self.processing_class(
-            prompt_texts,
+            prompts_text,
             return_tensors="pt",
             padding=True,
             padding_side="left",
@@ -124,12 +129,12 @@ class ArborGRPOTrainer(GRPOTrainer):
         )
         # Tokenize completions
-        completion_texts = [
+        completions_text = [
             prompt_completion_text["completion"]
             for prompt_completion_text in prompt_completion_texts
         ]
         completion_ids = self.processing_class(
-            completion_texts,
+            completions_text,
             return_tensors="pt",
             padding=True,
             add_special_tokens=False,
@@ -156,6 +161,30 @@ class ArborGRPOTrainer(GRPOTrainer):
         #     self._move_model_to_vllm()
         #     self._last_loaded_step = self.state.global_step
+        prompt_ids = broadcast_object_list(prompt_ids)
+        prompt_mask = broadcast_object_list(prompt_mask)
+        completion_ids = broadcast_object_list(completion_ids)
+        completion_mask = broadcast_object_list(completion_mask)
+        process_slice = slice(
+            self.accelerator.process_index * len(batch),
+            (self.accelerator.process_index + 1) * len(batch),
+        )
+        prompt_ids = prompt_ids[process_slice]
+        prompt_mask = prompt_mask[process_slice]
+        completion_ids = completion_ids[process_slice]
+        completion_mask = completion_mask[process_slice]
+        is_eos = completion_ids == self.processing_class.eos_token_id
+        # If mask_truncated_completions is enabled, zero out truncated completions in completion_mask
+        if self.mask_truncated_completions:
+            truncated_completions = ~is_eos.any(dim=1)
+            completion_mask = (
+                completion_mask * (~truncated_completions).unsqueeze(1).int()
+            )
         prompt_completion_ids = torch.cat([prompt_ids, completion_ids], dim=1)
         attention_mask = torch.cat([prompt_mask, completion_mask], dim=1)  # (B, P+C)
@@ -164,34 +193,29 @@ class ArborGRPOTrainer(GRPOTrainer):
         )
         logits_to_keep = completion_ids.size(1)
+        batch_size = (
+            self.args.per_device_train_batch_size
+            if mode == "train"
+            else self.args.per_device_eval_batch_size
+        )
         with torch.no_grad():
             # When using num_iterations == 1, old_per_token_logps == per_token_logps, so we can skip it's
             # computation here, and use per_token_logps.detach() instead.
-            if self.num_iterations > 1:
+            if (
+                self.num_iterations > 1
+                or self.args.steps_per_generation
+                > self.args.gradient_accumulation_steps
+            ):
                 old_per_token_logps = self._get_per_token_logps(
-                    self.model, prompt_completion_ids, attention_mask, logits_to_keep
-                )
-            else:
-                old_per_token_logps = None
-            if self.beta == 0.0:
-                ref_per_token_logps = None
-            elif self.ref_model is not None:
-                ref_per_token_logps = self._get_per_token_logps(
-                    self.ref_model,
+                    self.model,
                     prompt_completion_ids,
                     attention_mask,
                     logits_to_keep,
+                    batch_size,
                 )
             else:
-                with self.accelerator.unwrap_model(self.model).disable_adapter():
-                    ref_per_token_logps = self._get_per_token_logps(
-                        self.model,
-                        prompt_completion_ids,
-                        attention_mask,
-                        logits_to_keep,
-                    )
+                old_per_token_logps = None
         rewards = torch.tensor(
             [example["reward"] for example in batch], dtype=torch.float32
@@ -219,7 +243,56 @@ class ArborGRPOTrainer(GRPOTrainer):
         )
         advantages = advantages[process_slice]
-        ## Logged Metrics Removed Here
+        # Log the metrics
+        if mode == "train":
+            self.state.num_input_tokens_seen += (
+                self.accelerator.gather_for_metrics(attention_mask.sum()).sum().item()
+            )
+        self._metrics[mode]["num_tokens"] = [self.state.num_input_tokens_seen]
+        # log completion lengths, mean, min, max
+        agg_completion_mask = self.accelerator.gather_for_metrics(
+            completion_mask.sum(1)
+        )
+        self._metrics[mode]["completions/mean_length"].append(
+            agg_completion_mask.float().mean().item()
+        )
+        self._metrics[mode]["completions/min_length"].append(
+            agg_completion_mask.float().min().item()
+        )
+        self._metrics[mode]["completions/max_length"].append(
+            agg_completion_mask.float().max().item()
+        )
+        # identify sequences that terminated with EOS and log their lengths
+        agg_terminated_with_eos = self.accelerator.gather_for_metrics(is_eos.any(dim=1))
+        term_completion_mask = agg_completion_mask[agg_terminated_with_eos]
+        clipped_completions_ratio = 1 - len(term_completion_mask) / len(
+            agg_completion_mask
+        )
+        self._metrics[mode]["completions/clipped_ratio"].append(
+            clipped_completions_ratio
+        )
+        if len(term_completion_mask) == 0:
+            # edge case where no completed sequences are found
+            term_completion_mask = torch.zeros(1, device=device)
+        self._metrics[mode]["completions/mean_terminated_length"].append(
+            term_completion_mask.float().mean().item()
+        )
+        self._metrics[mode]["completions/min_terminated_length"].append(
+            term_completion_mask.float().min().item()
+        )
+        self._metrics[mode]["completions/max_terminated_length"].append(
+            term_completion_mask.float().max().item()
+        )
+        # Calculate mean reward
+        self._metrics[mode]["reward"].append(mean_grouped_rewards.mean().item())
+        self._metrics[mode]["reward_std"].append(std_grouped_rewards.mean().item())
+        # Log prompt and completion texts
+        self._textual_logs["prompt"].extend(gather_object(prompts_text))
+        self._textual_logs["completion"].extend(gather_object(completions_text))
         return {
             "prompt_ids": prompt_ids,
@@ -227,7 +300,6 @@ class ArborGRPOTrainer(GRPOTrainer):
             "completion_ids": completion_ids,
             "completion_mask": completion_mask,
             "old_per_token_logps": old_per_token_logps,
-            "ref_per_token_logps": ref_per_token_logps,
             "advantages": advantages,
         }
@@ -326,15 +398,10 @@ class CommandMonitor:
                     print(
                         f"[Training Script] Instructed to save model at {self.trainer.args.output_dir}"
                     )
-                    # Wait until data queue is empty before saving
                     while (
                         time_since_last_step() <= 10
                         or get_time_since_last_queue_pop() <= 10
                     ):
-                        # print(
-                        # f"Waiting for data queue to empty...{self.comms_handler.get_data_queue_size()}"
-                        # )
                         print(f"Waiting for steps to finish")
                         print(
                             f"Time since last step: {time_since_last_step():.1f} (needs to be >= 10)"
@@ -342,15 +409,12 @@ class CommandMonitor:
                         print(
                             f"Time since last queue pop: {get_time_since_last_queue_pop():.1f} (needs to be >= 10)"
                         )
-                        time.sleep(5)  # Small delay to prevent busy waiting)
+                        time.sleep(5)
                     print("[Training Script] Saving model...")
                     if self.trainer.peft_config:
                         self.trainer.save_model(
                             output_dir=self.trainer.args.output_dir + "/adapter/"
                         )
                         _model_to_merge = AutoPeftModelForCausalLM.from_pretrained(
                             self.trainer.args.output_dir + "/adapter/",
                             config=self.trainer.peft_config,
@@ -373,6 +437,56 @@ class CommandMonitor:
                             "output_dir": self.trainer.args.output_dir,
                         }
                     )
+                elif command.get("command") == "save_checkpoint":
+                    print(
+                        f"[Training Script] Instructed to save checkpoint {command.get('checkpoint_name')}"
+                    )
+                    while (
+                        time_since_last_step() <= 10
+                        or get_time_since_last_queue_pop() <= 10
+                    ):
+                        print(f"Waiting for steps to finish")
+                        print(
+                            f"Time since last step: {time_since_last_step():.1f} (needs to be >= 10)"
+                        )
+                        print(
+                            f"Time since last queue pop: {get_time_since_last_queue_pop():.1f} (needs to be >= 10)"
+                        )
+                        time.sleep(5)
+                    if self.trainer.peft_config:
+                        self.trainer.save_model(
+                            output_dir=self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/adapter/"
+                        )
+                        _model_to_merge = AutoPeftModelForCausalLM.from_pretrained(
+                            self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/adapter/",
+                            config=self.trainer.peft_config,
+                        )
+                        merged_model = _model_to_merge.merge_and_unload()
+                        merged_model.save_pretrained(
+                            self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/",
+                            safe_serialization=True,
+                        )
+                        self.trainer.processing_class.save_pretrained(
+                            self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/"
+                        )
+                    else:
+                        self.trainer.save_model(
+                            output_dir=self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/"
+                        )
+                    self.comms_handler.send_status(
+                        {
+                            "status": "checkpoint_saved",
+                            "checkpoint_name": command.get("checkpoint_name"),
+                            "output_dir": self.trainer.args.output_dir
+                            + f"/checkpoints/{command.get('checkpoint_name')}/",
+                        }
+                    )
         except Exception as e:
             print(e)
             self.comms_handler.send_status({"status": "error", "error": str(e)})
@@ -385,13 +499,15 @@ class CommandMonitor:
             for broadcast in self.comms_handler.receive_broadcast():
                 print(f"!!!Received broadcast: {broadcast}")
                 if broadcast.get("message") == "terminate":
-                    self.trainer.control.should_training_stop = True
-                    self.comms_handler.send_status(
-                        {
-                            "status": "Received termination command",
-                            "process_id": self.trainer.accelerator.process_index,
-                        }
-                    )
+                    # self.trainer.control.should_training_stop = True
+                    # self.comms_handler.send_status(
+                    #     {
+                    #         "status": "Received termination command",
+                    #         "process_id": self.trainer.accelerator.process_index,
+                    #     }
+                    # )
+                    if self.trainer.accelerator.is_main_process:
+                        self.trainer.accelerator.end_training()
         except Exception as e:
             self.comms_handler.send_status({"status": "error", "error": str(e)})

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.4
 Name: arbor-ai
-Version: 0.1.12
+Version: 0.1.13
 Summary: A framework for fine-tuning and managing language models
 Author-email: Noah Ziems <nziems2@nd.edu>
 Project-URL: Homepage, https://github.com/Ziems/arbor
@@ -15,7 +15,7 @@ Requires-Dist: python-multipart
 Requires-Dist: pydantic-settings
 Requires-Dist: torch
 Requires-Dist: transformers
-Requires-Dist: trl
+Requires-Dist: trl==0.17.0
 Requires-Dist: peft
 Requires-Dist: ray>=2.9
 Requires-Dist: setuptools<77.0.0,>=76.0.0
@@ -23,6 +23,7 @@ Requires-Dist: pyzmq>=26.4.0
 Requires-Dist: pyyaml>=6.0.2
 Requires-Dist: sglang[all]>=0.4.5.post3
 Requires-Dist: sglang-router
+Requires-Dist: wandb
 Dynamic: license-file
 <p align="center">

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/RECORD RENAMED Viewed

@@ -5,10 +5,10 @@ arbor/client/api.py,sha256=86bgHuGM_AvI1Uhic_QaCnpF4VFqXie9ZzxmbTXUPpQ,19
 arbor/server/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
 arbor/server/main.py,sha256=tY4Vlaaj4oq1FTGYOkbFMGF0quLEeR-VBaKaXhQ5mEE,382
 arbor/server/api/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
-arbor/server/api/models/schemas.py,sha256=s_G8sSb05FjkKEqpKpLlqaEd8NysJddHibRHhcnrKIk,5594
+arbor/server/api/models/schemas.py,sha256=KCHav1nPFbQEynrcO-MObhRmoOrdFvfGuVogApynOCA,6210
 arbor/server/api/routes/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 arbor/server/api/routes/files.py,sha256=DQC_ogH5zlzhHZSAA4Cj5wzK07XBIBVs2Po91W9rcDY,1835
-arbor/server/api/routes/grpo.py,sha256=VuEvSOwwrHegn9qM-1nbHFmmUnnC_BMwnIHsfIdiJyI,1877
+arbor/server/api/routes/grpo.py,sha256=AbQ_BHgk-Om5U0qSt_FeJfyBJ0vItpfrnCNtJgD6p5k,2245
 arbor/server/api/routes/inference.py,sha256=Zy4ciN6vdRgu0-sFFnEeTZB-4XnLjEDH-atU7roIKSs,1668
 arbor/server/api/routes/jobs.py,sha256=BNdaSYUBJX6xSd6Pj6qx1DQJiZ5EKVxxbXDbEkfkCpw,3634
 arbor/server/core/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
@@ -17,18 +17,18 @@ arbor/server/core/logging.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,
 arbor/server/services/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 arbor/server/services/dependencies.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 arbor/server/services/file_manager.py,sha256=Z9z4A4EzvPauid_DBfpim401DDtuJy_TbX4twTWDJWI,12119
-arbor/server/services/grpo_manager.py,sha256=TAU2BMHgbCgiAvKNVd2Y8N20SR4qEms3lChA4Z0ZzyY,13777
-arbor/server/services/inference_manager.py,sha256=q4RVUqh1snGfW-AADkCqW8hC5x3WAZNe0jwXKOY5joU,10685
+arbor/server/services/grpo_manager.py,sha256=-_0xjENvIrOAtHACkFPMYox9YAeckHbpX2FkrmKrWuU,15448
+arbor/server/services/inference_manager.py,sha256=NcsUI-pgf3cRhU6P3xlPx0dxhvgYrfGZkEEGORcHcis,12833
 arbor/server/services/job_manager.py,sha256=m_d4UPwN_82f7t7K443DaFpFoyv7JZSZKml8tawt1Bk,2186
 arbor/server/services/training_manager.py,sha256=oQdhpfxdgp_lCTb_lxhvjupdLrcg6HL3TEbct_q9F6I,21065
 arbor/server/services/comms/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 arbor/server/services/comms/comms.py,sha256=3KN3mzwPvfW2_L5hq02JdAk6yOMyhY0_pBz-DDr5A3o,7694
-arbor/server/services/scripts/grpo_training.py,sha256=Q9jwnbRdXAv_jVgrChLX6IiB3BLZU1F3BP6mBV0DVik,20889
+arbor/server/services/scripts/grpo_training.py,sha256=eMT5cIMolAzhukANH1WRmPdxIkvLbsbrggdGFCMGMHc,26474
 arbor/server/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 arbor/server/utils/helpers.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
-arbor_ai-0.1.12.dist-info/licenses/LICENSE,sha256=5vFGrbOFeXXM83JV9o16w7ohH4WLeu3-57GocJSz8ow,1067
-arbor_ai-0.1.12.dist-info/METADATA,sha256=upqnB_F9JDLytHm4AFrDnvPaOHdj8XiBCdrlam0rgRc,2413
-arbor_ai-0.1.12.dist-info/WHEEL,sha256=DnLRTWE75wApRYVsjgc6wsVswC54sMSJhAEd4xhDpBk,91
-arbor_ai-0.1.12.dist-info/entry_points.txt,sha256=PGBX-MfNwfIl8UPFgsX3gjtXLqSogRhOktKMpZUysD0,40
-arbor_ai-0.1.12.dist-info/top_level.txt,sha256=jzWdp3BRYqvZDMFsPajrcftvvlluzVDErkD8IMRfhYs,6
-arbor_ai-0.1.12.dist-info/RECORD,,
+arbor_ai-0.1.13.dist-info/licenses/LICENSE,sha256=5vFGrbOFeXXM83JV9o16w7ohH4WLeu3-57GocJSz8ow,1067
+arbor_ai-0.1.13.dist-info/METADATA,sha256=c0yScMpCiWYSFqVLjgk5TrRBuAVJK3aTBl0z0IPZ_8Y,2442
+arbor_ai-0.1.13.dist-info/WHEEL,sha256=QZxptf4Y1BKFRCEDxD4h2V0mBFQOVFLFEpvxHmIs52A,91
+arbor_ai-0.1.13.dist-info/entry_points.txt,sha256=PGBX-MfNwfIl8UPFgsX3gjtXLqSogRhOktKMpZUysD0,40
+arbor_ai-0.1.13.dist-info/top_level.txt,sha256=jzWdp3BRYqvZDMFsPajrcftvvlluzVDErkD8IMRfhYs,6
+arbor_ai-0.1.13.dist-info/RECORD,,

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/WHEEL RENAMED Viewed

@@ -1,5 +1,5 @@
 Wheel-Version: 1.0
-Generator: setuptools (80.4.0)
+Generator: setuptools (80.6.0)
 Root-Is-Purelib: true
 Tag: py3-none-any

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/entry_points.txt RENAMED Viewed

File without changes

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/licenses/LICENSE RENAMED Viewed

File without changes

{arbor_ai-0.1.12.dist-info → arbor_ai-0.1.13.dist-info}/top_level.txt RENAMED Viewed

File without changes

arbor-ai 0.1.12__py3-none-any.whl → 0.1.13__py3-none-any.whl

arbor-ai 0.1.12py3-none-any.whl → 0.1.13py3-none-any.whl