starforge-cli 0.1.6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. starforge_cli/__init__.py +3 -0
  2. starforge_cli/api_client.py +589 -0
  3. starforge_cli/auth.py +349 -0
  4. starforge_cli/catalog.py +124 -0
  5. starforge_cli/cli.py +74 -0
  6. starforge_cli/cli_ui.py +469 -0
  7. starforge_cli/client_device.py +104 -0
  8. starforge_cli/commands/__init__.py +1 -0
  9. starforge_cli/commands/admin.py +140 -0
  10. starforge_cli/commands/bench.py +94 -0
  11. starforge_cli/commands/common.py +178 -0
  12. starforge_cli/commands/dataset.py +150 -0
  13. starforge_cli/commands/exp.py +213 -0
  14. starforge_cli/commands/init.py +52 -0
  15. starforge_cli/commands/jobs.py +223 -0
  16. starforge_cli/commands/login.py +54 -0
  17. starforge_cli/commands/plugin.py +243 -0
  18. starforge_cli/commands/recipe.py +163 -0
  19. starforge_cli/commands/serve.py +79 -0
  20. starforge_cli/commands/submit.py +467 -0
  21. starforge_cli/commands/sweep.py +154 -0
  22. starforge_cli/config_resolve.py +17 -0
  23. starforge_cli/data_prep.py +60 -0
  24. starforge_cli/new_experiment.py +195 -0
  25. starforge_cli/packing.py +179 -0
  26. starforge_cli/plugins_lock.py +73 -0
  27. starforge_cli/project.py +130 -0
  28. starforge_cli/recipe_lock.py +453 -0
  29. starforge_cli/scaffold/agent-run.py.tmpl +146 -0
  30. starforge_cli/scaffold/custom-framework/train.sh +56 -0
  31. starforge_cli/scaffold/experiment-template/.gitkeep +0 -0
  32. starforge_cli/scaffold/experiment-template/README.md +36 -0
  33. starforge_cli/scaffold/experiment-template/config.yaml +44 -0
  34. starforge_cli/scaffold/project/common/README.md +12 -0
  35. starforge_cli/scaffold/project/common/__init__.py +0 -0
  36. starforge_cli/scaffold/project/configs/README.md +103 -0
  37. starforge_cli/scaffold/project/configs/base/README.md +24 -0
  38. starforge_cli/scaffold/project/configs/base/distillation_math.yaml +284 -0
  39. starforge_cli/scaffold/project/configs/base/grpo_lora.yaml +30 -0
  40. starforge_cli/scaffold/project/configs/base/grpo_math_1B.yaml +470 -0
  41. starforge_cli/scaffold/project/configs/base/grpo_megatron.yaml +43 -0
  42. starforge_cli/scaffold/project/configs/base/grpo_noncolocated.yaml +18 -0
  43. starforge_cli/scaffold/project/configs/base/grpo_sliding_puzzle.yaml +81 -0
  44. starforge_cli/scaffold/project/configs/base/ppo_math_1B.yaml +454 -0
  45. starforge_cli/scaffold/project/configs/base/rm.yaml +224 -0
  46. starforge_cli/scaffold/project/configs/base/sft.yaml +294 -0
  47. starforge_cli/scaffold/project/configs/models/README.md +16 -0
  48. starforge_cli/scaffold/project/configs/models/qwen3.5-4b.yaml +12 -0
  49. starforge_cli/scaffold/project/configs/models/qwen3.5-9b.yaml +10 -0
  50. starforge_cli/scaffold/project/gitignore +11 -0
  51. starforge_cli/spec_builder.py +372 -0
  52. starforge_cli-0.1.6.dist-info/METADATA +40 -0
  53. starforge_cli-0.1.6.dist-info/RECORD +55 -0
  54. starforge_cli-0.1.6.dist-info/WHEEL +4 -0
  55. starforge_cli-0.1.6.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,470 @@
1
+ # GRPO Algorithm Configuration
2
+ grpo:
3
+ num_prompts_per_step: 32
4
+ num_generations_per_prompt: 16
5
+ max_rollout_turns: 1 # for multi-turn rollouts. Math Environments just have 1 turn (answering the question)
6
+ max_num_epochs: 1
7
+ max_num_steps: 1000000
8
+ normalize_rewards: true
9
+ use_leave_one_out_baseline: true
10
+ val_period: 10
11
+ val_at_start: false
12
+ val_at_end: false
13
+ overlong_filtering: false
14
+ advantage_clip_low: null
15
+ advantage_clip_high: null
16
+ max_val_samples: 256
17
+ val_batch_size: 256
18
+ seed: 42
19
+ use_dynamic_sampling: false
20
+ dynamic_sampling_max_gen_batches: 10
21
+ batch_multiplier: 1
22
+ reward_shaping:
23
+ enabled: false
24
+ overlong_buffer_length: 128
25
+ overlong_buffer_penalty: 1
26
+ max_response_length: ${policy.max_total_sequence_length}
27
+ stop_properly_penalty_coef: null
28
+
29
+ # Advantage Estimator Configuration
30
+ # Options: "grpo" (default) or "reinforce_plus_plus"
31
+ adv_estimator:
32
+ name: "grpo" # Use "reinforce_plus_plus" for Reinforce++ estimator
33
+ normalize_rewards: ${grpo.normalize_rewards}
34
+ use_leave_one_out_baseline: ${grpo.use_leave_one_out_baseline}
35
+ minus_baseline: true # Reinforce++-baseline specific: subtract per-prompt mean baseline
36
+ reward_scaling:
37
+ enabled: false
38
+ source_min: 0.0
39
+ source_max: 1.0
40
+ target_min: 0.0
41
+ target_max: 1.0
42
+ seq_logprob_error_threshold: null
43
+ # Advantage value assigned to invalid tool call tokens (e.g. -5.0 to penalize). null disables.
44
+ invalid_tool_call_advantage: null
45
+ # Advantage value assigned to tokens with malformed <think>/</think> tags (e.g. -5.0). null disables.
46
+ malformed_thinking_advantage: null
47
+
48
+ async_grpo:
49
+ enabled: false # Set to true to enable async training mode
50
+ # Max age (in training steps) for trajectories used in training
51
+ max_trajectory_age_steps: 1
52
+ in_flight_weight_updates: false # Set to true to enable in-flight weight updates
53
+ recompute_kv_cache_after_weight_updates: false # Set to true to recompute kv cache after in-flight-weight-updates
54
+
55
+ loss_fn:
56
+ reference_policy_kl_penalty: 0.01
57
+ # Can be set to k1, k2, k3
58
+ # For more details, see http://joschu.net/blog/kl-approx.html
59
+ reference_policy_kl_type: "k3"
60
+ kl_input_clamp_value: 20.0
61
+ kl_output_clamp_value: 10.0
62
+ ratio_clip_min: 0.2
63
+ ratio_clip_max: 0.2
64
+ ratio_clip_c: null
65
+ # (default off) loss formulation improvements (docs/guides/grpo.md#loss)
66
+ use_on_policy_kl_approximation: false
67
+ # Async GRPO requires importance sampling correction enabled
68
+ # Set to true when async_grpo.enabled is true
69
+ use_importance_sampling_correction: false
70
+ # "tis" – clamp IS weights to [min, max], where min defaults to 0
71
+ # "icepop" – zero out tokens with IS weight outside [min, max]
72
+ # "seq-mask-tis" – zero out sequences by geometric-mean IS ratio, non-truncated token IS correction
73
+ truncated_importance_sampling_type: null
74
+ truncated_importance_sampling_ratio: null
75
+ # TIS lower bound; null means 0 for TIS. Set explicitly for ICE-POP / seq-mask-tis.
76
+ truncated_importance_sampling_ratio_min: null
77
+ sequence_level_importance_ratios: false
78
+ token_level_loss: true
79
+ force_on_policy_ratio: false # Set to true to force ratio=1.0 (requires train_global_batch_size == num_prompts_per_step * num_generations_per_prompt)
80
+ use_kl_in_reward: false # Reinforce++: add KL penalty to reward instead of loss
81
+ use_cispo: false # CISPO (https://arxiv.org/abs/2506.13585): clipped IS-weight policy optimization
82
+ # Disable PPO ratio (curr_logprobs - prev_logprobs).exp(); falls back to REINFORCE-style
83
+ # surrogate loss. Used by REINFORCE/RLOO. See https://arxiv.org/abs/2402.14740
84
+ disable_ppo_ratio: false
85
+ # VAPO positive-example NLL loss weight: L = L_PPO + mu * L_NLL(correct samples).
86
+ # 0.0 disables it. VAPO paper recommends 0.1.
87
+ positive_example_nll_weight: 0.0
88
+
89
+ checkpointing:
90
+ enabled: true
91
+ checkpoint_dir: "results/grpo"
92
+ metric_name: "val:accuracy" # one of "val:" or "train:" followed by the metric name
93
+ higher_is_better: true
94
+ keep_top_k: 3
95
+ save_period: 10
96
+ checkpoint_must_save_by: null
97
+ model_save_format: "safetensors"
98
+ save_consolidated: false
99
+ save_optimizer: true
100
+
101
+ policy:
102
+ model_name: "Qwen/Qwen2.5-1.5B"
103
+ tokenizer:
104
+ name: ${policy.model_name} ## specify if you'd like to use a tokenizer different from the model's default
105
+ chat_template_kwargs: null # can be used to pass kwargs to the chat template, e.g., enable_thinking=true
106
+ hf_config_overrides: {}
107
+ train_global_batch_size: 512
108
+ train_micro_batch_size: 4
109
+ generation_batch_size: 32 # Only used when generating using HF backend
110
+ logprob_batch_size: ${policy.train_micro_batch_size}
111
+ max_total_sequence_length: 512
112
+ precision: "bfloat16"
113
+ logprob_chunk_size: null
114
+ offload_optimizer_for_logprob: false # Only useful for non-colocated generation since colocated generation will always offload optimizer to cuda before refit
115
+
116
+ dtensor_cfg:
117
+ _v2: true
118
+ enabled: true
119
+ cpu_offload: False
120
+ sequence_parallel: false
121
+ activation_checkpointing: false
122
+ tensor_parallel_size: 1
123
+ context_parallel_size: 1
124
+ custom_parallel_plan: null
125
+
126
+ # Automodel kwargs passed to from_pretrained.
127
+ # Uncomment force_hf if the custom model's adapter doesn't support per-tensor
128
+ # weight conversion (e.g. Qwen2, Llama). Auto-detected if not set.
129
+ # See https://github.com/NVIDIA-NeMo/RL/issues/2072
130
+ automodel_kwargs: {}
131
+ # force_hf: true
132
+
133
+ # LoRA (Low-Rank Adaptation) Configuration
134
+ lora_cfg:
135
+ enabled: False # Set to True to enable LoRA fine-tuning
136
+ target_modules: [] # List of module names to apply LoRA (empty list with match_all_linear=true applies to all linear layers)
137
+ exclude_modules: [] # List of module names to exclude from LoRA
138
+ match_all_linear: true # If True, applies LoRA to all linear layers (overrides target_modules)
139
+ dim: 8 # LoRA rank (r): lower rank = fewer parameters but less capacity. Typical values: 4, 8, 16, 32, 64
140
+ alpha: 32 # LoRA scaling factor: effective learning rate multiplier = alpha/dim. Typical values: 16, 32, 64
141
+ dropout: 0.0 # Dropout probability applied to LoRA layers (0.0 = no dropout)
142
+ dropout_position: "post" # Where to apply dropout: "pre" (before LoRA) or "post" (after LoRA)
143
+ lora_A_init: "xavier" # Initialization method for LoRA A matrix: "xavier" or "uniform"
144
+ use_triton: true # Use Triton-optimized kernels for LoRA (faster but requires flash-attn). Disable when tensor_parallel_size > 1
145
+
146
+ megatron_cfg:
147
+ enabled: false
148
+ force_reconvert_from_hf: False # Set to True to force reconvert of the model from Hugging Face
149
+ empty_unused_memory_level: 1 # 1 is the minimum recommendation for RL since we almost always need to offload before beginning generation. Setting to 0 is faster, but you are more likely to run out of GPU memory.
150
+ activation_checkpointing: false
151
+ # recompute_granularity controls activation checkpointing depth.
152
+ # "full": recompute all activations (default, max memory savings).
153
+ # "selective": recompute only specific modules (see recompute_modules).
154
+ # Only takes effect when activation_checkpointing: true.
155
+ recompute_granularity: "full"
156
+ # Modules to selectively recompute when recompute_granularity="selective".
157
+ # MCore options: ["core_attn", "moe_act", "layernorm", "mla_up_proj", "mlp", "moe", "shared_experts"].
158
+ # Null defaults to ["core_attn"]. Full list and per-module constraints:
159
+ # https://github.com/NVIDIA/Megatron-LM/blob/d30c3ae5469fe3f6a64d4fd2e63b6e7f7844ea81/megatron/core/transformer/transformer_config.py#L483
160
+ # Use ["moe"] for MoE models to recompute expert activations only.
161
+ recompute_modules: null
162
+ tensor_model_parallel_size: 1
163
+ expert_tensor_parallel_size: 1
164
+ expert_model_parallel_size: 1
165
+ pipeline_model_parallel_size: 1
166
+ num_layers_in_first_pipeline_stage: null
167
+ num_layers_in_last_pipeline_stage: null
168
+ context_parallel_size: 1
169
+ pipeline_dtype: ${policy.precision}
170
+ sequence_parallel: false
171
+ freeze_moe_router: true
172
+ moe_router_dtype: "fp64"
173
+ moe_router_load_balancing_type: "none" # "seq_aux_loss" causes logprob error divergence for grpo
174
+ moe_router_bias_update_rate: 0.0 # by default, disable bias updates for grpo
175
+ moe_permute_fusion: true
176
+ # gives ~20% training perf speedup with sequence packing
177
+ apply_rope_fusion: True
178
+ # gives ~25% training perf speedup with sequence packing and apply_rope_fusion
179
+ bias_activation_fusion: True
180
+ defer_fp32_logits: False
181
+ moe_per_layer_logging: False
182
+ moe_enable_deepep: false
183
+ moe_token_dispatcher_type: "alltoall"
184
+ moe_shared_expert_overlap: false
185
+ # Multi-Token Prediction (MTP). mtp_num_layers=0 disables MTP.
186
+ mtp_num_layers: 0
187
+ # MTP loss weight added to the main next-token loss (0.0 disables the MTP loss contribution).
188
+ mtp_loss_scaling_factor: 0.0
189
+ # When true, repeat one MTP layer mtp_num_layers times instead of using distinct layers.
190
+ mtp_use_repeated_layer: false
191
+ # When true, detach MTP heads so MTP loss does not affect main-model gradients.
192
+ mtp_detach_heads: false
193
+ gradient_accumulation_fusion: false
194
+ use_fused_weighted_squared_relu: false
195
+
196
+ peft:
197
+ enabled: false
198
+ target_modules: []
199
+ exclude_modules: []
200
+ dim: 8
201
+ alpha: 32
202
+ dropout: 0.0
203
+ dropout_position: "post"
204
+ lora_A_init_method: "xavier"
205
+ lora_B_init_method: "zero"
206
+ a2a_experimental: false
207
+ lora_dtype: None
208
+
209
+ optimizer:
210
+ optimizer: "adam"
211
+ lr: 5.0e-6
212
+ min_lr: 5.0e-7
213
+ weight_decay: 0.01
214
+ bf16: true
215
+ fp16: false
216
+ params_dtype: "float32"
217
+
218
+ #adam
219
+ adam_beta1: 0.9
220
+ adam_beta2: 0.999
221
+ adam_eps: 1e-8
222
+
223
+ #sgd
224
+ sgd_momentum: 0.9
225
+
226
+ #distributed optimizer
227
+ use_distributed_optimizer: true
228
+ use_precision_aware_optimizer: true
229
+
230
+ clip_grad: ${policy.max_grad_norm}
231
+
232
+ # optimizer cpu offload
233
+ optimizer_cpu_offload: false
234
+ optimizer_offload_fraction: 0.0
235
+
236
+ scheduler:
237
+ start_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
238
+ end_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
239
+ weight_decay_incr_style: "constant"
240
+ lr_decay_style: "constant"
241
+ lr_decay_iters: 1000
242
+ lr_warmup_iters: 13
243
+ lr_warmup_init: 5.0e-7
244
+
245
+ distributed_data_parallel_config:
246
+ grad_reduce_in_fp32: false
247
+ overlap_grad_reduce: true
248
+ overlap_param_gather: true
249
+ use_custom_fsdp: false
250
+ data_parallel_sharding_strategy: "optim_grads_params"
251
+
252
+ fp8_cfg:
253
+ enabled: false
254
+ fp8: "e4m3"
255
+ fp8_recipe: "blockwise"
256
+ fp8_param: false
257
+
258
+ env_vars: null
259
+
260
+ draft:
261
+ enabled: false
262
+ model_name: null
263
+ loss_weight: 0.1
264
+ num_layers: null
265
+ aux_layer_indices: null
266
+
267
+ # See docs/design-docs/sequence-packing-and-dynamic-batching.md
268
+ # for more details on dynamic batching and sequence packing.
269
+ dynamic_batching:
270
+ enabled: False
271
+ train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
272
+ logprob_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.logprob_batch_size}}
273
+ sequence_length_round: 64
274
+
275
+ sequence_packing:
276
+ enabled: True
277
+ train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
278
+ logprob_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.logprob_batch_size}}
279
+ algorithm: "modified_first_fit_decreasing"
280
+ sequence_length_round: 64
281
+
282
+ # makes the training sequence length divisible by the tensor parallel size
283
+ # this is useful for sequence parallel training
284
+ make_sequence_length_divisible_by: ${policy.dtensor_cfg.tensor_parallel_size}
285
+ max_grad_norm: 1.0
286
+
287
+ optimizer:
288
+ name: "torch.optim.AdamW"
289
+ kwargs:
290
+ lr: 5.0e-6
291
+ weight_decay: 0.01
292
+ betas: [0.9, 0.999]
293
+ eps: 1e-8
294
+
295
+ scheduler:
296
+ - name: "torch.optim.lr_scheduler.LinearLR"
297
+ kwargs:
298
+ start_factor: 0.1
299
+ end_factor: 1.0
300
+ total_iters: 50
301
+ - name: "torch.optim.lr_scheduler.ConstantLR"
302
+ kwargs:
303
+ factor: 1.0
304
+ total_iters: 10000000000
305
+ - milestones: [50]
306
+
307
+ generation:
308
+ # Port range for vLLM HTTP servers, kept below the OS ephemeral range.
309
+ # See ray.sub for the full port layout.
310
+ port_range_low: 11001
311
+ port_range_high: 15000
312
+ backend: "vllm"
313
+ max_new_tokens: ${policy.max_total_sequence_length}
314
+ temperature: 1.0
315
+ top_p: 1.0
316
+ top_k: null
317
+ stop_token_ids: null
318
+ stop_strings: null
319
+ mcore_generation_config:
320
+ async_engine: false
321
+ max_model_len: ${policy.max_total_sequence_length} # Engine-side max sequence length.
322
+ cuda_graph_impl: "local" # CUDA-graph implementation. Options: "none", "local", "transformer_engine", "full_iteration".
323
+ inference_cuda_graph_scope: "block" # Inference CUDA-graph scope. Options: "none", "layer", "block".
324
+ buffer_size_gb: 10
325
+ num_cuda_graphs: 4 # Number of CUDA graphs to pre-compile for different batch sizes
326
+ block_size_tokens: 256 # Size of each KV cache block in tokens (affects memory granularity)
327
+ use_cuda_graphs_for_non_decode_steps: true # Enable CUDA graphs for prefill/context processing
328
+ enable_chunked_prefill: true # Split long prefills into chunks for better memory management
329
+ enable_prefix_caching: false # Reuse KV blocks across requests sharing a prompt prefix.
330
+ max_tokens: 16384 # Maximum number of tokens to use in a single step. Analogous to vllm's max_num_batched_tokens
331
+ kv_cache_management_mode: "persist" # KV cache lifecycle across suspend/resume. Options: "persist", "offload". To select "recompute", set grpo.async_grpo.recompute_kv_cache_after_weight_updates=true.
332
+ materialize_only_last_token_logits: true
333
+ num_speculative_tokens: 0
334
+ refit_backend: "nvshmem" # Copy-service backend for non-colocated megatron weight refit. Options: "gloo" or "nvshmem".
335
+ parsers: [] # OpenAI tool-call / response parser names exposed by the HTTP server (only used if expose_http_server=true).
336
+ expose_http_server: false # Start an OpenAI-compatible HTTP server alongside the persistent engine. Required for NeMo-Gym.
337
+ vllm_cfg:
338
+ async_engine: false
339
+ precision: ${policy.precision}
340
+ kv_cache_dtype: "auto"
341
+ tensor_parallel_size: 1
342
+ pipeline_parallel_size: 1
343
+ expert_parallel_size: 1 # When EP > 1, EP must be a multiple of TP since vLLM's EP = DP * TP
344
+ gpu_memory_utilization: 0.6
345
+ max_model_len: ${policy.max_total_sequence_length}
346
+ # when enforce_eager is False, it is optional to set ++policy.generation.vllm_kwargs.compilation_config.backend=eager for better accuracy,
347
+ # with the flag, vllm will use the custom CUDA kernels instead of the Triton kernels generated by torch.compile
348
+ # for more details, see convergence issue https://github.com/NVIDIA-NeMo/RL/issues/998
349
+ enforce_eager: False
350
+ use_tqdm: true # Set to false to suppress vLLM generation progress bars. Only applies when async_engine is false
351
+ use_deep_gemm: False
352
+ num_last_layers_in_bf16: 0
353
+ num_first_layers_in_bf16: 0
354
+ enable_vllm_metrics_logger: true # Set to true to enable vLLM internal metrics logger, turn off for better performance
355
+ vllm_metrics_logger_interval: 0.5 # Interval in seconds to collect vLLM logger metrics
356
+ vllm_kwargs: {}
357
+ colocated:
358
+ # true: generation shares training GPUs
359
+ # false: uses dedicated generation resources
360
+ enabled: true
361
+ # only relevant when enabled is false
362
+ resources:
363
+ gpus_per_node: null # Decides num gpus to be dedicated to generation when there is one node in the cluster i.e cluster.num_nodes == 1
364
+ num_nodes: null # Decides number of nodes to be dedicated to generation
365
+
366
+ data:
367
+ max_input_seq_length: ${policy.max_total_sequence_length} # upper bound, real truncation occurs at vllm.max_model_len
368
+ shuffle: true
369
+ num_workers: 1
370
+
371
+ # use multiple dataloader for train
372
+ # see https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/grpo.md#multiple-dataloaders for more details.
373
+ use_multiple_dataloader: false
374
+
375
+ # dataset
376
+ train:
377
+ dataset_name: OpenMathInstruct-2
378
+ split_validation_size: 0.05 # use 5% of the training data as validation data
379
+ seed: ${grpo.seed} # seed for train/validation split when split_validation_size > 0
380
+ validation: null
381
+ # default settings for all datasets
382
+ default:
383
+ prompt_file: "examples/prompts/cot.txt"
384
+ system_prompt_file: null
385
+ processor: "math_hf_data_processor"
386
+ env_name: "math"
387
+
388
+ # You can also use multiple datasets by using a list of datasets.
389
+ # See `examples/configs/grpo_multiple_datasets.yaml` for a full configuration example.
390
+
391
+ # You can use custom response datasets for training and validation. For example:
392
+ # train:
393
+ # # this dataset will override input_key and use the default values for other vars
394
+ # data_path: /path/to/local/train_dataset.jsonl
395
+ # input_key: question
396
+ # validation:
397
+ # # this dataset will use the default values for other vars except data_path
398
+ # data_path: /path/to/local/val_dataset.jsonl
399
+ # default:
400
+ # # will use below vars as default values if dataset doesn't specify it
401
+ # dataset_name: ResponseDataset
402
+ # input_key: input
403
+ # output_key: output
404
+ # prompt_file: null
405
+ # system_prompt_file: null
406
+ # processor: "math_hf_data_processor"
407
+ # env_name: math
408
+ # See https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/grpo.md#datasets for more details.
409
+
410
+ env:
411
+ math:
412
+ num_workers: 8
413
+ math_verify_impl: "hf_math_verify"
414
+
415
+ logger:
416
+ log_dir: "logs" # Base directory for all logs
417
+ num_val_samples_to_print: 0 # Number of validation samples to pretty print on terminal
418
+ wandb_enabled: false
419
+ tensorboard_enabled: false
420
+ mlflow_enabled: false # Disable MLflow logging
421
+ swanlab_enabled: false # Disable SwanLab logging
422
+ monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
423
+ wandb:
424
+ project: "grpo-dev"
425
+ name: "grpo-dev-logger"
426
+ swanlab:
427
+ project: "grpo-dev"
428
+ name: "grpo-dev-logger"
429
+ tensorboard: {}
430
+ mlflow:
431
+ experiment_name: "grpo-dev"
432
+ run_name: "grpo-dev-logger"
433
+ tracking_uri: "http://localhost:5000"
434
+ gpu_monitoring:
435
+ collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
436
+ flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)
437
+
438
+ cluster:
439
+ gpus_per_node: 1
440
+ num_nodes: 1
441
+ # Port range for the distributed master address (TCPStore / NCCL rendezvous)
442
+ # and per-worker available ports. Kept below the OS ephemeral range
443
+ # (32768-60999 on stock Linux). See ray.sub for the full port layout.
444
+ master_port_range_low: 25000
445
+ master_port_range_high: 28000
446
+ segment_size: null # Nodes per NVLink domain segment for topology-aware alignment; null to disable
447
+
448
+ # TransferQueue-mediated data plane for sync GRPO.
449
+ # Off by default — the legacy grpo_train trainer never engages this.
450
+ # Flip enabled=true and run grpo_train_sync to use TQ-mediated bulk
451
+ # transfer between rollout and train. See nemo_rl/data_plane/README.md.
452
+ data_plane:
453
+ enabled: false
454
+ impl: transfer_queue
455
+ backend: "simple" # TQ storage backend ('simple' or 'mooncake_cpu')
456
+ storage_capacity: 1000000 # max samples retained per partition
457
+ num_storage_units: 2 # storage shards
458
+ claim_meta_poll_interval_s: 0.5 # blocking-claim poll cadence
459
+ global_segment_size: 549755813888 # 512 GiB — used when backend == "mooncake_cpu"
460
+ local_buffer_size: 68719476736 # 64 GiB — used when backend == "mooncake_cpu"
461
+ # observability: # NotRequired
462
+ # enabled: false
463
+
464
+ # Multi-Teacher On-Policy Distillation (MOPD): distills from one or more teacher
465
+ # models into the policy via token-level teacher-minus-student logprob advantages,
466
+ # served by non-colocated teacher worker groups (OPD advantage estimator +
467
+ # nemo_gym). null = disabled (default). See
468
+ # examples/configs/recipes/llm/mopd-qwen3-1.7b-3n8g-megatron-pack.yaml for a full
469
+ # enabled example.
470
+ on_policy_distillation: null
@@ -0,0 +1,43 @@
1
+ # Megatron-Core 后端 + 低显存实测 GRPO 调优(partial 片段,本身不含 defaults)。
2
+ # 用法:在实验 config 的 defaults 列表里,把它加在 base / 模型片段「之后」,让它覆盖默认的 DTensor:
3
+ # defaults:
4
+ # - ../../configs/base/grpo_math_1B.yaml
5
+ # - ../../configs/models/qwen3.5-9b.yaml
6
+ # - ../../configs/base/grpo_megatron.yaml # ← 切 Megatron 后端 + 低显存调优
7
+ # 想切回 DTensor/FSDP:删掉这一行,并把 lr 写回 policy.optimizer.kwargs.lr。
8
+ #
9
+ # 下面分两块:
10
+ # (A) 切后端必需项——对照官方 examples/configs/grpo_math_*_megatron.yaml。
11
+ # (B) 低显存实测显存/性能调优——来自跑通的 grpo_math_{4B,9B}_megatron。
12
+ # 换硬件(如 H200)时,在该硬件的 cluster/<profile>/overrides.conf 里覆盖这些键即可
13
+ # (这些键此处已存在,CLI override 的 struct 模式不会报错)。
14
+ policy:
15
+ # (A) 后端切换
16
+ dtensor_cfg:
17
+ enabled: false # 关 DTensor / FSDP
18
+ optimizer: null # 去掉 FSDP torch 优化器,改用 policy.megatron_cfg.optimizer
19
+ scheduler: null # 去掉 FSDP 调度器,改用 policy.megatron_cfg.scheduler
20
+ # 训练序列长度对齐到 Megatron 的 TP(基底默认对齐 DTensor 的 TP)
21
+ make_sequence_length_divisible_by: ${policy.megatron_cfg.tensor_model_parallel_size}
22
+
23
+ # (B) 低显存:小 micro batch;关 sequence packing(实测两份配置都关)
24
+ train_micro_batch_size: 1
25
+ sequence_packing:
26
+ enabled: false
27
+
28
+ megatron_cfg:
29
+ enabled: true # 开 Megatron-Core
30
+ # converter_type 由 AutoBridge 按 HF 架构自动识别,无需按模型改(保留基底默认即可)
31
+ # (B) 低显存实测显存/性能:
32
+ activation_checkpointing: true # 重算换显存
33
+ apply_rope_fusion: false # 官方对该路径明确关闭
34
+ defer_fp32_logits: true # 延迟 fp32 logits,省显存
35
+ empty_unused_memory_level: 2 # 官方默认 1,显存紧张时升到 2
36
+ distributed_data_parallel_config:
37
+ overlap_grad_reduce: false # 跨节点关 overlap 更稳
38
+ overlap_param_gather: false
39
+
40
+ generation:
41
+ vllm_cfg:
42
+ enforce_eager: true # 关 CUDA graph:省显存、避免编译问题
43
+ gpu_memory_utilization: 0.4 # 与训练 colocated,按显存在 0.3~0.5 间调
@@ -0,0 +1,18 @@
1
+ # 非 colocated 生成 overlay:训练与 vLLM 生成各占独立 GPU/节点(实测 9B 用法)。
2
+ # 2 卡场景:1 卡专跑生成、1 卡训练 → 训练侧 TP=PP=1。
3
+ # 删此行即回 colocated(生成与训练共用 GPU)。
4
+ #
5
+ # 为什么显存紧张时 9B 选非 colocated:
6
+ # - colocated 单卡要在“训练态↔vLLM 生成态”间反复 offload/refit,模型一大(9B)
7
+ # 又紧张又易出问题;非 colocated 让两张卡角色隔离、各自常驻,更稳(代价是每步把权重
8
+ # 经网络同步给生成卡)。colocated 更省卡,适合小模型(如 4B 的 colocated+PP=2)。
9
+ #
10
+ # 注意:非 colocated 把 cluster 里一部分卡划给生成,剩下的才给训练。
11
+ # 在 2 卡集群上训练只剩 1 卡,所以无法再 PP=2(PP=2 需 2 张训练卡,要用就回 colocated)。
12
+ policy:
13
+ generation:
14
+ colocated:
15
+ enabled: false
16
+ resources:
17
+ gpus_per_node: 1 # 专给生成的卡数(每节点)
18
+ num_nodes: 1 # 专给生成的节点数
@@ -0,0 +1,81 @@
1
+ # GRPO Algorithm Configuration
2
+ defaults: "grpo_math_1B.yaml"
3
+
4
+ grpo:
5
+ num_prompts_per_step: 32
6
+ num_generations_per_prompt: 16
7
+ max_rollout_turns: 50 # Maximum turns allowed per rollout
8
+ max_num_steps: 10000
9
+ max_num_epochs: 1
10
+
11
+ checkpointing:
12
+ enabled: true
13
+ checkpoint_dir: "results/grpo-sliding-puzzle"
14
+ metric_name: "val:accuracy" # one of "val:" or "train:" followed by the metric name
15
+ higher_is_better: true
16
+ keep_top_k: 3
17
+ save_period: 10
18
+ checkpoint_must_save_by: null
19
+
20
+ policy:
21
+ model_name: "Qwen/Qwen2.5-1.5B-Instruct"
22
+ max_total_sequence_length: 1024
23
+
24
+ dtensor_cfg:
25
+ enabled: true
26
+ cpu_offload: true
27
+ activation_checkpointing: true
28
+ sequence_parallel: true
29
+
30
+ generation:
31
+ backend: "vllm"
32
+ max_new_tokens: ${policy.max_total_sequence_length}
33
+ temperature: 1.0
34
+ # Setting top_p/top_k to 0.999/10000 to strip out Qwen's special/illegal tokens
35
+ # https://github.com/NVIDIA-NeMo/RL/issues/237
36
+ top_p: 0.999
37
+ top_k: 10000
38
+ stop_token_ids: null
39
+ stop_strings: null
40
+ vllm_cfg:
41
+ async_engine: false
42
+ tensor_parallel_size: 1
43
+ pipeline_parallel_size: 1
44
+ expert_parallel_size: 1
45
+ gpu_memory_utilization: 0.6
46
+ max_model_len: ${policy.max_total_sequence_length}
47
+
48
+ data:
49
+ add_system_prompt: false
50
+ shuffle: false # disable dataloader shuffle, shuffle is handled within the dataset
51
+
52
+ env:
53
+ sliding_puzzle_game:
54
+ cfg:
55
+ game_config:
56
+ size: 5 # Size of the puzzle (e.g., 2 for 2x2, 3 for 3x3)
57
+ shuffle_moves: 10 # Number of random moves to shuffle the solved state
58
+ max_moves: 30 # Maximum moves allowed per episode
59
+
60
+ logger:
61
+ log_dir: "logs" # Base directory for all logs
62
+ num_val_samples_to_print: 0 # Number of validation samples to pretty print on terminal
63
+ wandb_enabled: false
64
+ tensorboard_enabled: false
65
+ mlflow_enabled: false
66
+ swanlab_enabled: false # Disable SwanLab logging
67
+ monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
68
+ wandb:
69
+ project: "grpo-dev"
70
+ name: "grpo-dev-sliding_puzzle"
71
+ swanlab:
72
+ project: "grpo-dev"
73
+ name: "grpo-dev-sliding_puzzle"
74
+ tensorboard: {}
75
+ mlflow:
76
+ experiment_name: "grpo-dev"
77
+ run_name: "grpo-dev-sliding_puzzle"
78
+ tracking_uri: "http://localhost:5000"
79
+ gpu_monitoring:
80
+ collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
81
+ flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)