starforge-cli 0.1.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- starforge_cli/__init__.py +3 -0
- starforge_cli/api_client.py +589 -0
- starforge_cli/auth.py +349 -0
- starforge_cli/catalog.py +124 -0
- starforge_cli/cli.py +74 -0
- starforge_cli/cli_ui.py +469 -0
- starforge_cli/client_device.py +104 -0
- starforge_cli/commands/__init__.py +1 -0
- starforge_cli/commands/admin.py +140 -0
- starforge_cli/commands/bench.py +94 -0
- starforge_cli/commands/common.py +178 -0
- starforge_cli/commands/dataset.py +150 -0
- starforge_cli/commands/exp.py +213 -0
- starforge_cli/commands/init.py +52 -0
- starforge_cli/commands/jobs.py +223 -0
- starforge_cli/commands/login.py +54 -0
- starforge_cli/commands/plugin.py +243 -0
- starforge_cli/commands/recipe.py +163 -0
- starforge_cli/commands/serve.py +79 -0
- starforge_cli/commands/submit.py +467 -0
- starforge_cli/commands/sweep.py +154 -0
- starforge_cli/config_resolve.py +17 -0
- starforge_cli/data_prep.py +60 -0
- starforge_cli/new_experiment.py +195 -0
- starforge_cli/packing.py +179 -0
- starforge_cli/plugins_lock.py +73 -0
- starforge_cli/project.py +130 -0
- starforge_cli/recipe_lock.py +453 -0
- starforge_cli/scaffold/agent-run.py.tmpl +146 -0
- starforge_cli/scaffold/custom-framework/train.sh +56 -0
- starforge_cli/scaffold/experiment-template/.gitkeep +0 -0
- starforge_cli/scaffold/experiment-template/README.md +36 -0
- starforge_cli/scaffold/experiment-template/config.yaml +44 -0
- starforge_cli/scaffold/project/common/README.md +12 -0
- starforge_cli/scaffold/project/common/__init__.py +0 -0
- starforge_cli/scaffold/project/configs/README.md +103 -0
- starforge_cli/scaffold/project/configs/base/README.md +24 -0
- starforge_cli/scaffold/project/configs/base/distillation_math.yaml +284 -0
- starforge_cli/scaffold/project/configs/base/grpo_lora.yaml +30 -0
- starforge_cli/scaffold/project/configs/base/grpo_math_1B.yaml +470 -0
- starforge_cli/scaffold/project/configs/base/grpo_megatron.yaml +43 -0
- starforge_cli/scaffold/project/configs/base/grpo_noncolocated.yaml +18 -0
- starforge_cli/scaffold/project/configs/base/grpo_sliding_puzzle.yaml +81 -0
- starforge_cli/scaffold/project/configs/base/ppo_math_1B.yaml +454 -0
- starforge_cli/scaffold/project/configs/base/rm.yaml +224 -0
- starforge_cli/scaffold/project/configs/base/sft.yaml +294 -0
- starforge_cli/scaffold/project/configs/models/README.md +16 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-4b.yaml +12 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-9b.yaml +10 -0
- starforge_cli/scaffold/project/gitignore +11 -0
- starforge_cli/spec_builder.py +372 -0
- starforge_cli-0.1.6.dist-info/METADATA +40 -0
- starforge_cli-0.1.6.dist-info/RECORD +55 -0
- starforge_cli-0.1.6.dist-info/WHEEL +4 -0
- starforge_cli-0.1.6.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
# GRPO Algorithm Configuration
|
|
2
|
+
grpo:
|
|
3
|
+
num_prompts_per_step: 32
|
|
4
|
+
num_generations_per_prompt: 16
|
|
5
|
+
max_rollout_turns: 1 # for multi-turn rollouts. Math Environments just have 1 turn (answering the question)
|
|
6
|
+
max_num_epochs: 1
|
|
7
|
+
max_num_steps: 1000000
|
|
8
|
+
normalize_rewards: true
|
|
9
|
+
use_leave_one_out_baseline: true
|
|
10
|
+
val_period: 10
|
|
11
|
+
val_at_start: false
|
|
12
|
+
val_at_end: false
|
|
13
|
+
overlong_filtering: false
|
|
14
|
+
advantage_clip_low: null
|
|
15
|
+
advantage_clip_high: null
|
|
16
|
+
max_val_samples: 256
|
|
17
|
+
val_batch_size: 256
|
|
18
|
+
seed: 42
|
|
19
|
+
use_dynamic_sampling: false
|
|
20
|
+
dynamic_sampling_max_gen_batches: 10
|
|
21
|
+
batch_multiplier: 1
|
|
22
|
+
reward_shaping:
|
|
23
|
+
enabled: false
|
|
24
|
+
overlong_buffer_length: 128
|
|
25
|
+
overlong_buffer_penalty: 1
|
|
26
|
+
max_response_length: ${policy.max_total_sequence_length}
|
|
27
|
+
stop_properly_penalty_coef: null
|
|
28
|
+
|
|
29
|
+
# Advantage Estimator Configuration
|
|
30
|
+
# Options: "grpo" (default) or "reinforce_plus_plus"
|
|
31
|
+
adv_estimator:
|
|
32
|
+
name: "grpo" # Use "reinforce_plus_plus" for Reinforce++ estimator
|
|
33
|
+
normalize_rewards: ${grpo.normalize_rewards}
|
|
34
|
+
use_leave_one_out_baseline: ${grpo.use_leave_one_out_baseline}
|
|
35
|
+
minus_baseline: true # Reinforce++-baseline specific: subtract per-prompt mean baseline
|
|
36
|
+
reward_scaling:
|
|
37
|
+
enabled: false
|
|
38
|
+
source_min: 0.0
|
|
39
|
+
source_max: 1.0
|
|
40
|
+
target_min: 0.0
|
|
41
|
+
target_max: 1.0
|
|
42
|
+
seq_logprob_error_threshold: null
|
|
43
|
+
# Advantage value assigned to invalid tool call tokens (e.g. -5.0 to penalize). null disables.
|
|
44
|
+
invalid_tool_call_advantage: null
|
|
45
|
+
# Advantage value assigned to tokens with malformed <think>/</think> tags (e.g. -5.0). null disables.
|
|
46
|
+
malformed_thinking_advantage: null
|
|
47
|
+
|
|
48
|
+
async_grpo:
|
|
49
|
+
enabled: false # Set to true to enable async training mode
|
|
50
|
+
# Max age (in training steps) for trajectories used in training
|
|
51
|
+
max_trajectory_age_steps: 1
|
|
52
|
+
in_flight_weight_updates: false # Set to true to enable in-flight weight updates
|
|
53
|
+
recompute_kv_cache_after_weight_updates: false # Set to true to recompute kv cache after in-flight-weight-updates
|
|
54
|
+
|
|
55
|
+
loss_fn:
|
|
56
|
+
reference_policy_kl_penalty: 0.01
|
|
57
|
+
# Can be set to k1, k2, k3
|
|
58
|
+
# For more details, see http://joschu.net/blog/kl-approx.html
|
|
59
|
+
reference_policy_kl_type: "k3"
|
|
60
|
+
kl_input_clamp_value: 20.0
|
|
61
|
+
kl_output_clamp_value: 10.0
|
|
62
|
+
ratio_clip_min: 0.2
|
|
63
|
+
ratio_clip_max: 0.2
|
|
64
|
+
ratio_clip_c: null
|
|
65
|
+
# (default off) loss formulation improvements (docs/guides/grpo.md#loss)
|
|
66
|
+
use_on_policy_kl_approximation: false
|
|
67
|
+
# Async GRPO requires importance sampling correction enabled
|
|
68
|
+
# Set to true when async_grpo.enabled is true
|
|
69
|
+
use_importance_sampling_correction: false
|
|
70
|
+
# "tis" – clamp IS weights to [min, max], where min defaults to 0
|
|
71
|
+
# "icepop" – zero out tokens with IS weight outside [min, max]
|
|
72
|
+
# "seq-mask-tis" – zero out sequences by geometric-mean IS ratio, non-truncated token IS correction
|
|
73
|
+
truncated_importance_sampling_type: null
|
|
74
|
+
truncated_importance_sampling_ratio: null
|
|
75
|
+
# TIS lower bound; null means 0 for TIS. Set explicitly for ICE-POP / seq-mask-tis.
|
|
76
|
+
truncated_importance_sampling_ratio_min: null
|
|
77
|
+
sequence_level_importance_ratios: false
|
|
78
|
+
token_level_loss: true
|
|
79
|
+
force_on_policy_ratio: false # Set to true to force ratio=1.0 (requires train_global_batch_size == num_prompts_per_step * num_generations_per_prompt)
|
|
80
|
+
use_kl_in_reward: false # Reinforce++: add KL penalty to reward instead of loss
|
|
81
|
+
use_cispo: false # CISPO (https://arxiv.org/abs/2506.13585): clipped IS-weight policy optimization
|
|
82
|
+
# Disable PPO ratio (curr_logprobs - prev_logprobs).exp(); falls back to REINFORCE-style
|
|
83
|
+
# surrogate loss. Used by REINFORCE/RLOO. See https://arxiv.org/abs/2402.14740
|
|
84
|
+
disable_ppo_ratio: false
|
|
85
|
+
# VAPO positive-example NLL loss weight: L = L_PPO + mu * L_NLL(correct samples).
|
|
86
|
+
# 0.0 disables it. VAPO paper recommends 0.1.
|
|
87
|
+
positive_example_nll_weight: 0.0
|
|
88
|
+
|
|
89
|
+
checkpointing:
|
|
90
|
+
enabled: true
|
|
91
|
+
checkpoint_dir: "results/grpo"
|
|
92
|
+
metric_name: "val:accuracy" # one of "val:" or "train:" followed by the metric name
|
|
93
|
+
higher_is_better: true
|
|
94
|
+
keep_top_k: 3
|
|
95
|
+
save_period: 10
|
|
96
|
+
checkpoint_must_save_by: null
|
|
97
|
+
model_save_format: "safetensors"
|
|
98
|
+
save_consolidated: false
|
|
99
|
+
save_optimizer: true
|
|
100
|
+
|
|
101
|
+
policy:
|
|
102
|
+
model_name: "Qwen/Qwen2.5-1.5B"
|
|
103
|
+
tokenizer:
|
|
104
|
+
name: ${policy.model_name} ## specify if you'd like to use a tokenizer different from the model's default
|
|
105
|
+
chat_template_kwargs: null # can be used to pass kwargs to the chat template, e.g., enable_thinking=true
|
|
106
|
+
hf_config_overrides: {}
|
|
107
|
+
train_global_batch_size: 512
|
|
108
|
+
train_micro_batch_size: 4
|
|
109
|
+
generation_batch_size: 32 # Only used when generating using HF backend
|
|
110
|
+
logprob_batch_size: ${policy.train_micro_batch_size}
|
|
111
|
+
max_total_sequence_length: 512
|
|
112
|
+
precision: "bfloat16"
|
|
113
|
+
logprob_chunk_size: null
|
|
114
|
+
offload_optimizer_for_logprob: false # Only useful for non-colocated generation since colocated generation will always offload optimizer to cuda before refit
|
|
115
|
+
|
|
116
|
+
dtensor_cfg:
|
|
117
|
+
_v2: true
|
|
118
|
+
enabled: true
|
|
119
|
+
cpu_offload: False
|
|
120
|
+
sequence_parallel: false
|
|
121
|
+
activation_checkpointing: false
|
|
122
|
+
tensor_parallel_size: 1
|
|
123
|
+
context_parallel_size: 1
|
|
124
|
+
custom_parallel_plan: null
|
|
125
|
+
|
|
126
|
+
# Automodel kwargs passed to from_pretrained.
|
|
127
|
+
# Uncomment force_hf if the custom model's adapter doesn't support per-tensor
|
|
128
|
+
# weight conversion (e.g. Qwen2, Llama). Auto-detected if not set.
|
|
129
|
+
# See https://github.com/NVIDIA-NeMo/RL/issues/2072
|
|
130
|
+
automodel_kwargs: {}
|
|
131
|
+
# force_hf: true
|
|
132
|
+
|
|
133
|
+
# LoRA (Low-Rank Adaptation) Configuration
|
|
134
|
+
lora_cfg:
|
|
135
|
+
enabled: False # Set to True to enable LoRA fine-tuning
|
|
136
|
+
target_modules: [] # List of module names to apply LoRA (empty list with match_all_linear=true applies to all linear layers)
|
|
137
|
+
exclude_modules: [] # List of module names to exclude from LoRA
|
|
138
|
+
match_all_linear: true # If True, applies LoRA to all linear layers (overrides target_modules)
|
|
139
|
+
dim: 8 # LoRA rank (r): lower rank = fewer parameters but less capacity. Typical values: 4, 8, 16, 32, 64
|
|
140
|
+
alpha: 32 # LoRA scaling factor: effective learning rate multiplier = alpha/dim. Typical values: 16, 32, 64
|
|
141
|
+
dropout: 0.0 # Dropout probability applied to LoRA layers (0.0 = no dropout)
|
|
142
|
+
dropout_position: "post" # Where to apply dropout: "pre" (before LoRA) or "post" (after LoRA)
|
|
143
|
+
lora_A_init: "xavier" # Initialization method for LoRA A matrix: "xavier" or "uniform"
|
|
144
|
+
use_triton: true # Use Triton-optimized kernels for LoRA (faster but requires flash-attn). Disable when tensor_parallel_size > 1
|
|
145
|
+
|
|
146
|
+
megatron_cfg:
|
|
147
|
+
enabled: false
|
|
148
|
+
force_reconvert_from_hf: False # Set to True to force reconvert of the model from Hugging Face
|
|
149
|
+
empty_unused_memory_level: 1 # 1 is the minimum recommendation for RL since we almost always need to offload before beginning generation. Setting to 0 is faster, but you are more likely to run out of GPU memory.
|
|
150
|
+
activation_checkpointing: false
|
|
151
|
+
# recompute_granularity controls activation checkpointing depth.
|
|
152
|
+
# "full": recompute all activations (default, max memory savings).
|
|
153
|
+
# "selective": recompute only specific modules (see recompute_modules).
|
|
154
|
+
# Only takes effect when activation_checkpointing: true.
|
|
155
|
+
recompute_granularity: "full"
|
|
156
|
+
# Modules to selectively recompute when recompute_granularity="selective".
|
|
157
|
+
# MCore options: ["core_attn", "moe_act", "layernorm", "mla_up_proj", "mlp", "moe", "shared_experts"].
|
|
158
|
+
# Null defaults to ["core_attn"]. Full list and per-module constraints:
|
|
159
|
+
# https://github.com/NVIDIA/Megatron-LM/blob/d30c3ae5469fe3f6a64d4fd2e63b6e7f7844ea81/megatron/core/transformer/transformer_config.py#L483
|
|
160
|
+
# Use ["moe"] for MoE models to recompute expert activations only.
|
|
161
|
+
recompute_modules: null
|
|
162
|
+
tensor_model_parallel_size: 1
|
|
163
|
+
expert_tensor_parallel_size: 1
|
|
164
|
+
expert_model_parallel_size: 1
|
|
165
|
+
pipeline_model_parallel_size: 1
|
|
166
|
+
num_layers_in_first_pipeline_stage: null
|
|
167
|
+
num_layers_in_last_pipeline_stage: null
|
|
168
|
+
context_parallel_size: 1
|
|
169
|
+
pipeline_dtype: ${policy.precision}
|
|
170
|
+
sequence_parallel: false
|
|
171
|
+
freeze_moe_router: true
|
|
172
|
+
moe_router_dtype: "fp64"
|
|
173
|
+
moe_router_load_balancing_type: "none" # "seq_aux_loss" causes logprob error divergence for grpo
|
|
174
|
+
moe_router_bias_update_rate: 0.0 # by default, disable bias updates for grpo
|
|
175
|
+
moe_permute_fusion: true
|
|
176
|
+
# gives ~20% training perf speedup with sequence packing
|
|
177
|
+
apply_rope_fusion: True
|
|
178
|
+
# gives ~25% training perf speedup with sequence packing and apply_rope_fusion
|
|
179
|
+
bias_activation_fusion: True
|
|
180
|
+
defer_fp32_logits: False
|
|
181
|
+
moe_per_layer_logging: False
|
|
182
|
+
moe_enable_deepep: false
|
|
183
|
+
moe_token_dispatcher_type: "alltoall"
|
|
184
|
+
moe_shared_expert_overlap: false
|
|
185
|
+
# Multi-Token Prediction (MTP). mtp_num_layers=0 disables MTP.
|
|
186
|
+
mtp_num_layers: 0
|
|
187
|
+
# MTP loss weight added to the main next-token loss (0.0 disables the MTP loss contribution).
|
|
188
|
+
mtp_loss_scaling_factor: 0.0
|
|
189
|
+
# When true, repeat one MTP layer mtp_num_layers times instead of using distinct layers.
|
|
190
|
+
mtp_use_repeated_layer: false
|
|
191
|
+
# When true, detach MTP heads so MTP loss does not affect main-model gradients.
|
|
192
|
+
mtp_detach_heads: false
|
|
193
|
+
gradient_accumulation_fusion: false
|
|
194
|
+
use_fused_weighted_squared_relu: false
|
|
195
|
+
|
|
196
|
+
peft:
|
|
197
|
+
enabled: false
|
|
198
|
+
target_modules: []
|
|
199
|
+
exclude_modules: []
|
|
200
|
+
dim: 8
|
|
201
|
+
alpha: 32
|
|
202
|
+
dropout: 0.0
|
|
203
|
+
dropout_position: "post"
|
|
204
|
+
lora_A_init_method: "xavier"
|
|
205
|
+
lora_B_init_method: "zero"
|
|
206
|
+
a2a_experimental: false
|
|
207
|
+
lora_dtype: None
|
|
208
|
+
|
|
209
|
+
optimizer:
|
|
210
|
+
optimizer: "adam"
|
|
211
|
+
lr: 5.0e-6
|
|
212
|
+
min_lr: 5.0e-7
|
|
213
|
+
weight_decay: 0.01
|
|
214
|
+
bf16: true
|
|
215
|
+
fp16: false
|
|
216
|
+
params_dtype: "float32"
|
|
217
|
+
|
|
218
|
+
#adam
|
|
219
|
+
adam_beta1: 0.9
|
|
220
|
+
adam_beta2: 0.999
|
|
221
|
+
adam_eps: 1e-8
|
|
222
|
+
|
|
223
|
+
#sgd
|
|
224
|
+
sgd_momentum: 0.9
|
|
225
|
+
|
|
226
|
+
#distributed optimizer
|
|
227
|
+
use_distributed_optimizer: true
|
|
228
|
+
use_precision_aware_optimizer: true
|
|
229
|
+
|
|
230
|
+
clip_grad: ${policy.max_grad_norm}
|
|
231
|
+
|
|
232
|
+
# optimizer cpu offload
|
|
233
|
+
optimizer_cpu_offload: false
|
|
234
|
+
optimizer_offload_fraction: 0.0
|
|
235
|
+
|
|
236
|
+
scheduler:
|
|
237
|
+
start_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
238
|
+
end_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
239
|
+
weight_decay_incr_style: "constant"
|
|
240
|
+
lr_decay_style: "constant"
|
|
241
|
+
lr_decay_iters: 1000
|
|
242
|
+
lr_warmup_iters: 13
|
|
243
|
+
lr_warmup_init: 5.0e-7
|
|
244
|
+
|
|
245
|
+
distributed_data_parallel_config:
|
|
246
|
+
grad_reduce_in_fp32: false
|
|
247
|
+
overlap_grad_reduce: true
|
|
248
|
+
overlap_param_gather: true
|
|
249
|
+
use_custom_fsdp: false
|
|
250
|
+
data_parallel_sharding_strategy: "optim_grads_params"
|
|
251
|
+
|
|
252
|
+
fp8_cfg:
|
|
253
|
+
enabled: false
|
|
254
|
+
fp8: "e4m3"
|
|
255
|
+
fp8_recipe: "blockwise"
|
|
256
|
+
fp8_param: false
|
|
257
|
+
|
|
258
|
+
env_vars: null
|
|
259
|
+
|
|
260
|
+
draft:
|
|
261
|
+
enabled: false
|
|
262
|
+
model_name: null
|
|
263
|
+
loss_weight: 0.1
|
|
264
|
+
num_layers: null
|
|
265
|
+
aux_layer_indices: null
|
|
266
|
+
|
|
267
|
+
# See docs/design-docs/sequence-packing-and-dynamic-batching.md
|
|
268
|
+
# for more details on dynamic batching and sequence packing.
|
|
269
|
+
dynamic_batching:
|
|
270
|
+
enabled: False
|
|
271
|
+
train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
|
|
272
|
+
logprob_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.logprob_batch_size}}
|
|
273
|
+
sequence_length_round: 64
|
|
274
|
+
|
|
275
|
+
sequence_packing:
|
|
276
|
+
enabled: True
|
|
277
|
+
train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
|
|
278
|
+
logprob_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.logprob_batch_size}}
|
|
279
|
+
algorithm: "modified_first_fit_decreasing"
|
|
280
|
+
sequence_length_round: 64
|
|
281
|
+
|
|
282
|
+
# makes the training sequence length divisible by the tensor parallel size
|
|
283
|
+
# this is useful for sequence parallel training
|
|
284
|
+
make_sequence_length_divisible_by: ${policy.dtensor_cfg.tensor_parallel_size}
|
|
285
|
+
max_grad_norm: 1.0
|
|
286
|
+
|
|
287
|
+
optimizer:
|
|
288
|
+
name: "torch.optim.AdamW"
|
|
289
|
+
kwargs:
|
|
290
|
+
lr: 5.0e-6
|
|
291
|
+
weight_decay: 0.01
|
|
292
|
+
betas: [0.9, 0.999]
|
|
293
|
+
eps: 1e-8
|
|
294
|
+
|
|
295
|
+
scheduler:
|
|
296
|
+
- name: "torch.optim.lr_scheduler.LinearLR"
|
|
297
|
+
kwargs:
|
|
298
|
+
start_factor: 0.1
|
|
299
|
+
end_factor: 1.0
|
|
300
|
+
total_iters: 50
|
|
301
|
+
- name: "torch.optim.lr_scheduler.ConstantLR"
|
|
302
|
+
kwargs:
|
|
303
|
+
factor: 1.0
|
|
304
|
+
total_iters: 10000000000
|
|
305
|
+
- milestones: [50]
|
|
306
|
+
|
|
307
|
+
generation:
|
|
308
|
+
# Port range for vLLM HTTP servers, kept below the OS ephemeral range.
|
|
309
|
+
# See ray.sub for the full port layout.
|
|
310
|
+
port_range_low: 11001
|
|
311
|
+
port_range_high: 15000
|
|
312
|
+
backend: "vllm"
|
|
313
|
+
max_new_tokens: ${policy.max_total_sequence_length}
|
|
314
|
+
temperature: 1.0
|
|
315
|
+
top_p: 1.0
|
|
316
|
+
top_k: null
|
|
317
|
+
stop_token_ids: null
|
|
318
|
+
stop_strings: null
|
|
319
|
+
mcore_generation_config:
|
|
320
|
+
async_engine: false
|
|
321
|
+
max_model_len: ${policy.max_total_sequence_length} # Engine-side max sequence length.
|
|
322
|
+
cuda_graph_impl: "local" # CUDA-graph implementation. Options: "none", "local", "transformer_engine", "full_iteration".
|
|
323
|
+
inference_cuda_graph_scope: "block" # Inference CUDA-graph scope. Options: "none", "layer", "block".
|
|
324
|
+
buffer_size_gb: 10
|
|
325
|
+
num_cuda_graphs: 4 # Number of CUDA graphs to pre-compile for different batch sizes
|
|
326
|
+
block_size_tokens: 256 # Size of each KV cache block in tokens (affects memory granularity)
|
|
327
|
+
use_cuda_graphs_for_non_decode_steps: true # Enable CUDA graphs for prefill/context processing
|
|
328
|
+
enable_chunked_prefill: true # Split long prefills into chunks for better memory management
|
|
329
|
+
enable_prefix_caching: false # Reuse KV blocks across requests sharing a prompt prefix.
|
|
330
|
+
max_tokens: 16384 # Maximum number of tokens to use in a single step. Analogous to vllm's max_num_batched_tokens
|
|
331
|
+
kv_cache_management_mode: "persist" # KV cache lifecycle across suspend/resume. Options: "persist", "offload". To select "recompute", set grpo.async_grpo.recompute_kv_cache_after_weight_updates=true.
|
|
332
|
+
materialize_only_last_token_logits: true
|
|
333
|
+
num_speculative_tokens: 0
|
|
334
|
+
refit_backend: "nvshmem" # Copy-service backend for non-colocated megatron weight refit. Options: "gloo" or "nvshmem".
|
|
335
|
+
parsers: [] # OpenAI tool-call / response parser names exposed by the HTTP server (only used if expose_http_server=true).
|
|
336
|
+
expose_http_server: false # Start an OpenAI-compatible HTTP server alongside the persistent engine. Required for NeMo-Gym.
|
|
337
|
+
vllm_cfg:
|
|
338
|
+
async_engine: false
|
|
339
|
+
precision: ${policy.precision}
|
|
340
|
+
kv_cache_dtype: "auto"
|
|
341
|
+
tensor_parallel_size: 1
|
|
342
|
+
pipeline_parallel_size: 1
|
|
343
|
+
expert_parallel_size: 1 # When EP > 1, EP must be a multiple of TP since vLLM's EP = DP * TP
|
|
344
|
+
gpu_memory_utilization: 0.6
|
|
345
|
+
max_model_len: ${policy.max_total_sequence_length}
|
|
346
|
+
# when enforce_eager is False, it is optional to set ++policy.generation.vllm_kwargs.compilation_config.backend=eager for better accuracy,
|
|
347
|
+
# with the flag, vllm will use the custom CUDA kernels instead of the Triton kernels generated by torch.compile
|
|
348
|
+
# for more details, see convergence issue https://github.com/NVIDIA-NeMo/RL/issues/998
|
|
349
|
+
enforce_eager: False
|
|
350
|
+
use_tqdm: true # Set to false to suppress vLLM generation progress bars. Only applies when async_engine is false
|
|
351
|
+
use_deep_gemm: False
|
|
352
|
+
num_last_layers_in_bf16: 0
|
|
353
|
+
num_first_layers_in_bf16: 0
|
|
354
|
+
enable_vllm_metrics_logger: true # Set to true to enable vLLM internal metrics logger, turn off for better performance
|
|
355
|
+
vllm_metrics_logger_interval: 0.5 # Interval in seconds to collect vLLM logger metrics
|
|
356
|
+
vllm_kwargs: {}
|
|
357
|
+
colocated:
|
|
358
|
+
# true: generation shares training GPUs
|
|
359
|
+
# false: uses dedicated generation resources
|
|
360
|
+
enabled: true
|
|
361
|
+
# only relevant when enabled is false
|
|
362
|
+
resources:
|
|
363
|
+
gpus_per_node: null # Decides num gpus to be dedicated to generation when there is one node in the cluster i.e cluster.num_nodes == 1
|
|
364
|
+
num_nodes: null # Decides number of nodes to be dedicated to generation
|
|
365
|
+
|
|
366
|
+
data:
|
|
367
|
+
max_input_seq_length: ${policy.max_total_sequence_length} # upper bound, real truncation occurs at vllm.max_model_len
|
|
368
|
+
shuffle: true
|
|
369
|
+
num_workers: 1
|
|
370
|
+
|
|
371
|
+
# use multiple dataloader for train
|
|
372
|
+
# see https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/grpo.md#multiple-dataloaders for more details.
|
|
373
|
+
use_multiple_dataloader: false
|
|
374
|
+
|
|
375
|
+
# dataset
|
|
376
|
+
train:
|
|
377
|
+
dataset_name: OpenMathInstruct-2
|
|
378
|
+
split_validation_size: 0.05 # use 5% of the training data as validation data
|
|
379
|
+
seed: ${grpo.seed} # seed for train/validation split when split_validation_size > 0
|
|
380
|
+
validation: null
|
|
381
|
+
# default settings for all datasets
|
|
382
|
+
default:
|
|
383
|
+
prompt_file: "examples/prompts/cot.txt"
|
|
384
|
+
system_prompt_file: null
|
|
385
|
+
processor: "math_hf_data_processor"
|
|
386
|
+
env_name: "math"
|
|
387
|
+
|
|
388
|
+
# You can also use multiple datasets by using a list of datasets.
|
|
389
|
+
# See `examples/configs/grpo_multiple_datasets.yaml` for a full configuration example.
|
|
390
|
+
|
|
391
|
+
# You can use custom response datasets for training and validation. For example:
|
|
392
|
+
# train:
|
|
393
|
+
# # this dataset will override input_key and use the default values for other vars
|
|
394
|
+
# data_path: /path/to/local/train_dataset.jsonl
|
|
395
|
+
# input_key: question
|
|
396
|
+
# validation:
|
|
397
|
+
# # this dataset will use the default values for other vars except data_path
|
|
398
|
+
# data_path: /path/to/local/val_dataset.jsonl
|
|
399
|
+
# default:
|
|
400
|
+
# # will use below vars as default values if dataset doesn't specify it
|
|
401
|
+
# dataset_name: ResponseDataset
|
|
402
|
+
# input_key: input
|
|
403
|
+
# output_key: output
|
|
404
|
+
# prompt_file: null
|
|
405
|
+
# system_prompt_file: null
|
|
406
|
+
# processor: "math_hf_data_processor"
|
|
407
|
+
# env_name: math
|
|
408
|
+
# See https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/grpo.md#datasets for more details.
|
|
409
|
+
|
|
410
|
+
env:
|
|
411
|
+
math:
|
|
412
|
+
num_workers: 8
|
|
413
|
+
math_verify_impl: "hf_math_verify"
|
|
414
|
+
|
|
415
|
+
logger:
|
|
416
|
+
log_dir: "logs" # Base directory for all logs
|
|
417
|
+
num_val_samples_to_print: 0 # Number of validation samples to pretty print on terminal
|
|
418
|
+
wandb_enabled: false
|
|
419
|
+
tensorboard_enabled: false
|
|
420
|
+
mlflow_enabled: false # Disable MLflow logging
|
|
421
|
+
swanlab_enabled: false # Disable SwanLab logging
|
|
422
|
+
monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
|
|
423
|
+
wandb:
|
|
424
|
+
project: "grpo-dev"
|
|
425
|
+
name: "grpo-dev-logger"
|
|
426
|
+
swanlab:
|
|
427
|
+
project: "grpo-dev"
|
|
428
|
+
name: "grpo-dev-logger"
|
|
429
|
+
tensorboard: {}
|
|
430
|
+
mlflow:
|
|
431
|
+
experiment_name: "grpo-dev"
|
|
432
|
+
run_name: "grpo-dev-logger"
|
|
433
|
+
tracking_uri: "http://localhost:5000"
|
|
434
|
+
gpu_monitoring:
|
|
435
|
+
collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
|
|
436
|
+
flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)
|
|
437
|
+
|
|
438
|
+
cluster:
|
|
439
|
+
gpus_per_node: 1
|
|
440
|
+
num_nodes: 1
|
|
441
|
+
# Port range for the distributed master address (TCPStore / NCCL rendezvous)
|
|
442
|
+
# and per-worker available ports. Kept below the OS ephemeral range
|
|
443
|
+
# (32768-60999 on stock Linux). See ray.sub for the full port layout.
|
|
444
|
+
master_port_range_low: 25000
|
|
445
|
+
master_port_range_high: 28000
|
|
446
|
+
segment_size: null # Nodes per NVLink domain segment for topology-aware alignment; null to disable
|
|
447
|
+
|
|
448
|
+
# TransferQueue-mediated data plane for sync GRPO.
|
|
449
|
+
# Off by default — the legacy grpo_train trainer never engages this.
|
|
450
|
+
# Flip enabled=true and run grpo_train_sync to use TQ-mediated bulk
|
|
451
|
+
# transfer between rollout and train. See nemo_rl/data_plane/README.md.
|
|
452
|
+
data_plane:
|
|
453
|
+
enabled: false
|
|
454
|
+
impl: transfer_queue
|
|
455
|
+
backend: "simple" # TQ storage backend ('simple' or 'mooncake_cpu')
|
|
456
|
+
storage_capacity: 1000000 # max samples retained per partition
|
|
457
|
+
num_storage_units: 2 # storage shards
|
|
458
|
+
claim_meta_poll_interval_s: 0.5 # blocking-claim poll cadence
|
|
459
|
+
global_segment_size: 549755813888 # 512 GiB — used when backend == "mooncake_cpu"
|
|
460
|
+
local_buffer_size: 68719476736 # 64 GiB — used when backend == "mooncake_cpu"
|
|
461
|
+
# observability: # NotRequired
|
|
462
|
+
# enabled: false
|
|
463
|
+
|
|
464
|
+
# Multi-Teacher On-Policy Distillation (MOPD): distills from one or more teacher
|
|
465
|
+
# models into the policy via token-level teacher-minus-student logprob advantages,
|
|
466
|
+
# served by non-colocated teacher worker groups (OPD advantage estimator +
|
|
467
|
+
# nemo_gym). null = disabled (default). See
|
|
468
|
+
# examples/configs/recipes/llm/mopd-qwen3-1.7b-3n8g-megatron-pack.yaml for a full
|
|
469
|
+
# enabled example.
|
|
470
|
+
on_policy_distillation: null
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Megatron-Core 后端 + 低显存实测 GRPO 调优(partial 片段,本身不含 defaults)。
|
|
2
|
+
# 用法:在实验 config 的 defaults 列表里,把它加在 base / 模型片段「之后」,让它覆盖默认的 DTensor:
|
|
3
|
+
# defaults:
|
|
4
|
+
# - ../../configs/base/grpo_math_1B.yaml
|
|
5
|
+
# - ../../configs/models/qwen3.5-9b.yaml
|
|
6
|
+
# - ../../configs/base/grpo_megatron.yaml # ← 切 Megatron 后端 + 低显存调优
|
|
7
|
+
# 想切回 DTensor/FSDP:删掉这一行,并把 lr 写回 policy.optimizer.kwargs.lr。
|
|
8
|
+
#
|
|
9
|
+
# 下面分两块:
|
|
10
|
+
# (A) 切后端必需项——对照官方 examples/configs/grpo_math_*_megatron.yaml。
|
|
11
|
+
# (B) 低显存实测显存/性能调优——来自跑通的 grpo_math_{4B,9B}_megatron。
|
|
12
|
+
# 换硬件(如 H200)时,在该硬件的 cluster/<profile>/overrides.conf 里覆盖这些键即可
|
|
13
|
+
# (这些键此处已存在,CLI override 的 struct 模式不会报错)。
|
|
14
|
+
policy:
|
|
15
|
+
# (A) 后端切换
|
|
16
|
+
dtensor_cfg:
|
|
17
|
+
enabled: false # 关 DTensor / FSDP
|
|
18
|
+
optimizer: null # 去掉 FSDP torch 优化器,改用 policy.megatron_cfg.optimizer
|
|
19
|
+
scheduler: null # 去掉 FSDP 调度器,改用 policy.megatron_cfg.scheduler
|
|
20
|
+
# 训练序列长度对齐到 Megatron 的 TP(基底默认对齐 DTensor 的 TP)
|
|
21
|
+
make_sequence_length_divisible_by: ${policy.megatron_cfg.tensor_model_parallel_size}
|
|
22
|
+
|
|
23
|
+
# (B) 低显存:小 micro batch;关 sequence packing(实测两份配置都关)
|
|
24
|
+
train_micro_batch_size: 1
|
|
25
|
+
sequence_packing:
|
|
26
|
+
enabled: false
|
|
27
|
+
|
|
28
|
+
megatron_cfg:
|
|
29
|
+
enabled: true # 开 Megatron-Core
|
|
30
|
+
# converter_type 由 AutoBridge 按 HF 架构自动识别,无需按模型改(保留基底默认即可)
|
|
31
|
+
# (B) 低显存实测显存/性能:
|
|
32
|
+
activation_checkpointing: true # 重算换显存
|
|
33
|
+
apply_rope_fusion: false # 官方对该路径明确关闭
|
|
34
|
+
defer_fp32_logits: true # 延迟 fp32 logits,省显存
|
|
35
|
+
empty_unused_memory_level: 2 # 官方默认 1,显存紧张时升到 2
|
|
36
|
+
distributed_data_parallel_config:
|
|
37
|
+
overlap_grad_reduce: false # 跨节点关 overlap 更稳
|
|
38
|
+
overlap_param_gather: false
|
|
39
|
+
|
|
40
|
+
generation:
|
|
41
|
+
vllm_cfg:
|
|
42
|
+
enforce_eager: true # 关 CUDA graph:省显存、避免编译问题
|
|
43
|
+
gpu_memory_utilization: 0.4 # 与训练 colocated,按显存在 0.3~0.5 间调
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# 非 colocated 生成 overlay:训练与 vLLM 生成各占独立 GPU/节点(实测 9B 用法)。
|
|
2
|
+
# 2 卡场景:1 卡专跑生成、1 卡训练 → 训练侧 TP=PP=1。
|
|
3
|
+
# 删此行即回 colocated(生成与训练共用 GPU)。
|
|
4
|
+
#
|
|
5
|
+
# 为什么显存紧张时 9B 选非 colocated:
|
|
6
|
+
# - colocated 单卡要在“训练态↔vLLM 生成态”间反复 offload/refit,模型一大(9B)
|
|
7
|
+
# 又紧张又易出问题;非 colocated 让两张卡角色隔离、各自常驻,更稳(代价是每步把权重
|
|
8
|
+
# 经网络同步给生成卡)。colocated 更省卡,适合小模型(如 4B 的 colocated+PP=2)。
|
|
9
|
+
#
|
|
10
|
+
# 注意:非 colocated 把 cluster 里一部分卡划给生成,剩下的才给训练。
|
|
11
|
+
# 在 2 卡集群上训练只剩 1 卡,所以无法再 PP=2(PP=2 需 2 张训练卡,要用就回 colocated)。
|
|
12
|
+
policy:
|
|
13
|
+
generation:
|
|
14
|
+
colocated:
|
|
15
|
+
enabled: false
|
|
16
|
+
resources:
|
|
17
|
+
gpus_per_node: 1 # 专给生成的卡数(每节点)
|
|
18
|
+
num_nodes: 1 # 专给生成的节点数
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# GRPO Algorithm Configuration
|
|
2
|
+
defaults: "grpo_math_1B.yaml"
|
|
3
|
+
|
|
4
|
+
grpo:
|
|
5
|
+
num_prompts_per_step: 32
|
|
6
|
+
num_generations_per_prompt: 16
|
|
7
|
+
max_rollout_turns: 50 # Maximum turns allowed per rollout
|
|
8
|
+
max_num_steps: 10000
|
|
9
|
+
max_num_epochs: 1
|
|
10
|
+
|
|
11
|
+
checkpointing:
|
|
12
|
+
enabled: true
|
|
13
|
+
checkpoint_dir: "results/grpo-sliding-puzzle"
|
|
14
|
+
metric_name: "val:accuracy" # one of "val:" or "train:" followed by the metric name
|
|
15
|
+
higher_is_better: true
|
|
16
|
+
keep_top_k: 3
|
|
17
|
+
save_period: 10
|
|
18
|
+
checkpoint_must_save_by: null
|
|
19
|
+
|
|
20
|
+
policy:
|
|
21
|
+
model_name: "Qwen/Qwen2.5-1.5B-Instruct"
|
|
22
|
+
max_total_sequence_length: 1024
|
|
23
|
+
|
|
24
|
+
dtensor_cfg:
|
|
25
|
+
enabled: true
|
|
26
|
+
cpu_offload: true
|
|
27
|
+
activation_checkpointing: true
|
|
28
|
+
sequence_parallel: true
|
|
29
|
+
|
|
30
|
+
generation:
|
|
31
|
+
backend: "vllm"
|
|
32
|
+
max_new_tokens: ${policy.max_total_sequence_length}
|
|
33
|
+
temperature: 1.0
|
|
34
|
+
# Setting top_p/top_k to 0.999/10000 to strip out Qwen's special/illegal tokens
|
|
35
|
+
# https://github.com/NVIDIA-NeMo/RL/issues/237
|
|
36
|
+
top_p: 0.999
|
|
37
|
+
top_k: 10000
|
|
38
|
+
stop_token_ids: null
|
|
39
|
+
stop_strings: null
|
|
40
|
+
vllm_cfg:
|
|
41
|
+
async_engine: false
|
|
42
|
+
tensor_parallel_size: 1
|
|
43
|
+
pipeline_parallel_size: 1
|
|
44
|
+
expert_parallel_size: 1
|
|
45
|
+
gpu_memory_utilization: 0.6
|
|
46
|
+
max_model_len: ${policy.max_total_sequence_length}
|
|
47
|
+
|
|
48
|
+
data:
|
|
49
|
+
add_system_prompt: false
|
|
50
|
+
shuffle: false # disable dataloader shuffle, shuffle is handled within the dataset
|
|
51
|
+
|
|
52
|
+
env:
|
|
53
|
+
sliding_puzzle_game:
|
|
54
|
+
cfg:
|
|
55
|
+
game_config:
|
|
56
|
+
size: 5 # Size of the puzzle (e.g., 2 for 2x2, 3 for 3x3)
|
|
57
|
+
shuffle_moves: 10 # Number of random moves to shuffle the solved state
|
|
58
|
+
max_moves: 30 # Maximum moves allowed per episode
|
|
59
|
+
|
|
60
|
+
logger:
|
|
61
|
+
log_dir: "logs" # Base directory for all logs
|
|
62
|
+
num_val_samples_to_print: 0 # Number of validation samples to pretty print on terminal
|
|
63
|
+
wandb_enabled: false
|
|
64
|
+
tensorboard_enabled: false
|
|
65
|
+
mlflow_enabled: false
|
|
66
|
+
swanlab_enabled: false # Disable SwanLab logging
|
|
67
|
+
monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
|
|
68
|
+
wandb:
|
|
69
|
+
project: "grpo-dev"
|
|
70
|
+
name: "grpo-dev-sliding_puzzle"
|
|
71
|
+
swanlab:
|
|
72
|
+
project: "grpo-dev"
|
|
73
|
+
name: "grpo-dev-sliding_puzzle"
|
|
74
|
+
tensorboard: {}
|
|
75
|
+
mlflow:
|
|
76
|
+
experiment_name: "grpo-dev"
|
|
77
|
+
run_name: "grpo-dev-sliding_puzzle"
|
|
78
|
+
tracking_uri: "http://localhost:5000"
|
|
79
|
+
gpu_monitoring:
|
|
80
|
+
collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
|
|
81
|
+
flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)
|