starforge-cli 0.1.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- starforge_cli/__init__.py +3 -0
- starforge_cli/api_client.py +589 -0
- starforge_cli/auth.py +349 -0
- starforge_cli/catalog.py +124 -0
- starforge_cli/cli.py +74 -0
- starforge_cli/cli_ui.py +469 -0
- starforge_cli/client_device.py +104 -0
- starforge_cli/commands/__init__.py +1 -0
- starforge_cli/commands/admin.py +140 -0
- starforge_cli/commands/bench.py +94 -0
- starforge_cli/commands/common.py +178 -0
- starforge_cli/commands/dataset.py +150 -0
- starforge_cli/commands/exp.py +213 -0
- starforge_cli/commands/init.py +52 -0
- starforge_cli/commands/jobs.py +223 -0
- starforge_cli/commands/login.py +54 -0
- starforge_cli/commands/plugin.py +243 -0
- starforge_cli/commands/recipe.py +163 -0
- starforge_cli/commands/serve.py +79 -0
- starforge_cli/commands/submit.py +467 -0
- starforge_cli/commands/sweep.py +154 -0
- starforge_cli/config_resolve.py +17 -0
- starforge_cli/data_prep.py +60 -0
- starforge_cli/new_experiment.py +195 -0
- starforge_cli/packing.py +179 -0
- starforge_cli/plugins_lock.py +73 -0
- starforge_cli/project.py +130 -0
- starforge_cli/recipe_lock.py +453 -0
- starforge_cli/scaffold/agent-run.py.tmpl +146 -0
- starforge_cli/scaffold/custom-framework/train.sh +56 -0
- starforge_cli/scaffold/experiment-template/.gitkeep +0 -0
- starforge_cli/scaffold/experiment-template/README.md +36 -0
- starforge_cli/scaffold/experiment-template/config.yaml +44 -0
- starforge_cli/scaffold/project/common/README.md +12 -0
- starforge_cli/scaffold/project/common/__init__.py +0 -0
- starforge_cli/scaffold/project/configs/README.md +103 -0
- starforge_cli/scaffold/project/configs/base/README.md +24 -0
- starforge_cli/scaffold/project/configs/base/distillation_math.yaml +284 -0
- starforge_cli/scaffold/project/configs/base/grpo_lora.yaml +30 -0
- starforge_cli/scaffold/project/configs/base/grpo_math_1B.yaml +470 -0
- starforge_cli/scaffold/project/configs/base/grpo_megatron.yaml +43 -0
- starforge_cli/scaffold/project/configs/base/grpo_noncolocated.yaml +18 -0
- starforge_cli/scaffold/project/configs/base/grpo_sliding_puzzle.yaml +81 -0
- starforge_cli/scaffold/project/configs/base/ppo_math_1B.yaml +454 -0
- starforge_cli/scaffold/project/configs/base/rm.yaml +224 -0
- starforge_cli/scaffold/project/configs/base/sft.yaml +294 -0
- starforge_cli/scaffold/project/configs/models/README.md +16 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-4b.yaml +12 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-9b.yaml +10 -0
- starforge_cli/scaffold/project/gitignore +11 -0
- starforge_cli/spec_builder.py +372 -0
- starforge_cli-0.1.6.dist-info/METADATA +40 -0
- starforge_cli-0.1.6.dist-info/RECORD +55 -0
- starforge_cli-0.1.6.dist-info/WHEEL +4 -0
- starforge_cli-0.1.6.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,294 @@
|
|
|
1
|
+
# SFT Algorithm Configuration
|
|
2
|
+
sft:
|
|
3
|
+
## total number of steps to train will equal
|
|
4
|
+
## min((max_num_epochs * len(train_dataloader)), max_num_steps)
|
|
5
|
+
max_num_epochs: 1
|
|
6
|
+
max_num_steps: 60
|
|
7
|
+
|
|
8
|
+
val_period: 10
|
|
9
|
+
val_batches: 8
|
|
10
|
+
val_global_batch_size: 32
|
|
11
|
+
val_micro_batch_size: 1
|
|
12
|
+
val_at_start: true
|
|
13
|
+
val_at_end: false
|
|
14
|
+
seed: 42
|
|
15
|
+
only_unmask_final: false
|
|
16
|
+
|
|
17
|
+
checkpointing:
|
|
18
|
+
enabled: true
|
|
19
|
+
checkpoint_dir: "results/sft"
|
|
20
|
+
metric_name: "val:val_loss" # one of "val:" or "train:" followed by the metric name
|
|
21
|
+
higher_is_better: false
|
|
22
|
+
keep_top_k: 3
|
|
23
|
+
save_period: 10
|
|
24
|
+
checkpoint_must_save_by: null
|
|
25
|
+
save_optimizer: true
|
|
26
|
+
# pretrained_checkpoint: load weights from an existing megatron-bridge or
|
|
27
|
+
# Megatron-LM checkpoint instead of converting from HuggingFace.
|
|
28
|
+
# pretrained_checkpoint:
|
|
29
|
+
# path: "/path/to/checkpoint" # iter dir or checkpoint root
|
|
30
|
+
# format: "megatron_bridge" # or "megatron_lm"
|
|
31
|
+
|
|
32
|
+
policy:
|
|
33
|
+
model_name: "meta-llama/Llama-3.2-1B"
|
|
34
|
+
tokenizer:
|
|
35
|
+
name: ${policy.model_name} ## specify if you'd like to use a tokenizer different from the model's default
|
|
36
|
+
# chat_template can be a Jinja template string or path to a .jinja file
|
|
37
|
+
chat_template: "{% for message in messages %}{%- if message['role'] == 'system' %}{{'Context: ' + message['content'].strip()}}{%- elif message['role'] == 'user' %}{{' Question: ' + message['content'].strip() + ' Answer:'}}{%- elif message['role'] == 'assistant' %}{{' ' + message['content'].strip()}}{%- endif %}{% endfor %}"
|
|
38
|
+
chat_template_kwargs: null # can be used to pass kwargs to the chat template, e.g., enable_thinking=true
|
|
39
|
+
train_global_batch_size: 32
|
|
40
|
+
train_micro_batch_size: 1
|
|
41
|
+
max_total_sequence_length: 1024
|
|
42
|
+
precision: "bfloat16"
|
|
43
|
+
|
|
44
|
+
offload_optimizer_for_logprob: false
|
|
45
|
+
|
|
46
|
+
dtensor_cfg:
|
|
47
|
+
_v2: true
|
|
48
|
+
enabled: true
|
|
49
|
+
env_vars: {}
|
|
50
|
+
cpu_offload: False
|
|
51
|
+
sequence_parallel: false
|
|
52
|
+
activation_checkpointing: false
|
|
53
|
+
tensor_parallel_size: 1
|
|
54
|
+
context_parallel_size: 1
|
|
55
|
+
# HSDP replicate dimension size for Automodel DTensor v2.
|
|
56
|
+
# Typically set to the number of nodes; set to 1 to disable HSDP.
|
|
57
|
+
dp_replicate_size: 1
|
|
58
|
+
custom_parallel_plan: null
|
|
59
|
+
|
|
60
|
+
# Automodel kwargs passed to from_pretrained.
|
|
61
|
+
# Uncomment force_hf if the custom model's adapter doesn't support per-tensor
|
|
62
|
+
# weight conversion (e.g. Qwen2, Llama). Auto-detected if not set.
|
|
63
|
+
# See https://github.com/NVIDIA-NeMo/RL/issues/2072
|
|
64
|
+
automodel_kwargs: {}
|
|
65
|
+
# force_hf: true
|
|
66
|
+
|
|
67
|
+
# LoRA (Low-Rank Adaptation) Configuration
|
|
68
|
+
lora_cfg:
|
|
69
|
+
enabled: False # Set to True to enable LoRA fine-tuning
|
|
70
|
+
target_modules: [] # List of module names to apply LoRA (empty list with match_all_linear=true applies to all linear layers)
|
|
71
|
+
exclude_modules: [] # List of module names to exclude from LoRA
|
|
72
|
+
match_all_linear: true # If True, applies LoRA to all linear layers (overrides target_modules)
|
|
73
|
+
dim: 8 # LoRA rank (r): lower rank = fewer parameters but less capacity. Typical values: 4, 8, 16, 32, 64
|
|
74
|
+
alpha: 32 # LoRA scaling factor: effective learning rate multiplier = alpha/dim. Typical values: 16, 32, 64
|
|
75
|
+
dropout: 0.0 # Dropout probability applied to LoRA layers (0.0 = no dropout)
|
|
76
|
+
dropout_position: "post" # Where to apply dropout: "pre" (before LoRA) or "post" (after LoRA)
|
|
77
|
+
lora_A_init: "xavier" # Initialization method for LoRA A matrix: "xavier" or "uniform"
|
|
78
|
+
use_triton: true # Use Triton-optimized kernels for LoRA (faster but requires flash-attn). Disable when tensor_parallel_size > 1
|
|
79
|
+
|
|
80
|
+
dynamic_batching:
|
|
81
|
+
enabled: false
|
|
82
|
+
train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
|
|
83
|
+
sequence_length_round: 64
|
|
84
|
+
|
|
85
|
+
sequence_packing:
|
|
86
|
+
enabled: False
|
|
87
|
+
train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
|
|
88
|
+
algorithm: "modified_first_fit_decreasing"
|
|
89
|
+
sequence_length_round: 64
|
|
90
|
+
|
|
91
|
+
# makes the training sequence length divisible by the tensor parallel size
|
|
92
|
+
# this is useful for sequence parallel training
|
|
93
|
+
make_sequence_length_divisible_by: ${policy.dtensor_cfg.tensor_parallel_size}
|
|
94
|
+
max_grad_norm: 1.0
|
|
95
|
+
|
|
96
|
+
optimizer:
|
|
97
|
+
name: "torch.optim.AdamW"
|
|
98
|
+
kwargs:
|
|
99
|
+
lr: 5.0e-6
|
|
100
|
+
weight_decay: 0.1
|
|
101
|
+
betas: [0.9, 0.98]
|
|
102
|
+
eps: 1e-5
|
|
103
|
+
# when using Dtensor, we need to set foreach
|
|
104
|
+
# and fused to False
|
|
105
|
+
foreach: False
|
|
106
|
+
fused: False
|
|
107
|
+
|
|
108
|
+
## ignored since enabled=false, but needed for testing purposes
|
|
109
|
+
megatron_cfg:
|
|
110
|
+
enabled: false
|
|
111
|
+
use_linear_ce_fusion_loss: false
|
|
112
|
+
linear_ce_fusion_chunk_size: 256
|
|
113
|
+
force_reconvert_from_hf: False # Set to True to force reconvert of the model from Hugging Face
|
|
114
|
+
env_vars: {}
|
|
115
|
+
empty_unused_memory_level: 1
|
|
116
|
+
activation_checkpointing: false
|
|
117
|
+
# recompute_granularity controls activation checkpointing depth.
|
|
118
|
+
# "full": recompute all activations (default, max memory savings).
|
|
119
|
+
# "selective": recompute only specific modules (see recompute_modules).
|
|
120
|
+
# Only takes effect when activation_checkpointing: true.
|
|
121
|
+
recompute_granularity: "full"
|
|
122
|
+
# Modules to selectively recompute when recompute_granularity="selective".
|
|
123
|
+
# MCore options: ["core_attn", "moe_act", "layernorm", "mla_up_proj", "mlp", "moe", "shared_experts"].
|
|
124
|
+
# Null defaults to ["core_attn"]. Full list and per-module constraints:
|
|
125
|
+
# https://github.com/NVIDIA/Megatron-LM/blob/d30c3ae5469fe3f6a64d4fd2e63b6e7f7844ea81/megatron/core/transformer/transformer_config.py#L483
|
|
126
|
+
# Use ["moe"] for MoE models to recompute expert activations only.
|
|
127
|
+
recompute_modules: null
|
|
128
|
+
tensor_model_parallel_size: 1
|
|
129
|
+
expert_tensor_parallel_size: 1
|
|
130
|
+
expert_model_parallel_size: 1
|
|
131
|
+
pipeline_model_parallel_size: 1
|
|
132
|
+
context_parallel_size: 1
|
|
133
|
+
pipeline_dtype: ${policy.precision}
|
|
134
|
+
num_layers_in_first_pipeline_stage: null
|
|
135
|
+
num_layers_in_last_pipeline_stage: null
|
|
136
|
+
sequence_parallel: false
|
|
137
|
+
freeze_moe_router: false
|
|
138
|
+
moe_router_dtype: null
|
|
139
|
+
moe_router_load_balancing_type: "aux_loss"
|
|
140
|
+
moe_router_bias_update_rate: 1e-3
|
|
141
|
+
moe_permute_fusion: true
|
|
142
|
+
#gives ~20% training perf speedup with sequence packing
|
|
143
|
+
apply_rope_fusion: True
|
|
144
|
+
# gives ~25% training perf speedup with sequence packing and apply_rope_fusion
|
|
145
|
+
bias_activation_fusion: True
|
|
146
|
+
defer_fp32_logits: False
|
|
147
|
+
moe_per_layer_logging: False
|
|
148
|
+
moe_enable_deepep: false
|
|
149
|
+
moe_token_dispatcher_type: "alltoall"
|
|
150
|
+
moe_shared_expert_overlap: false
|
|
151
|
+
gradient_accumulation_fusion: false
|
|
152
|
+
use_fused_weighted_squared_relu: false
|
|
153
|
+
|
|
154
|
+
peft:
|
|
155
|
+
enabled: false
|
|
156
|
+
target_modules: []
|
|
157
|
+
exclude_modules: []
|
|
158
|
+
dim: 8
|
|
159
|
+
alpha: 32
|
|
160
|
+
dropout: 0.0
|
|
161
|
+
dropout_position: "post"
|
|
162
|
+
lora_A_init_method: "xavier"
|
|
163
|
+
lora_B_init_method: "zero"
|
|
164
|
+
a2a_experimental: false
|
|
165
|
+
lora_dtype: None
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
optimizer:
|
|
169
|
+
optimizer: "adam" # When weight decay is set, it actually uses AdamW
|
|
170
|
+
lr: 5.0e-6
|
|
171
|
+
min_lr: 4.9999e-6
|
|
172
|
+
weight_decay: 0.1 # When weight decay is set, it actually uses AdamW
|
|
173
|
+
bf16: true
|
|
174
|
+
fp16: false
|
|
175
|
+
params_dtype: "float32"
|
|
176
|
+
|
|
177
|
+
#adam
|
|
178
|
+
adam_beta1: 0.9
|
|
179
|
+
adam_beta2: 0.98
|
|
180
|
+
adam_eps: 1e-5
|
|
181
|
+
|
|
182
|
+
#sgd
|
|
183
|
+
sgd_momentum: 0.9
|
|
184
|
+
|
|
185
|
+
#distributed optimizer
|
|
186
|
+
use_distributed_optimizer: true
|
|
187
|
+
use_precision_aware_optimizer: true
|
|
188
|
+
|
|
189
|
+
clip_grad: ${policy.max_grad_norm}
|
|
190
|
+
|
|
191
|
+
# optimizer cpu offload
|
|
192
|
+
optimizer_cpu_offload: false
|
|
193
|
+
optimizer_offload_fraction: 0.0
|
|
194
|
+
|
|
195
|
+
scheduler:
|
|
196
|
+
start_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
197
|
+
end_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
198
|
+
weight_decay_incr_style: "constant"
|
|
199
|
+
lr_decay_style: "constant"
|
|
200
|
+
lr_decay_iters: 1000
|
|
201
|
+
lr_warmup_iters: 50
|
|
202
|
+
lr_warmup_init: 4.9999e-6
|
|
203
|
+
|
|
204
|
+
distributed_data_parallel_config:
|
|
205
|
+
grad_reduce_in_fp32: false
|
|
206
|
+
overlap_grad_reduce: true
|
|
207
|
+
overlap_param_gather: true
|
|
208
|
+
data_parallel_sharding_strategy: "optim_grads_params"
|
|
209
|
+
use_custom_fsdp: false
|
|
210
|
+
|
|
211
|
+
fp8_cfg:
|
|
212
|
+
enabled: false
|
|
213
|
+
fp8: "e4m3"
|
|
214
|
+
fp8_recipe: "blockwise"
|
|
215
|
+
fp8_param: false
|
|
216
|
+
|
|
217
|
+
data:
|
|
218
|
+
max_input_seq_length: ${policy.max_total_sequence_length}
|
|
219
|
+
add_bos: true
|
|
220
|
+
add_eos: true
|
|
221
|
+
add_generation_prompt: false
|
|
222
|
+
shuffle: true
|
|
223
|
+
num_workers: 1
|
|
224
|
+
|
|
225
|
+
# dataset
|
|
226
|
+
train:
|
|
227
|
+
dataset_name: "squad"
|
|
228
|
+
split: "train"
|
|
229
|
+
validation:
|
|
230
|
+
dataset_name: "squad"
|
|
231
|
+
split: "validation"
|
|
232
|
+
# default settings for all datasets
|
|
233
|
+
default:
|
|
234
|
+
prompt_file: null
|
|
235
|
+
system_prompt_file: null
|
|
236
|
+
processor: "sft_processor"
|
|
237
|
+
|
|
238
|
+
# You can also use multiple datasets by using a list of datasets.
|
|
239
|
+
# See `examples/configs/grpo_multiple_datasets.yaml` for a full configuration example.
|
|
240
|
+
|
|
241
|
+
# You can use custom response datasets for training and validation. For example:
|
|
242
|
+
# train:
|
|
243
|
+
# # this dataset will override input_key and use the default values for other vars
|
|
244
|
+
# data_path: /path/to/local/train_dataset.jsonl
|
|
245
|
+
# input_key: question
|
|
246
|
+
# validation:
|
|
247
|
+
# # this dataset will use the default values for other vars except data_path
|
|
248
|
+
# data_path: /path/to/local/val_dataset.jsonl
|
|
249
|
+
# default:
|
|
250
|
+
# # will use below vars as default values if dataset doesn't specify it
|
|
251
|
+
# dataset_name: ResponseDataset
|
|
252
|
+
# input_key: input
|
|
253
|
+
# output_key: output
|
|
254
|
+
# prompt_file: null
|
|
255
|
+
# system_prompt_file: null
|
|
256
|
+
# processor: "sft_processor"
|
|
257
|
+
# See https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/sft.md#datasets for more details.
|
|
258
|
+
|
|
259
|
+
# OpenAI format specific configs
|
|
260
|
+
# train_data_path: "/path/to/train.jsonl" # Path to training data
|
|
261
|
+
# val_data_path: "/path/to/val.jsonl" # Path to validation data
|
|
262
|
+
# chat_key: "messages" # Key for messages in the data
|
|
263
|
+
# system_key: null # Key for system message (optional)
|
|
264
|
+
# system_prompt: null # Default system prompt (optional)
|
|
265
|
+
# tool_key: "tools" # Key for tools in the data
|
|
266
|
+
# use_preserving_dataset: false # If true, uses PreservingDataset to preserve heterogeneous schemas (e.g., tool calls with varying argument structures)
|
|
267
|
+
|
|
268
|
+
logger:
|
|
269
|
+
log_dir: "logs" # Base directory for all logs
|
|
270
|
+
wandb_enabled: true # Make sure you do a ``wandb login [Your API key]'' before running
|
|
271
|
+
tensorboard_enabled: true
|
|
272
|
+
mlflow_enabled: false
|
|
273
|
+
swanlab_enabled: false # Disable SwanLab logging
|
|
274
|
+
monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
|
|
275
|
+
wandb:
|
|
276
|
+
project: "sft-dev"
|
|
277
|
+
name: "sft-dev-${data.train.dataset_name}"
|
|
278
|
+
swanlab:
|
|
279
|
+
project: "sft-dev"
|
|
280
|
+
name: "sft-dev-${data.train.dataset_name}"
|
|
281
|
+
tensorboard:
|
|
282
|
+
log_dir: "tb_logs-sft-dev-${data.train.dataset_name}"
|
|
283
|
+
mlflow:
|
|
284
|
+
experiment_name: "sft-dev"
|
|
285
|
+
run_name: "sft-dev-${data.train.dataset_name}"
|
|
286
|
+
tracking_uri: "http://localhost:5000"
|
|
287
|
+
gpu_monitoring:
|
|
288
|
+
collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
|
|
289
|
+
flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)
|
|
290
|
+
|
|
291
|
+
cluster:
|
|
292
|
+
gpus_per_node: 1
|
|
293
|
+
num_nodes: 1
|
|
294
|
+
segment_size: null # Nodes per NVLink domain segment for topology-aware alignment; null to disable
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# configs/models — 各基础模型的公共片段
|
|
2
|
+
|
|
3
|
+
每个基础模型一个 partial 配置(只含该模型通用字段,如 `policy.model_name`、tokenizer、
|
|
4
|
+
该模型适配的序列长度 / 并行 / 显存策略)。**不含 `defaults`**,作为实验「多继承」的一项使用。
|
|
5
|
+
|
|
6
|
+
实验 `config.yaml` 里:
|
|
7
|
+
|
|
8
|
+
```yaml
|
|
9
|
+
defaults:
|
|
10
|
+
- ../../configs/base/grpo_math_1B.yaml # 方法基底(官方 v0.6.0)
|
|
11
|
+
- ../../configs/models/qwen3.5-9b.yaml # 模型片段(覆盖基底里的模型字段)
|
|
12
|
+
# 下面再写本实验差异(数据集 / lr / kl / swanlab ...)
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
多继承中**后面的覆盖前面的**,实验自身的顶层键再覆盖两者。新增模型就加一个
|
|
16
|
+
`<model>.yaml`,命名与 `docs/naming-convention.md` 的 model 字段一致。
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# 模型片段:qwen3.5-4b
|
|
2
|
+
# 作为实验「多继承」的一项,覆盖基底里的模型相关字段。本身不含 defaults,是个 partial。
|
|
3
|
+
# 用法(实验 config.yaml):
|
|
4
|
+
# defaults:
|
|
5
|
+
# - ../../configs/base/grpo_math_1B.yaml
|
|
6
|
+
# - ../../configs/models/qwen3.5-4b.yaml
|
|
7
|
+
policy:
|
|
8
|
+
model_name: "Qwen/Qwen3.5-4B"
|
|
9
|
+
tokenizer:
|
|
10
|
+
name: "Qwen/Qwen3.5-4B"
|
|
11
|
+
# 该模型常用的默认值可放这里(实验仍可再覆盖)
|
|
12
|
+
# max_total_sequence_length: 4096
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# 模型片段:qwen3.5-9b
|
|
2
|
+
# 作为实验「多继承」的一项,覆盖基底里的模型相关字段。本身不含 defaults,是个 partial。
|
|
3
|
+
# 注:你实测 GRPO 用的是 base 模型 "Qwen/Qwen3.5-9B-Base";如需从 base 起强化训练,改下面 model_name。
|
|
4
|
+
policy:
|
|
5
|
+
model_name: "Qwen/Qwen3.5-9B"
|
|
6
|
+
tokenizer:
|
|
7
|
+
name: "Qwen/Qwen3.5-9B"
|
|
8
|
+
# 9B 低显存策略(实测):
|
|
9
|
+
logprob_chunk_size: 256 # 分块算 logprob,省显存
|
|
10
|
+
offload_optimizer_for_logprob: true # 非 colocated 生成时把优化器 offload 到 CPU(colocated 下为 no-op)
|