starforge-cli 0.1.6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. starforge_cli/__init__.py +3 -0
  2. starforge_cli/api_client.py +589 -0
  3. starforge_cli/auth.py +349 -0
  4. starforge_cli/catalog.py +124 -0
  5. starforge_cli/cli.py +74 -0
  6. starforge_cli/cli_ui.py +469 -0
  7. starforge_cli/client_device.py +104 -0
  8. starforge_cli/commands/__init__.py +1 -0
  9. starforge_cli/commands/admin.py +140 -0
  10. starforge_cli/commands/bench.py +94 -0
  11. starforge_cli/commands/common.py +178 -0
  12. starforge_cli/commands/dataset.py +150 -0
  13. starforge_cli/commands/exp.py +213 -0
  14. starforge_cli/commands/init.py +52 -0
  15. starforge_cli/commands/jobs.py +223 -0
  16. starforge_cli/commands/login.py +54 -0
  17. starforge_cli/commands/plugin.py +243 -0
  18. starforge_cli/commands/recipe.py +163 -0
  19. starforge_cli/commands/serve.py +79 -0
  20. starforge_cli/commands/submit.py +467 -0
  21. starforge_cli/commands/sweep.py +154 -0
  22. starforge_cli/config_resolve.py +17 -0
  23. starforge_cli/data_prep.py +60 -0
  24. starforge_cli/new_experiment.py +195 -0
  25. starforge_cli/packing.py +179 -0
  26. starforge_cli/plugins_lock.py +73 -0
  27. starforge_cli/project.py +130 -0
  28. starforge_cli/recipe_lock.py +453 -0
  29. starforge_cli/scaffold/agent-run.py.tmpl +146 -0
  30. starforge_cli/scaffold/custom-framework/train.sh +56 -0
  31. starforge_cli/scaffold/experiment-template/.gitkeep +0 -0
  32. starforge_cli/scaffold/experiment-template/README.md +36 -0
  33. starforge_cli/scaffold/experiment-template/config.yaml +44 -0
  34. starforge_cli/scaffold/project/common/README.md +12 -0
  35. starforge_cli/scaffold/project/common/__init__.py +0 -0
  36. starforge_cli/scaffold/project/configs/README.md +103 -0
  37. starforge_cli/scaffold/project/configs/base/README.md +24 -0
  38. starforge_cli/scaffold/project/configs/base/distillation_math.yaml +284 -0
  39. starforge_cli/scaffold/project/configs/base/grpo_lora.yaml +30 -0
  40. starforge_cli/scaffold/project/configs/base/grpo_math_1B.yaml +470 -0
  41. starforge_cli/scaffold/project/configs/base/grpo_megatron.yaml +43 -0
  42. starforge_cli/scaffold/project/configs/base/grpo_noncolocated.yaml +18 -0
  43. starforge_cli/scaffold/project/configs/base/grpo_sliding_puzzle.yaml +81 -0
  44. starforge_cli/scaffold/project/configs/base/ppo_math_1B.yaml +454 -0
  45. starforge_cli/scaffold/project/configs/base/rm.yaml +224 -0
  46. starforge_cli/scaffold/project/configs/base/sft.yaml +294 -0
  47. starforge_cli/scaffold/project/configs/models/README.md +16 -0
  48. starforge_cli/scaffold/project/configs/models/qwen3.5-4b.yaml +12 -0
  49. starforge_cli/scaffold/project/configs/models/qwen3.5-9b.yaml +10 -0
  50. starforge_cli/scaffold/project/gitignore +11 -0
  51. starforge_cli/spec_builder.py +372 -0
  52. starforge_cli-0.1.6.dist-info/METADATA +40 -0
  53. starforge_cli-0.1.6.dist-info/RECORD +55 -0
  54. starforge_cli-0.1.6.dist-info/WHEEL +4 -0
  55. starforge_cli-0.1.6.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,294 @@
1
+ # SFT Algorithm Configuration
2
+ sft:
3
+ ## total number of steps to train will equal
4
+ ## min((max_num_epochs * len(train_dataloader)), max_num_steps)
5
+ max_num_epochs: 1
6
+ max_num_steps: 60
7
+
8
+ val_period: 10
9
+ val_batches: 8
10
+ val_global_batch_size: 32
11
+ val_micro_batch_size: 1
12
+ val_at_start: true
13
+ val_at_end: false
14
+ seed: 42
15
+ only_unmask_final: false
16
+
17
+ checkpointing:
18
+ enabled: true
19
+ checkpoint_dir: "results/sft"
20
+ metric_name: "val:val_loss" # one of "val:" or "train:" followed by the metric name
21
+ higher_is_better: false
22
+ keep_top_k: 3
23
+ save_period: 10
24
+ checkpoint_must_save_by: null
25
+ save_optimizer: true
26
+ # pretrained_checkpoint: load weights from an existing megatron-bridge or
27
+ # Megatron-LM checkpoint instead of converting from HuggingFace.
28
+ # pretrained_checkpoint:
29
+ # path: "/path/to/checkpoint" # iter dir or checkpoint root
30
+ # format: "megatron_bridge" # or "megatron_lm"
31
+
32
+ policy:
33
+ model_name: "meta-llama/Llama-3.2-1B"
34
+ tokenizer:
35
+ name: ${policy.model_name} ## specify if you'd like to use a tokenizer different from the model's default
36
+ # chat_template can be a Jinja template string or path to a .jinja file
37
+ chat_template: "{% for message in messages %}{%- if message['role'] == 'system' %}{{'Context: ' + message['content'].strip()}}{%- elif message['role'] == 'user' %}{{' Question: ' + message['content'].strip() + ' Answer:'}}{%- elif message['role'] == 'assistant' %}{{' ' + message['content'].strip()}}{%- endif %}{% endfor %}"
38
+ chat_template_kwargs: null # can be used to pass kwargs to the chat template, e.g., enable_thinking=true
39
+ train_global_batch_size: 32
40
+ train_micro_batch_size: 1
41
+ max_total_sequence_length: 1024
42
+ precision: "bfloat16"
43
+
44
+ offload_optimizer_for_logprob: false
45
+
46
+ dtensor_cfg:
47
+ _v2: true
48
+ enabled: true
49
+ env_vars: {}
50
+ cpu_offload: False
51
+ sequence_parallel: false
52
+ activation_checkpointing: false
53
+ tensor_parallel_size: 1
54
+ context_parallel_size: 1
55
+ # HSDP replicate dimension size for Automodel DTensor v2.
56
+ # Typically set to the number of nodes; set to 1 to disable HSDP.
57
+ dp_replicate_size: 1
58
+ custom_parallel_plan: null
59
+
60
+ # Automodel kwargs passed to from_pretrained.
61
+ # Uncomment force_hf if the custom model's adapter doesn't support per-tensor
62
+ # weight conversion (e.g. Qwen2, Llama). Auto-detected if not set.
63
+ # See https://github.com/NVIDIA-NeMo/RL/issues/2072
64
+ automodel_kwargs: {}
65
+ # force_hf: true
66
+
67
+ # LoRA (Low-Rank Adaptation) Configuration
68
+ lora_cfg:
69
+ enabled: False # Set to True to enable LoRA fine-tuning
70
+ target_modules: [] # List of module names to apply LoRA (empty list with match_all_linear=true applies to all linear layers)
71
+ exclude_modules: [] # List of module names to exclude from LoRA
72
+ match_all_linear: true # If True, applies LoRA to all linear layers (overrides target_modules)
73
+ dim: 8 # LoRA rank (r): lower rank = fewer parameters but less capacity. Typical values: 4, 8, 16, 32, 64
74
+ alpha: 32 # LoRA scaling factor: effective learning rate multiplier = alpha/dim. Typical values: 16, 32, 64
75
+ dropout: 0.0 # Dropout probability applied to LoRA layers (0.0 = no dropout)
76
+ dropout_position: "post" # Where to apply dropout: "pre" (before LoRA) or "post" (after LoRA)
77
+ lora_A_init: "xavier" # Initialization method for LoRA A matrix: "xavier" or "uniform"
78
+ use_triton: true # Use Triton-optimized kernels for LoRA (faster but requires flash-attn). Disable when tensor_parallel_size > 1
79
+
80
+ dynamic_batching:
81
+ enabled: false
82
+ train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
83
+ sequence_length_round: 64
84
+
85
+ sequence_packing:
86
+ enabled: False
87
+ train_mb_tokens: ${mul:${policy.max_total_sequence_length}, ${policy.train_micro_batch_size}}
88
+ algorithm: "modified_first_fit_decreasing"
89
+ sequence_length_round: 64
90
+
91
+ # makes the training sequence length divisible by the tensor parallel size
92
+ # this is useful for sequence parallel training
93
+ make_sequence_length_divisible_by: ${policy.dtensor_cfg.tensor_parallel_size}
94
+ max_grad_norm: 1.0
95
+
96
+ optimizer:
97
+ name: "torch.optim.AdamW"
98
+ kwargs:
99
+ lr: 5.0e-6
100
+ weight_decay: 0.1
101
+ betas: [0.9, 0.98]
102
+ eps: 1e-5
103
+ # when using Dtensor, we need to set foreach
104
+ # and fused to False
105
+ foreach: False
106
+ fused: False
107
+
108
+ ## ignored since enabled=false, but needed for testing purposes
109
+ megatron_cfg:
110
+ enabled: false
111
+ use_linear_ce_fusion_loss: false
112
+ linear_ce_fusion_chunk_size: 256
113
+ force_reconvert_from_hf: False # Set to True to force reconvert of the model from Hugging Face
114
+ env_vars: {}
115
+ empty_unused_memory_level: 1
116
+ activation_checkpointing: false
117
+ # recompute_granularity controls activation checkpointing depth.
118
+ # "full": recompute all activations (default, max memory savings).
119
+ # "selective": recompute only specific modules (see recompute_modules).
120
+ # Only takes effect when activation_checkpointing: true.
121
+ recompute_granularity: "full"
122
+ # Modules to selectively recompute when recompute_granularity="selective".
123
+ # MCore options: ["core_attn", "moe_act", "layernorm", "mla_up_proj", "mlp", "moe", "shared_experts"].
124
+ # Null defaults to ["core_attn"]. Full list and per-module constraints:
125
+ # https://github.com/NVIDIA/Megatron-LM/blob/d30c3ae5469fe3f6a64d4fd2e63b6e7f7844ea81/megatron/core/transformer/transformer_config.py#L483
126
+ # Use ["moe"] for MoE models to recompute expert activations only.
127
+ recompute_modules: null
128
+ tensor_model_parallel_size: 1
129
+ expert_tensor_parallel_size: 1
130
+ expert_model_parallel_size: 1
131
+ pipeline_model_parallel_size: 1
132
+ context_parallel_size: 1
133
+ pipeline_dtype: ${policy.precision}
134
+ num_layers_in_first_pipeline_stage: null
135
+ num_layers_in_last_pipeline_stage: null
136
+ sequence_parallel: false
137
+ freeze_moe_router: false
138
+ moe_router_dtype: null
139
+ moe_router_load_balancing_type: "aux_loss"
140
+ moe_router_bias_update_rate: 1e-3
141
+ moe_permute_fusion: true
142
+ #gives ~20% training perf speedup with sequence packing
143
+ apply_rope_fusion: True
144
+ # gives ~25% training perf speedup with sequence packing and apply_rope_fusion
145
+ bias_activation_fusion: True
146
+ defer_fp32_logits: False
147
+ moe_per_layer_logging: False
148
+ moe_enable_deepep: false
149
+ moe_token_dispatcher_type: "alltoall"
150
+ moe_shared_expert_overlap: false
151
+ gradient_accumulation_fusion: false
152
+ use_fused_weighted_squared_relu: false
153
+
154
+ peft:
155
+ enabled: false
156
+ target_modules: []
157
+ exclude_modules: []
158
+ dim: 8
159
+ alpha: 32
160
+ dropout: 0.0
161
+ dropout_position: "post"
162
+ lora_A_init_method: "xavier"
163
+ lora_B_init_method: "zero"
164
+ a2a_experimental: false
165
+ lora_dtype: None
166
+
167
+
168
+ optimizer:
169
+ optimizer: "adam" # When weight decay is set, it actually uses AdamW
170
+ lr: 5.0e-6
171
+ min_lr: 4.9999e-6
172
+ weight_decay: 0.1 # When weight decay is set, it actually uses AdamW
173
+ bf16: true
174
+ fp16: false
175
+ params_dtype: "float32"
176
+
177
+ #adam
178
+ adam_beta1: 0.9
179
+ adam_beta2: 0.98
180
+ adam_eps: 1e-5
181
+
182
+ #sgd
183
+ sgd_momentum: 0.9
184
+
185
+ #distributed optimizer
186
+ use_distributed_optimizer: true
187
+ use_precision_aware_optimizer: true
188
+
189
+ clip_grad: ${policy.max_grad_norm}
190
+
191
+ # optimizer cpu offload
192
+ optimizer_cpu_offload: false
193
+ optimizer_offload_fraction: 0.0
194
+
195
+ scheduler:
196
+ start_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
197
+ end_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
198
+ weight_decay_incr_style: "constant"
199
+ lr_decay_style: "constant"
200
+ lr_decay_iters: 1000
201
+ lr_warmup_iters: 50
202
+ lr_warmup_init: 4.9999e-6
203
+
204
+ distributed_data_parallel_config:
205
+ grad_reduce_in_fp32: false
206
+ overlap_grad_reduce: true
207
+ overlap_param_gather: true
208
+ data_parallel_sharding_strategy: "optim_grads_params"
209
+ use_custom_fsdp: false
210
+
211
+ fp8_cfg:
212
+ enabled: false
213
+ fp8: "e4m3"
214
+ fp8_recipe: "blockwise"
215
+ fp8_param: false
216
+
217
+ data:
218
+ max_input_seq_length: ${policy.max_total_sequence_length}
219
+ add_bos: true
220
+ add_eos: true
221
+ add_generation_prompt: false
222
+ shuffle: true
223
+ num_workers: 1
224
+
225
+ # dataset
226
+ train:
227
+ dataset_name: "squad"
228
+ split: "train"
229
+ validation:
230
+ dataset_name: "squad"
231
+ split: "validation"
232
+ # default settings for all datasets
233
+ default:
234
+ prompt_file: null
235
+ system_prompt_file: null
236
+ processor: "sft_processor"
237
+
238
+ # You can also use multiple datasets by using a list of datasets.
239
+ # See `examples/configs/grpo_multiple_datasets.yaml` for a full configuration example.
240
+
241
+ # You can use custom response datasets for training and validation. For example:
242
+ # train:
243
+ # # this dataset will override input_key and use the default values for other vars
244
+ # data_path: /path/to/local/train_dataset.jsonl
245
+ # input_key: question
246
+ # validation:
247
+ # # this dataset will use the default values for other vars except data_path
248
+ # data_path: /path/to/local/val_dataset.jsonl
249
+ # default:
250
+ # # will use below vars as default values if dataset doesn't specify it
251
+ # dataset_name: ResponseDataset
252
+ # input_key: input
253
+ # output_key: output
254
+ # prompt_file: null
255
+ # system_prompt_file: null
256
+ # processor: "sft_processor"
257
+ # See https://github.com/NVIDIA-NeMo/RL/blob/main/docs/guides/sft.md#datasets for more details.
258
+
259
+ # OpenAI format specific configs
260
+ # train_data_path: "/path/to/train.jsonl" # Path to training data
261
+ # val_data_path: "/path/to/val.jsonl" # Path to validation data
262
+ # chat_key: "messages" # Key for messages in the data
263
+ # system_key: null # Key for system message (optional)
264
+ # system_prompt: null # Default system prompt (optional)
265
+ # tool_key: "tools" # Key for tools in the data
266
+ # use_preserving_dataset: false # If true, uses PreservingDataset to preserve heterogeneous schemas (e.g., tool calls with varying argument structures)
267
+
268
+ logger:
269
+ log_dir: "logs" # Base directory for all logs
270
+ wandb_enabled: true # Make sure you do a ``wandb login [Your API key]'' before running
271
+ tensorboard_enabled: true
272
+ mlflow_enabled: false
273
+ swanlab_enabled: false # Disable SwanLab logging
274
+ monitor_gpus: true # If true, will monitor GPU usage and log to wandb and/or tensorboard
275
+ wandb:
276
+ project: "sft-dev"
277
+ name: "sft-dev-${data.train.dataset_name}"
278
+ swanlab:
279
+ project: "sft-dev"
280
+ name: "sft-dev-${data.train.dataset_name}"
281
+ tensorboard:
282
+ log_dir: "tb_logs-sft-dev-${data.train.dataset_name}"
283
+ mlflow:
284
+ experiment_name: "sft-dev"
285
+ run_name: "sft-dev-${data.train.dataset_name}"
286
+ tracking_uri: "http://localhost:5000"
287
+ gpu_monitoring:
288
+ collection_interval: 10 # How often to collect GPU usage metrics (in seconds)
289
+ flush_interval: 10 # How often to flush GPU usage metrics to the loggers (in seconds)
290
+
291
+ cluster:
292
+ gpus_per_node: 1
293
+ num_nodes: 1
294
+ segment_size: null # Nodes per NVLink domain segment for topology-aware alignment; null to disable
@@ -0,0 +1,16 @@
1
+ # configs/models — 各基础模型的公共片段
2
+
3
+ 每个基础模型一个 partial 配置(只含该模型通用字段,如 `policy.model_name`、tokenizer、
4
+ 该模型适配的序列长度 / 并行 / 显存策略)。**不含 `defaults`**,作为实验「多继承」的一项使用。
5
+
6
+ 实验 `config.yaml` 里:
7
+
8
+ ```yaml
9
+ defaults:
10
+ - ../../configs/base/grpo_math_1B.yaml # 方法基底(官方 v0.6.0)
11
+ - ../../configs/models/qwen3.5-9b.yaml # 模型片段(覆盖基底里的模型字段)
12
+ # 下面再写本实验差异(数据集 / lr / kl / swanlab ...)
13
+ ```
14
+
15
+ 多继承中**后面的覆盖前面的**,实验自身的顶层键再覆盖两者。新增模型就加一个
16
+ `<model>.yaml`,命名与 `docs/naming-convention.md` 的 model 字段一致。
@@ -0,0 +1,12 @@
1
+ # 模型片段:qwen3.5-4b
2
+ # 作为实验「多继承」的一项,覆盖基底里的模型相关字段。本身不含 defaults,是个 partial。
3
+ # 用法(实验 config.yaml):
4
+ # defaults:
5
+ # - ../../configs/base/grpo_math_1B.yaml
6
+ # - ../../configs/models/qwen3.5-4b.yaml
7
+ policy:
8
+ model_name: "Qwen/Qwen3.5-4B"
9
+ tokenizer:
10
+ name: "Qwen/Qwen3.5-4B"
11
+ # 该模型常用的默认值可放这里(实验仍可再覆盖)
12
+ # max_total_sequence_length: 4096
@@ -0,0 +1,10 @@
1
+ # 模型片段:qwen3.5-9b
2
+ # 作为实验「多继承」的一项,覆盖基底里的模型相关字段。本身不含 defaults,是个 partial。
3
+ # 注:你实测 GRPO 用的是 base 模型 "Qwen/Qwen3.5-9B-Base";如需从 base 起强化训练,改下面 model_name。
4
+ policy:
5
+ model_name: "Qwen/Qwen3.5-9B"
6
+ tokenizer:
7
+ name: "Qwen/Qwen3.5-9B"
8
+ # 9B 低显存策略(实测):
9
+ logprob_chunk_size: 256 # 分块算 logprob,省显存
10
+ offload_optimizer_for_logprob: true # 非 colocated 生成时把优化器 offload 到 CPU(colocated 下为 no-op)
@@ -0,0 +1,11 @@
1
+ # StarForge 项目
2
+ outputs/
3
+ datasets/
4
+ forge_plugins/
5
+ __pycache__/
6
+ *.pyc
7
+ .venv/
8
+ .env
9
+ *.pem
10
+ id_rsa*
11
+ .DS_Store