starforge-cli 0.1.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- starforge_cli/__init__.py +3 -0
- starforge_cli/api_client.py +589 -0
- starforge_cli/auth.py +349 -0
- starforge_cli/catalog.py +124 -0
- starforge_cli/cli.py +74 -0
- starforge_cli/cli_ui.py +469 -0
- starforge_cli/client_device.py +104 -0
- starforge_cli/commands/__init__.py +1 -0
- starforge_cli/commands/admin.py +140 -0
- starforge_cli/commands/bench.py +94 -0
- starforge_cli/commands/common.py +178 -0
- starforge_cli/commands/dataset.py +150 -0
- starforge_cli/commands/exp.py +213 -0
- starforge_cli/commands/init.py +52 -0
- starforge_cli/commands/jobs.py +223 -0
- starforge_cli/commands/login.py +54 -0
- starforge_cli/commands/plugin.py +243 -0
- starforge_cli/commands/recipe.py +163 -0
- starforge_cli/commands/serve.py +79 -0
- starforge_cli/commands/submit.py +467 -0
- starforge_cli/commands/sweep.py +154 -0
- starforge_cli/config_resolve.py +17 -0
- starforge_cli/data_prep.py +60 -0
- starforge_cli/new_experiment.py +195 -0
- starforge_cli/packing.py +179 -0
- starforge_cli/plugins_lock.py +73 -0
- starforge_cli/project.py +130 -0
- starforge_cli/recipe_lock.py +453 -0
- starforge_cli/scaffold/agent-run.py.tmpl +146 -0
- starforge_cli/scaffold/custom-framework/train.sh +56 -0
- starforge_cli/scaffold/experiment-template/.gitkeep +0 -0
- starforge_cli/scaffold/experiment-template/README.md +36 -0
- starforge_cli/scaffold/experiment-template/config.yaml +44 -0
- starforge_cli/scaffold/project/common/README.md +12 -0
- starforge_cli/scaffold/project/common/__init__.py +0 -0
- starforge_cli/scaffold/project/configs/README.md +103 -0
- starforge_cli/scaffold/project/configs/base/README.md +24 -0
- starforge_cli/scaffold/project/configs/base/distillation_math.yaml +284 -0
- starforge_cli/scaffold/project/configs/base/grpo_lora.yaml +30 -0
- starforge_cli/scaffold/project/configs/base/grpo_math_1B.yaml +470 -0
- starforge_cli/scaffold/project/configs/base/grpo_megatron.yaml +43 -0
- starforge_cli/scaffold/project/configs/base/grpo_noncolocated.yaml +18 -0
- starforge_cli/scaffold/project/configs/base/grpo_sliding_puzzle.yaml +81 -0
- starforge_cli/scaffold/project/configs/base/ppo_math_1B.yaml +454 -0
- starforge_cli/scaffold/project/configs/base/rm.yaml +224 -0
- starforge_cli/scaffold/project/configs/base/sft.yaml +294 -0
- starforge_cli/scaffold/project/configs/models/README.md +16 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-4b.yaml +12 -0
- starforge_cli/scaffold/project/configs/models/qwen3.5-9b.yaml +10 -0
- starforge_cli/scaffold/project/gitignore +11 -0
- starforge_cli/spec_builder.py +372 -0
- starforge_cli-0.1.6.dist-info/METADATA +40 -0
- starforge_cli-0.1.6.dist-info/RECORD +55 -0
- starforge_cli-0.1.6.dist-info/WHEEL +4 -0
- starforge_cli-0.1.6.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# <method>_<model>_<dataset>_<tag>
|
|
2
|
+
|
|
3
|
+
> 复制本模板新建实验:`sf new <新实验名> --method <framework>/<method>`
|
|
4
|
+
> 实验名遵循 `docs/naming-convention.md`。
|
|
5
|
+
|
|
6
|
+
## 目标
|
|
7
|
+
|
|
8
|
+
一句话说明这个实验要验证 / 达成什么。
|
|
9
|
+
|
|
10
|
+
## 配置(NeMo-RL 0.6.0,配置继承)
|
|
11
|
+
|
|
12
|
+
- `config.yaml` 通过 `defaults` 继承基底(`configs/base/`)+ 模型片段(`configs/models/`),
|
|
13
|
+
**只写本实验差异**;不断调参就改 `config.yaml` 的「本实验差异」部分。
|
|
14
|
+
- 训练入口、指标与产物契约由 SDK 中的版本化 recipe 声明;实验目录不写 `framework`
|
|
15
|
+
标记,也不按 `run.py`/`train.sh` 是否存在猜测框架。
|
|
16
|
+
- 硬件与资源:提交时 `sf submit --profile 名称[:总卡数]` 一个参数说清(如 `h200`、`h200:4`);
|
|
17
|
+
卡型、默认形状、env/overrides 均由 Console 服务端注册表下发。
|
|
18
|
+
|
|
19
|
+
## 监控
|
|
20
|
+
|
|
21
|
+
- project:`<实验名>`
|
|
22
|
+
- run:`<超参组合>`
|
|
23
|
+
- 链接:<贴上监控面板链接>
|
|
24
|
+
|
|
25
|
+
## 运行
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
uv run sf submit <实验名> --profile h100
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
产物(checkpoint / 日志)落到本目录 `outputs/`(已 .gitignore)。
|
|
32
|
+
|
|
33
|
+
## 结果与结论
|
|
34
|
+
|
|
35
|
+
- 关键指标:
|
|
36
|
+
- 结论 / 下一步:
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# 实验配置:继承基底 + 模型片段,只写本实验差异(NeMo-RL 0.6.0 原生 defaults 机制)。
|
|
2
|
+
# 调参就改下面「本实验差异」部分;基底不动。
|
|
3
|
+
#
|
|
4
|
+
# 切换方法:
|
|
5
|
+
# - SFT : 使用 sft recipe 模板;入口由 recipe 固定为 examples/run_sft.py
|
|
6
|
+
# - 多轮 Agent: 换成 ../../configs/base/grpo_sliding_puzzle.yaml(已含 max_rollout_turns/env 范例)
|
|
7
|
+
defaults:
|
|
8
|
+
- ../../configs/base/grpo_math_1B.yaml
|
|
9
|
+
- ../../configs/models/qwen3.5-4b.yaml
|
|
10
|
+
|
|
11
|
+
# ===================== 本实验差异 =====================
|
|
12
|
+
policy:
|
|
13
|
+
max_total_sequence_length: 4096
|
|
14
|
+
optimizer:
|
|
15
|
+
kwargs:
|
|
16
|
+
lr: 1.0e-6
|
|
17
|
+
|
|
18
|
+
grpo:
|
|
19
|
+
num_prompts_per_step: 32
|
|
20
|
+
num_generations_per_prompt: 16
|
|
21
|
+
max_rollout_turns: 1 # 多轮 Agent 设 >1
|
|
22
|
+
|
|
23
|
+
loss_fn:
|
|
24
|
+
reference_policy_kl_penalty: 0.01
|
|
25
|
+
|
|
26
|
+
# 数据集(按实际填写,见 common/data/README.md 与官方 docs/guides/grpo.md)
|
|
27
|
+
# data:
|
|
28
|
+
# train:
|
|
29
|
+
# data_path: /abs/path/train.jsonl
|
|
30
|
+
# input_key: question
|
|
31
|
+
# default:
|
|
32
|
+
# dataset_name: ResponseDataset
|
|
33
|
+
# env_name: math
|
|
34
|
+
|
|
35
|
+
logger:
|
|
36
|
+
swanlab_enabled: true
|
|
37
|
+
swanlab:
|
|
38
|
+
project: "<method>_<model>_<dataset>_<tag>" # 对齐实验目录名
|
|
39
|
+
name: "<超参组合>" # 如 lr1e6-g16-kl0.01
|
|
40
|
+
monitor_gpus: true
|
|
41
|
+
# =====================================================
|
|
42
|
+
# 注:checkpoint 目录与 log 目录由统一 launcher 指到本实验 outputs/(已 .gitignore),
|
|
43
|
+
# 集群/硬件相关(节点数、并行度)由服务端 profile 注册表在运行时叠加(FORGE_PROFILE_OVERRIDES)。
|
|
44
|
+
# 提交时 sf submit --profile 指定目标集群;下列数值应按该卡显存调。
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# common/ — 项目共享代码
|
|
2
|
+
|
|
3
|
+
实验通过 `common.*` 引用的自定义代码放这里,随作业包上传到集群:
|
|
4
|
+
|
|
5
|
+
| 目录 | 放什么 | 何时需要 |
|
|
6
|
+
| --- | --- | --- |
|
|
7
|
+
| `common/data/` | 数据预处理脚本 `prepare_*.py`(`sf dataset prepare` 自动发现) | 需要预处理内部数据时 |
|
|
8
|
+
| `common/environments/` | 多轮 Agent 环境(工具调用 / 可验证任务) | Agent RL 实验 |
|
|
9
|
+
| `common/rewards/` | 自定义奖励函数(关键词 / LLM 裁判) | RL 实验需要自定义奖励时 |
|
|
10
|
+
|
|
11
|
+
纯 SFT / 数学 GRPO 实验用官方基底即可,不需要写 common 代码。
|
|
12
|
+
Agent 环境与裁判/沙箱的写法见平台文档「Agent 环境套件」。
|
|
File without changes
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# configs/ — 配置继承体系
|
|
2
|
+
|
|
3
|
+
NeMo-RL 0.7.0 **原生支持配置继承**(`nemo_rl/utils/config.py` 的 `load_config`):配置里写
|
|
4
|
+
`defaults: parent.yaml` 即可继承,支持多继承、嵌套继承、`${}` 插值、`_override_` 整段覆盖,
|
|
5
|
+
再叠加命令行 Hydra override。官方自己就这么用(`grpo_sliding_puzzle.yaml` 继承 `grpo_math_1B.yaml`)。
|
|
6
|
+
|
|
7
|
+
所以每个模型 / 每个实验都有自己的配置、不断调参,但**只写差异**,公共部分继承。
|
|
8
|
+
|
|
9
|
+
## 三层结构
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
configs/
|
|
13
|
+
├── base/ 祖父:官方 v0.7.0 example 原样副本(version-locked,升级时从 NeMo-RL 源码目录整体拷入,勿手改)
|
|
14
|
+
│ ├── grpo_math_1B.yaml # GRPO 基底
|
|
15
|
+
│ ├── sft.yaml # SFT 基底
|
|
16
|
+
│ ├── grpo_sliding_puzzle.yaml # 多轮 Agent 基底(已继承 grpo_math_1B)
|
|
17
|
+
│ ├── grpo_megatron.yaml # ★自定义 overlay:Megatron 后端 + 低显存实测显存/性能调优
|
|
18
|
+
│ ├── grpo_lora.yaml # ★自定义 overlay:LoRA/PEFT(低显存机型上让 9B 能跑起来)
|
|
19
|
+
│ └── grpo_noncolocated.yaml # ★自定义 overlay:非 colocated 生成(1 卡生成 / 1 卡训练)
|
|
20
|
+
└── models/ 父:各基础模型的公共片段(model_name / tokenizer / 显存策略…)
|
|
21
|
+
├── qwen3.5-4b.yaml
|
|
22
|
+
└── qwen3.5-9b.yaml
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## 训练后端 + LoRA(两个 overlay)
|
|
26
|
+
|
|
27
|
+
本仓库 GRPO 实验默认 **Megatron-Core + LoRA**,靠两个 overlay 叠加(都来自低显存机型上跑通的配置):
|
|
28
|
+
|
|
29
|
+
```yaml
|
|
30
|
+
defaults:
|
|
31
|
+
- ../../configs/base/grpo_math_1B.yaml
|
|
32
|
+
- ../../configs/models/qwen3.5-9b.yaml
|
|
33
|
+
- ../../configs/base/grpo_megatron.yaml # ① Megatron 后端 + 低显存显存/性能调优
|
|
34
|
+
- ../../configs/base/grpo_lora.yaml # ② LoRA(lr 1e-4 / wd 0 / cosine)
|
|
35
|
+
policy:
|
|
36
|
+
max_total_sequence_length: 1250
|
|
37
|
+
train_global_batch_size: 32 # = num_prompts_per_step * num_generations_per_prompt
|
|
38
|
+
grpo:
|
|
39
|
+
num_prompts_per_step: 4
|
|
40
|
+
num_generations_per_prompt: 8
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
- `grpo_megatron.yaml`:关 DTensor、开 Megatron,置空 FSDP 的 `policy.optimizer/scheduler`,并带上低显存实测显存项(`activation_checkpointing` / `empty_unused_memory_level=2` / `apply_rope_fusion=false` / `defer_fp32_logits` / `enforce_eager` / 关 sequence packing 等)。
|
|
44
|
+
- `grpo_lora.yaml`:开 `megatron_cfg.peft`,LoRA 学习率写在 `megatron_cfg.optimizer.lr`(1e-4,比全参数高 2 个量级)。
|
|
45
|
+
- `grpo_noncolocated.yaml`(可选第三层):生成与训练各占独立 GPU(实测 9B 用法)。2 卡场景 = 1 卡生成 / 1 卡训练 → 训练侧 PP=1。删此行即回 colocated(共用 GPU)。
|
|
46
|
+
|
|
47
|
+
切回方式:
|
|
48
|
+
- **回全参数**:删 `grpo_lora.yaml` 一行,并把 lr 写到 `policy.megatron_cfg.optimizer.lr`(如 1e-6)。
|
|
49
|
+
- **回 DTensor/FSDP**:删 `grpo_megatron.yaml` 一行,并把 lr 写回 `policy.optimizer.kwargs.lr`。
|
|
50
|
+
|
|
51
|
+
> 并行度(TP/PP/CP)放 `cluster/<profile>/overrides.conf`;9B 实测 PP=1,4B 实测 PP=2。
|
|
52
|
+
> `converter_type` 无需按模型改:Megatron 用 AutoBridge 按 HF 架构自动识别。
|
|
53
|
+
> 显存类调优放在 overlay(merge)而非 `overrides.conf`:CLI override 是 struct 模式且对 SFT 也生效,而 SFT 没有 `policy.generation` 等键。
|
|
54
|
+
|
|
55
|
+
实验目录里的 `config.yaml` 是**子**:
|
|
56
|
+
|
|
57
|
+
```yaml
|
|
58
|
+
defaults:
|
|
59
|
+
- ../../configs/base/grpo_math_1B.yaml # 方法基底
|
|
60
|
+
- ../../configs/models/qwen3.5-9b.yaml # 模型片段
|
|
61
|
+
# 下面只写本实验差异:数据集 / lr / kl / swanlab / 步数 ...
|
|
62
|
+
grpo:
|
|
63
|
+
num_generations_per_prompt: 16
|
|
64
|
+
loss_fn:
|
|
65
|
+
reference_policy_kl_penalty: 0.01
|
|
66
|
+
logger:
|
|
67
|
+
swanlab_enabled: true
|
|
68
|
+
swanlab: { project: "grpo_qwen3.5-9b_gsm8k_v1", name: "lr1e6-g16-kl0.01" }
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
> 合并规则:多继承中**后面覆盖前面**,实验自身再覆盖基底/模型。需要整段替换(不合并)时,
|
|
72
|
+
> 在该段加 `_override_: true`。
|
|
73
|
+
|
|
74
|
+
## 方法 → 入口 → 基底
|
|
75
|
+
|
|
76
|
+
| 方法 | `ENTRY`(run.sh) | 基底(defaults 第一项) |
|
|
77
|
+
| --- | --- | --- |
|
|
78
|
+
| SFT | `examples/run_sft.py` | `../../configs/base/sft.yaml` |
|
|
79
|
+
| GRPO | `examples/run_grpo.py` | `../../configs/base/grpo_math_1B.yaml` |
|
|
80
|
+
| 多轮 Agent | `examples/run_grpo.py` | `../../configs/base/grpo_sliding_puzzle.yaml` |
|
|
81
|
+
|
|
82
|
+
> 更大模型可加 `grpo_math_8B.yaml` 等官方基底:把文件名加进 `scripts/sync_base_configs.sh` 同步。
|
|
83
|
+
|
|
84
|
+
## 集群 / 硬件
|
|
85
|
+
|
|
86
|
+
不进 `config.yaml`,由 `cluster/<profile>/overrides.conf` 在运行时以 CLI override 叠加(见 `cluster/README.md`),
|
|
87
|
+
这样切硬件(H100 ↔ H200)不动实验配置。
|
|
88
|
+
|
|
89
|
+
## 常用字段(0.7.0)
|
|
90
|
+
|
|
91
|
+
| 作用 | key |
|
|
92
|
+
| --- | --- |
|
|
93
|
+
| 基础模型 | `policy.model_name` |
|
|
94
|
+
| 序列长度 | `policy.max_total_sequence_length` |
|
|
95
|
+
| 学习率(Megatron/LoRA,默认) | `policy.megatron_cfg.optimizer.lr`(LoRA 用 1e-4) |
|
|
96
|
+
| 学习率(DTensor 后端) | `policy.optimizer.kwargs.lr` |
|
|
97
|
+
| LoRA 开关 / 秩 | `policy.megatron_cfg.peft.enabled` / `.dim` / `.alpha` |
|
|
98
|
+
| 全局 batch(须整除 prompts×gen) | `policy.train_global_batch_size` |
|
|
99
|
+
| KL 惩罚 | `loss_fn.reference_policy_kl_penalty` |
|
|
100
|
+
| 每步 prompt 数 / 每 prompt 采样数 | `grpo.num_prompts_per_step` / `grpo.num_generations_per_prompt` |
|
|
101
|
+
| 多轮轮数 | `grpo.max_rollout_turns` |
|
|
102
|
+
| 节点 / 卡数 | `cluster.num_nodes` / `cluster.gpus_per_node` |
|
|
103
|
+
| 启用 SwanLab | `logger.swanlab_enabled` + `logger.swanlab.{project,name}` |
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# configs/base — 供继承的基底配置
|
|
2
|
+
|
|
3
|
+
这里是 **NeMo-RL v0.7.0 官方 example 配置的原样副本**(version-locked),作为所有实验
|
|
4
|
+
`defaults` 继承的「祖父配置」。实验只写差异,调参不动这里。
|
|
5
|
+
|
|
6
|
+
| 文件 | 来源(v0.7.0) | 用途 |
|
|
7
|
+
| --- | --- | --- |
|
|
8
|
+
| `grpo_math_1B.yaml` | `examples/configs/grpo_math_1B.yaml` | GRPO 基底(最常用的祖父) |
|
|
9
|
+
| `sft.yaml` | `examples/configs/sft.yaml` | SFT 基底 |
|
|
10
|
+
| `grpo_sliding_puzzle.yaml` | `examples/configs/grpo_sliding_puzzle.yaml` | 多轮 Agent 基底(本身 `defaults: grpo_math_1B.yaml`) |
|
|
11
|
+
|
|
12
|
+
> 官方就是这么用继承的:`grpo_sliding_puzzle.yaml` 第一行即 `defaults: "grpo_math_1B.yaml"`,
|
|
13
|
+
> 只覆盖 `grpo.max_rollout_turns`、`env` 等差异。
|
|
14
|
+
|
|
15
|
+
## 更新基底(升级 NeMo-RL 版本时)
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
# 从本地 NeMo-RL 源码把 example 配置原样拷入(升级是低频运维操作,直接 cp,不做 CLI 包装)
|
|
19
|
+
cp /path/to/NeMo-RL/examples/configs/grpo_math_1B.yaml configs/base/
|
|
20
|
+
cp /path/to/NeMo-RL/examples/configs/sft.yaml configs/base/
|
|
21
|
+
cp /path/to/NeMo-RL/examples/configs/grpo_sliding_puzzle.yaml configs/base/
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
**不要手改这些文件**——要调参请在实验的 `config.yaml` 里覆盖。
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
# Distillation Algorithm Configuration
|
|
2
|
+
distillation:
|
|
3
|
+
num_prompts_per_step: 128
|
|
4
|
+
num_generations_per_prompt: 1
|
|
5
|
+
max_rollout_turns: 1 # for multi-turn rollouts. Math Environments just have 1 turn (answering the question)
|
|
6
|
+
max_num_steps: 1000
|
|
7
|
+
max_num_epochs: 10
|
|
8
|
+
val_batch_size: 64
|
|
9
|
+
val_period: 20
|
|
10
|
+
val_at_start: false
|
|
11
|
+
val_at_end: false
|
|
12
|
+
max_val_samples: 512
|
|
13
|
+
topk_logits_k: 64
|
|
14
|
+
seed: 42
|
|
15
|
+
|
|
16
|
+
loss_fn:
|
|
17
|
+
kl_type: "mixed" # forward, reverse, mixed
|
|
18
|
+
mixed_kl_weight: 0.5 # when kl_type is "mixed", this is the weight of the forward KL
|
|
19
|
+
zero_outside_topk: false # zero out the teacher logits outside the top k when calculate forward KL loss
|
|
20
|
+
|
|
21
|
+
checkpointing:
|
|
22
|
+
enabled: true
|
|
23
|
+
checkpoint_dir: "checkpoints/distillation-${policy.model_name}"
|
|
24
|
+
metric_name: "val:accuracy" # one of "val:" or "train:" followed by the metric name
|
|
25
|
+
higher_is_better: true
|
|
26
|
+
keep_top_k: 3
|
|
27
|
+
save_period: 10
|
|
28
|
+
checkpoint_must_save_by: null
|
|
29
|
+
model_save_format: "safetensors"
|
|
30
|
+
save_consolidated: false
|
|
31
|
+
save_optimizer: true
|
|
32
|
+
|
|
33
|
+
policy: &POLICY_BASE
|
|
34
|
+
model_name: "Qwen/Qwen3-1.7B-Base"
|
|
35
|
+
tokenizer:
|
|
36
|
+
name: ${..model_name} ## specify if you'd like to use a tokenizer different from the model's default
|
|
37
|
+
chat_template_kwargs: null # can be used to pass kwargs to the chat template, e.g., enable_thinking=true
|
|
38
|
+
train_global_batch_size: 64
|
|
39
|
+
train_micro_batch_size: 1
|
|
40
|
+
generation_batch_size: 64
|
|
41
|
+
logprob_batch_size: 1
|
|
42
|
+
max_total_sequence_length: 8192
|
|
43
|
+
precision: "bfloat16"
|
|
44
|
+
logprob_chunk_size: null
|
|
45
|
+
|
|
46
|
+
offload_optimizer_for_logprob: false
|
|
47
|
+
|
|
48
|
+
dtensor_cfg: &DTENSOR_BASE
|
|
49
|
+
enabled: true
|
|
50
|
+
_v2: true
|
|
51
|
+
cpu_offload: False
|
|
52
|
+
sequence_parallel: false
|
|
53
|
+
activation_checkpointing: true
|
|
54
|
+
tensor_parallel_size: 2
|
|
55
|
+
context_parallel_size: 2
|
|
56
|
+
custom_parallel_plan: null
|
|
57
|
+
|
|
58
|
+
dynamic_batching:
|
|
59
|
+
enabled: true
|
|
60
|
+
train_mb_tokens: ${mul:${..max_total_sequence_length}, ${..train_micro_batch_size}}
|
|
61
|
+
logprob_mb_tokens: ${mul:${..max_total_sequence_length}, ${..logprob_batch_size}}
|
|
62
|
+
sequence_length_round: 64
|
|
63
|
+
|
|
64
|
+
sequence_packing:
|
|
65
|
+
enabled: false
|
|
66
|
+
train_mb_tokens: ${mul:${..max_total_sequence_length}, ${..train_micro_batch_size}}
|
|
67
|
+
logprob_mb_tokens: ${mul:${..max_total_sequence_length}, ${..logprob_batch_size}}
|
|
68
|
+
algorithm: "modified_first_fit_decreasing"
|
|
69
|
+
sequence_length_round: 64
|
|
70
|
+
|
|
71
|
+
max_grad_norm: 1.0
|
|
72
|
+
# makes the training sequence length divisible by the tensor parallel size
|
|
73
|
+
# this is useful for sequence parallel training
|
|
74
|
+
# must be divisible by 2*cp
|
|
75
|
+
make_sequence_length_divisible_by: ${mul:${mul:${.dtensor_cfg.tensor_parallel_size}, ${.dtensor_cfg.context_parallel_size}}, 2}
|
|
76
|
+
optimizer:
|
|
77
|
+
name: "torch.optim.AdamW"
|
|
78
|
+
kwargs:
|
|
79
|
+
lr: 2.0e-5
|
|
80
|
+
weight_decay: 0.01
|
|
81
|
+
betas: [0.9, 0.999]
|
|
82
|
+
eps: 1e-8
|
|
83
|
+
# when using Dtensor, we need to set foreach
|
|
84
|
+
# and fused to False
|
|
85
|
+
foreach: False
|
|
86
|
+
fused: False
|
|
87
|
+
|
|
88
|
+
megatron_cfg: &MEGATRON_BASE
|
|
89
|
+
enabled: false
|
|
90
|
+
force_reconvert_from_hf: False # Set to True to force reconvert of the model from Hugging Face
|
|
91
|
+
empty_unused_memory_level: 0
|
|
92
|
+
activation_checkpointing: false
|
|
93
|
+
# recompute_granularity controls activation checkpointing depth.
|
|
94
|
+
# "full": recompute all activations (default, max memory savings).
|
|
95
|
+
# "selective": recompute only specific modules (see recompute_modules).
|
|
96
|
+
# Only takes effect when activation_checkpointing: true.
|
|
97
|
+
recompute_granularity: "full"
|
|
98
|
+
# Modules to selectively recompute when recompute_granularity="selective".
|
|
99
|
+
# MCore options: ["core_attn", "moe_act", "layernorm", "mla_up_proj", "mlp", "moe", "shared_experts"].
|
|
100
|
+
# Null defaults to ["core_attn"]. Full list and per-module constraints:
|
|
101
|
+
# https://github.com/NVIDIA/Megatron-LM/blob/d30c3ae5469fe3f6a64d4fd2e63b6e7f7844ea81/megatron/core/transformer/transformer_config.py#L483
|
|
102
|
+
# Use ["moe"] for MoE models to recompute expert activations only.
|
|
103
|
+
recompute_modules: null
|
|
104
|
+
tensor_model_parallel_size: 2
|
|
105
|
+
expert_tensor_parallel_size: 1
|
|
106
|
+
expert_model_parallel_size: 1
|
|
107
|
+
pipeline_model_parallel_size: 2
|
|
108
|
+
num_layers_in_first_pipeline_stage: null
|
|
109
|
+
num_layers_in_last_pipeline_stage: null
|
|
110
|
+
context_parallel_size: 2
|
|
111
|
+
pipeline_dtype: ${policy.precision}
|
|
112
|
+
sequence_parallel: false
|
|
113
|
+
freeze_moe_router: true
|
|
114
|
+
moe_router_dtype: "fp64"
|
|
115
|
+
moe_router_load_balancing_type: "none" # "seq_aux_loss" causes logprob error divergence for grpo
|
|
116
|
+
moe_router_bias_update_rate: 0.0 # by default, disable bias updates for grpo
|
|
117
|
+
moe_permute_fusion: true
|
|
118
|
+
#gives ~20% training perf speedup with sequence packing
|
|
119
|
+
apply_rope_fusion: True
|
|
120
|
+
bias_activation_fusion: True
|
|
121
|
+
defer_fp32_logits: False
|
|
122
|
+
moe_per_layer_logging: False
|
|
123
|
+
moe_enable_deepep: false
|
|
124
|
+
moe_token_dispatcher_type: "alltoall"
|
|
125
|
+
moe_shared_expert_overlap: false
|
|
126
|
+
gradient_accumulation_fusion: false
|
|
127
|
+
use_fused_weighted_squared_relu: false
|
|
128
|
+
optimizer:
|
|
129
|
+
optimizer: "adam"
|
|
130
|
+
lr: 2.00001e-5
|
|
131
|
+
min_lr: 2.0e-5
|
|
132
|
+
weight_decay: 0.01
|
|
133
|
+
bf16: true
|
|
134
|
+
fp16: false
|
|
135
|
+
params_dtype: "float32"
|
|
136
|
+
|
|
137
|
+
#adam
|
|
138
|
+
adam_beta1: 0.9
|
|
139
|
+
adam_beta2: 0.999
|
|
140
|
+
adam_eps: 1e-8
|
|
141
|
+
|
|
142
|
+
#sgd
|
|
143
|
+
sgd_momentum: 0.9
|
|
144
|
+
|
|
145
|
+
#distributed optimizer
|
|
146
|
+
use_distributed_optimizer: true
|
|
147
|
+
use_precision_aware_optimizer: true
|
|
148
|
+
|
|
149
|
+
# optimizer cpu offload
|
|
150
|
+
optimizer_cpu_offload: false
|
|
151
|
+
optimizer_offload_fraction: 0.0
|
|
152
|
+
|
|
153
|
+
clip_grad: ${policy.max_grad_norm}
|
|
154
|
+
|
|
155
|
+
scheduler:
|
|
156
|
+
start_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
157
|
+
end_weight_decay: ${policy.megatron_cfg.optimizer.weight_decay}
|
|
158
|
+
weight_decay_incr_style: "constant"
|
|
159
|
+
lr_decay_style: "constant"
|
|
160
|
+
lr_decay_iters: 1000
|
|
161
|
+
lr_warmup_iters: 10
|
|
162
|
+
lr_warmup_init: 2.0e-6
|
|
163
|
+
|
|
164
|
+
distributed_data_parallel_config:
|
|
165
|
+
grad_reduce_in_fp32: false
|
|
166
|
+
overlap_grad_reduce: true
|
|
167
|
+
overlap_param_gather: true
|
|
168
|
+
use_custom_fsdp: false
|
|
169
|
+
data_parallel_sharding_strategy: "optim_grads_params"
|
|
170
|
+
|
|
171
|
+
fp8_cfg:
|
|
172
|
+
enabled: false
|
|
173
|
+
fp8: "e4m3"
|
|
174
|
+
fp8_recipe: "blockwise"
|
|
175
|
+
fp8_param: false
|
|
176
|
+
|
|
177
|
+
scheduler:
|
|
178
|
+
- name: "torch.optim.lr_scheduler.LinearLR"
|
|
179
|
+
kwargs:
|
|
180
|
+
start_factor: 0.1
|
|
181
|
+
end_factor: 1.0
|
|
182
|
+
total_iters: 10
|
|
183
|
+
- name: "torch.optim.lr_scheduler.ConstantLR"
|
|
184
|
+
kwargs:
|
|
185
|
+
factor: 1.0
|
|
186
|
+
total_iters: 10000000000
|
|
187
|
+
- milestones: [10]
|
|
188
|
+
|
|
189
|
+
generation:
|
|
190
|
+
# Port range for vLLM HTTP servers, kept below the OS ephemeral range.
|
|
191
|
+
# See ray.sub for the full port layout.
|
|
192
|
+
port_range_low: 11001
|
|
193
|
+
port_range_high: 15000
|
|
194
|
+
backend: "vllm"
|
|
195
|
+
max_new_tokens: ${..max_total_sequence_length} # refer to local policy/teacher config
|
|
196
|
+
temperature: 1.0
|
|
197
|
+
top_p: 1.0
|
|
198
|
+
top_k: null
|
|
199
|
+
stop_token_ids: null
|
|
200
|
+
stop_strings: null
|
|
201
|
+
vllm_cfg:
|
|
202
|
+
async_engine: false
|
|
203
|
+
precision: ${...precision}
|
|
204
|
+
kv_cache_dtype: "auto"
|
|
205
|
+
tensor_parallel_size: 1
|
|
206
|
+
pipeline_parallel_size: 1
|
|
207
|
+
expert_parallel_size: 1 # When EP > 1, EP must be a multiple of TP since vLLM's EP = DP * TP
|
|
208
|
+
gpu_memory_utilization: 0.6
|
|
209
|
+
max_model_len: ${...max_total_sequence_length} # refer to local policy/teacher config
|
|
210
|
+
enforce_eager: False
|
|
211
|
+
use_deep_gemm: False
|
|
212
|
+
num_last_layers_in_bf16: 0
|
|
213
|
+
num_first_layers_in_bf16: 0
|
|
214
|
+
distributed_executor_backend: null
|
|
215
|
+
vllm_kwargs: {}
|
|
216
|
+
|
|
217
|
+
colocated:
|
|
218
|
+
# true: generation shares training GPUs
|
|
219
|
+
# false: uses dedicated generation resources
|
|
220
|
+
enabled: true
|
|
221
|
+
# only relevant when enabled is false
|
|
222
|
+
resources:
|
|
223
|
+
gpus_per_node: null # Decides num gpus to be dedicated to generation when there is one node in the cluster i.e cluster.num_nodes == 1
|
|
224
|
+
num_nodes: null # Decides number of nodes to be dedicated to generation
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
teacher:
|
|
228
|
+
<<: *POLICY_BASE
|
|
229
|
+
model_name: "Qwen/Qwen3-4B"
|
|
230
|
+
dtensor_cfg:
|
|
231
|
+
<<: *DTENSOR_BASE
|
|
232
|
+
context_parallel_size: 2
|
|
233
|
+
tensor_parallel_size: 4
|
|
234
|
+
|
|
235
|
+
data:
|
|
236
|
+
max_input_seq_length: ${policy.max_total_sequence_length} # upper bound, real truncation occurs at vllm.max_model_len
|
|
237
|
+
shuffle: true
|
|
238
|
+
|
|
239
|
+
# dataset
|
|
240
|
+
train:
|
|
241
|
+
dataset_name: DeepScaler
|
|
242
|
+
validation:
|
|
243
|
+
dataset_name: AIME2024
|
|
244
|
+
repeat: 16
|
|
245
|
+
# default settings for all datasets
|
|
246
|
+
default:
|
|
247
|
+
prompt_file: "examples/prompts/cot.txt"
|
|
248
|
+
system_prompt_file: null
|
|
249
|
+
env_name: "math"
|
|
250
|
+
|
|
251
|
+
env:
|
|
252
|
+
math:
|
|
253
|
+
num_workers: 8
|
|
254
|
+
|
|
255
|
+
logger:
|
|
256
|
+
log_dir: "logs/distillation"
|
|
257
|
+
num_val_samples_to_print: 5
|
|
258
|
+
wandb_enabled: true
|
|
259
|
+
tensorboard_enabled: true
|
|
260
|
+
mlflow_enabled: false
|
|
261
|
+
swanlab_enabled: false
|
|
262
|
+
monitor_gpus: true
|
|
263
|
+
wandb:
|
|
264
|
+
project: "nemo-distillation"
|
|
265
|
+
name: "distillation-${data.train.dataset_name}-${teacher.model_name}-${policy.model_name}-${loss_fn.kl_type}-${distillation.topk_logits_k}"
|
|
266
|
+
swanlab:
|
|
267
|
+
project: "nemo-distillation"
|
|
268
|
+
name: "distillation-${data.train.dataset_name}-${teacher.model_name}-${policy.model_name}-${loss_fn.kl_type}-${distillation.topk_logits_k}"
|
|
269
|
+
tensorboard:
|
|
270
|
+
log_dir: "tb_logs-distillation-${data.train.dataset_name}"
|
|
271
|
+
mlflow:
|
|
272
|
+
experiment_name: "distillation-dev"
|
|
273
|
+
run_name: "distillation-math-cl-logger"
|
|
274
|
+
tracking_uri: "http://localhost:5000"
|
|
275
|
+
gpu_monitoring:
|
|
276
|
+
collection_interval: 10
|
|
277
|
+
flush_interval: 10
|
|
278
|
+
|
|
279
|
+
cluster:
|
|
280
|
+
gpus_per_node: 8
|
|
281
|
+
num_nodes: 1
|
|
282
|
+
master_port_range_low: 25000
|
|
283
|
+
master_port_range_high: 28000
|
|
284
|
+
segment_size: null # Nodes per NVLink domain segment for topology-aware alignment; null to disable
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# LoRA / PEFT overlay(Megatron 后端)——实测在低显存机型上让 9B GRPO 能跑起来。
|
|
2
|
+
# 用法:在实验 defaults 里放在 grpo_megatron.yaml「之后」。删掉这一行即回到全参数微调。
|
|
3
|
+
# defaults:
|
|
4
|
+
# - ../../configs/base/grpo_math_1B.yaml
|
|
5
|
+
# - ../../configs/models/qwen3.5-9b.yaml
|
|
6
|
+
# - ../../configs/base/grpo_megatron.yaml # Megatron 后端 + 低显存调优
|
|
7
|
+
# - ../../configs/base/grpo_lora.yaml # ← LoRA(删此行即全参数)
|
|
8
|
+
#
|
|
9
|
+
# 要点(来自跑通的 grpo_math_9B_megatron):
|
|
10
|
+
# - LoRA 学习率比全参数高 1~2 个量级(这里 1e-4;4B 实测用 2e-4)
|
|
11
|
+
# - weight_decay=0,cosine 衰减,warmup 比全参数多
|
|
12
|
+
# - target_modules 用 Megatron 的命名(linear_qkv/linear_proj/linear_fc1/linear_fc2)
|
|
13
|
+
policy:
|
|
14
|
+
megatron_cfg:
|
|
15
|
+
peft:
|
|
16
|
+
enabled: true
|
|
17
|
+
target_modules: ["linear_qkv", "linear_proj", "linear_fc1", "linear_fc2"]
|
|
18
|
+
exclude_modules: []
|
|
19
|
+
dim: 8
|
|
20
|
+
alpha: 16
|
|
21
|
+
dropout: 0.0
|
|
22
|
+
lora_dtype: "bfloat16"
|
|
23
|
+
optimizer:
|
|
24
|
+
lr: 1.0e-4
|
|
25
|
+
min_lr: 1.0e-5
|
|
26
|
+
weight_decay: 0.0
|
|
27
|
+
scheduler:
|
|
28
|
+
lr_decay_style: "cosine"
|
|
29
|
+
lr_warmup_iters: 50
|
|
30
|
+
lr_decay_iters: 1000
|