Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
defaults: ../../grpo_math_1B.yaml
grpo:
val_period: -1
loss_fn:
reference_policy_kl_penalty: 0.04
checkpointing:
checkpoint_dir: results/grpo-moonlight-16b-a3b-instruct-automodel-1n8g-ep8
enabled: false
save_period: 10000
policy:
model_name: moonshotai/Moonlight-16B-A3B-Instruct
train_micro_batch_size: 1
max_total_sequence_length: 4096
activation_checkpointing_enabled: false
dtensor_cfg:
expert_parallel_size: 8
clear_cache_every_n_steps: 50
automodel_kwargs:
use_liger_kernel: false
backend:
_target_: nemo_automodel.components.moe.utils.BackendConfig
attn: te
linear: te
rms_norm: te
enable_deepep: true
fake_balanced_gate: false
enable_hf_state_dict_adapter: true
Comment thread
yuki-97 marked this conversation as resolved.
enable_fsdp_optimizations: false
gate_precision: float64
dynamic_batching:
enabled: true
sequence_packing:
enabled: false
make_sequence_length_divisible_by: 4
optimizer:
kwargs:
lr: 3.0e-07
scheduler:
- name: torch.optim.lr_scheduler.LinearLR
kwargs:
start_factor: 0.1
end_factor: 1
total_iters: 13
- name: torch.optim.lr_scheduler.ConstantLR
kwargs:
factor: 1
total_iters: 10000000000
- milestones:
- 13
data:
max_input_seq_length: 4096
cluster:
gpus_per_node: 8
logger:
log_dir: logs/grpo-moonlight-16b-a3b-instruct-automodel-1n8g-ep8
wandb_enabled: true
tensorboard_enabled: true
wandb:
project: nemo-rl
name: grpo-moonlight-16b-a3b-instruct-automodel-1n8g-ep8
12 changes: 11 additions & 1 deletion nemo_rl/models/policy/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,13 @@ class LoRAConfig(TypedDict):


class AutomodelBackendConfig(TypedDict):
"""Configuration for custom MoE implementation backend in Automodel.

Used when setting the backend in automodel_kwargs in your config.
Alternatively, pass `force_hf: true` in automodel_kwargs to fall back
to the HuggingFace implementation.
"""

# Hydra target class path (e.g., "nemo_automodel.components.moe.utils.BackendConfig")
_target_: str
# Attention implementation: "te" (Transformer Engine), "flex" (FlexAttention), etc.
Expand All @@ -47,7 +54,8 @@ class AutomodelBackendConfig(TypedDict):
enable_deepep: NotRequired[bool]
# Use fake balanced gate for testing/debugging MoE
fake_balanced_gate: NotRequired[bool]
# Enable HuggingFace state dict adapter for checkpoint loading
# Enable HuggingFace state dict adapter for checkpoint saving/loading plus refit support for RL
# This should almost always be set to True when using a custom MoE implementation. Set to False only for specific use cases like debugging or performance testing.
enable_hf_state_dict_adapter: NotRequired[bool]
# Enable FSDP-specific optimizations
enable_fsdp_optimizations: NotRequired[bool]
Expand All @@ -60,6 +68,8 @@ class AutomodelKwargs(TypedDict):
use_liger_kernel: NotRequired[bool]
# Backend configuration for MoE models
backend: NotRequired[AutomodelBackendConfig]
# Whether to force use of the HuggingFace implementation for MoE models
force_hf: NotRequired[bool]
Comment thread
yuki-97 marked this conversation as resolved.


class DTensorConfigDisabled(TypedDict):
Expand Down
Loading
Loading