Files
Motrixlab/motrix_rl/template/skrl_config.yaml
motphys-developers e1421d1055 chore: release v0.3.0
2026-04-02 03:45:10 +00:00

76 lines
2.1 KiB
YAML

seed: 42
# Models are instantiated using skrl's model instantiator utility
# https://skrl.readthedocs.io/en/latest/api/utils/model_instantiators.html
models:
separate: False
policy: # gaussian model
class: "GaussianMixin"
clip_actions: True
clip_log_std: True
initial_log_std: 0.0
min_log_std: -20.0
max_log_std: 2.0
input: "STATES"
hiddens: [32, 32]
hidden_activation: ["elu", "elu"]
output: "ACTIONS"
output_activation: "tanh"
output_scale: 1.0
value: # deterministic model
class: "DeterministicMixin"
clip_actions: False
input: "STATES"
hiddens: [32, 32]
hidden_activation: ["elu", "elu"]
output: "ONE"
output_activation: ""
output_scale: 1.0
# Memory
# https://skrl.readthedocs.io/en/latest/api/memories/random.html
memory:
class: "RandomMemory"
memory_size: -1 # -1: automatically determined value
# PPO agent configuration (field names are from PPO_DEFAULT_CONFIG)
# https://skrl.readthedocs.io/en/latest/api/agents/ppo.html
agent:
class: "PPO"
rollouts: 16
learning_epochs: 8
mini_batches: 1
discount_factor: 0.99
lambda: 0.95
learning_rate: 3.e-4
learning_rate_scheduler: "KLAdaptiveLR"
learning_rate_scheduler_kwargs:
kl_threshold: 0.008
random_timesteps: 0 # random exploration steps
learning_starts: 0 # learning starts after this many steps
grad_norm_clip: 1.0
ratio_clip: 0.2
value_clip: 0.2
clip_predicted_values: True
entropy_loss_scale: 0.0
value_loss_scale: 2.0
kl_threshold: 0
rewards_shaper_scale: 1.0
time_limit_bootstrap: False
# logging and checkpoint
experiment:
directory: "runs"
experiment_name: ""
write_interval: 16
checkpoint_interval: 80
store_separately: False
wandb: False
wandb_kwargs: null
# Sequential trainer
# https://skrl.readthedocs.io/en/latest/api/trainers/sequential.html
trainer:
class: "SequentialTrainer"
timesteps: 1600
environment_info: "log"