mode: train # set to 'train' for training service
project: test # set your project name, must be the same as in Explorer
name: test # set your experiment name, must be the same as in Explorer
checkpoint_root_dir: CHECKPOINT_ROOT_DIR # set the root directory for checkpoints, must be the same as in Explorer
model:
model_path: /path/to/your/model # set the path to your base model, must be the same as in Explorer
max_model_len: 8192 # must be the same as in Explorer
max_response_tokens: 2048 # must be the same as in Explorer
temperature: 0.7 # must be the same as in Explorer
algorithm:
algorithm_type: "ppo" # current version only supports ppo for online training (group is not supported yet)
cluster:
node_num: 1
gpu_per_node: 4 # suppose you have 4 GPUs on the node
buffer:
train_batch_size: 32 # trainer consumes 16 samples per step
trainer_input:
experience_buffer:
name: exp_buffer # table name in the database, must be the same as in Explorer
storage_type: sql
# path: your_db_url # if not provided, use a sqlite database in checkpoint_root_dir/project/name/buffer
trainer:
save_interval: 16 # save checkpoint every step
ulysses_sequence_parallel_size: 1 # set according to your model and hardware
save_hf_checkpoint: always
max_checkpoints_to_keep: 5
trainer_config:
trainer:
balance_batch: false
max_actor_ckpt_to_keep: 5
max_critic_ckpt_to_keep: 5
synchronizer:
sync_method: checkpoint
sync_interval: 1
monitor:
monitor_type: tensorboard