max_arm_step: 5
rollout_length: 50
penalty_coef: 5
auto_alpha: False
alpha: 1.0
n_epochs: 3000