decrease extreme explorationin training
This commit is contained in:
+3
-3
@@ -87,7 +87,7 @@ def main():
|
|||||||
args.pretrained_model,
|
args.pretrained_model,
|
||||||
env=vec_env,
|
env=vec_env,
|
||||||
learning_rate=5e-5, # Lower learning rate so RL fine-tunes without destroying base gait
|
learning_rate=5e-5, # Lower learning rate so RL fine-tunes without destroying base gait
|
||||||
ent_coef=0.001,
|
ent_coef=0.0,
|
||||||
target_kl=0.05,
|
target_kl=0.05,
|
||||||
vf_coef=0.5,
|
vf_coef=0.5,
|
||||||
max_grad_norm=0.5,
|
max_grad_norm=0.5,
|
||||||
@@ -107,7 +107,7 @@ def main():
|
|||||||
gamma=0.99,
|
gamma=0.99,
|
||||||
gae_lambda=0.95,
|
gae_lambda=0.95,
|
||||||
clip_range=0.2,
|
clip_range=0.2,
|
||||||
ent_coef=0.001,
|
ent_coef=0.0,
|
||||||
target_kl=0.05,
|
target_kl=0.05,
|
||||||
vf_coef=0.5,
|
vf_coef=0.5,
|
||||||
max_grad_norm=0.5,
|
max_grad_norm=0.5,
|
||||||
@@ -115,7 +115,7 @@ def main():
|
|||||||
tensorboard_log=args.log_dir,
|
tensorboard_log=args.log_dir,
|
||||||
device="cpu",
|
device="cpu",
|
||||||
)
|
)
|
||||||
|
model.policy.log_std.data.fill_(-2.0)
|
||||||
# Setup Callbacks with ppo<number> naming
|
# Setup Callbacks with ppo<number> naming
|
||||||
checkpoint_callback = CheckpointCallback(
|
checkpoint_callback = CheckpointCallback(
|
||||||
save_freq=max(1, args.save_freq // args.num_workers),
|
save_freq=max(1, args.save_freq // args.num_workers),
|
||||||
|
|||||||
Reference in New Issue
Block a user