{ "total_timesteps": 20000000, "policy": "MlpPolicy", "policy_kwargs": { "log_std_init": -3, "activation_fn": "nn.ReLU", "net_arch": [256, 256] }, "learning_rate": "linear_schedule(1e-3,5e-4)", "batch_size": 256, "gamma": 0.99, "learning_starts": 10000, "buffer_size": 1000000, "tau": 0.005, "ent_coef": "auto", "train_freq": 1, "gradient_steps": 1, "use_sde": true }