From 5ff176b2f183819a89aa06c71bb83ec3ef0be90b Mon Sep 17 00:00:00 2001 From: Antonin RAFFIN Date: Thu, 16 Jul 2020 14:14:22 +0200 Subject: [PATCH] Implement DDPG (#92) * Add DDPG + TD3 with any number of critics * Allow any number of critics for SAC * Update doc * [ci skip] Update DDPG example * Remove unused parameter * Add DDPG to identity test * Fix computation with n_critics=1,3 * Update doc * Apply suggestions from code review Co-authored-by: Adam Gleave * Update docstrings for off-policy algos * Add check for sde Co-authored-by: Adam Gleave --- README.md | 4 +- docs/guide/algos.rst | 3 +- docs/guide/custom_policy.rst | 4 +- docs/index.rst | 3 +- docs/misc/changelog.rst | 6 +- docs/modules/ddpg.rst | 104 ++++++++++++++++ docs/spelling_wordlist.txt | 2 + setup.cfg | 1 + stable_baselines3/__init__.py | 3 +- stable_baselines3/common/base_class.py | 4 + stable_baselines3/common/noise.py | 2 +- .../common/off_policy_algorithm.py | 9 +- stable_baselines3/ddpg/__init__.py | 2 + stable_baselines3/ddpg/ddpg.py | 116 ++++++++++++++++++ stable_baselines3/ddpg/policies.py | 2 + stable_baselines3/dqn/dqn.py | 9 +- stable_baselines3/sac/sac.py | 29 +++-- stable_baselines3/td3/td3.py | 24 ++-- stable_baselines3/version.txt | 2 +- tests/test_callbacks.py | 4 +- tests/test_identity.py | 7 +- tests/test_run.py | 20 ++- tests/test_save_load.py | 3 +- 23 files changed, 315 insertions(+), 48 deletions(-) create mode 100644 docs/modules/ddpg.rst create mode 100644 stable_baselines3/ddpg/__init__.py create mode 100644 stable_baselines3/ddpg/ddpg.py create mode 100644 stable_baselines3/ddpg/policies.py diff --git a/README.md b/README.md index 048f2e0..d6338af 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,6 @@ These algorithms will make it easier for the research community and industry to Please look at the issue for more details. Planned features: -- [ ] DDPG (you can use its successor TD3 for now) - [ ] HER ### Planned features (v1.1+) @@ -152,6 +151,7 @@ All the following examples can be executed online using Google colab notebooks: - [Monitor Training and Plotting](https://colab.research.google.com/github/Stable-Baselines-Team/rl-colab-notebooks/blob/sb3/monitor_training.ipynb) - [Atari Games](https://colab.research.google.com/github/Stable-Baselines-Team/rl-colab-notebooks/blob/sb3/atari_games.ipynb) - [RL Baselines Zoo](https://colab.research.google.com/github/Stable-Baselines-Team/rl-colab-notebooks/blob/sb3/rl-baselines-zoo.ipynb) +- [PyBullet](https://colab.research.google.com/github/Stable-Baselines-Team/rl-colab-notebooks/blob/sb3/pybullet.ipynb) ## Implemented Algorithms @@ -159,6 +159,8 @@ All the following examples can be executed online using Google colab notebooks: | **Name** | **Recurrent** | `Box` | `Discrete` | `MultiDiscrete` | `MultiBinary` | **Multi Processing** | | ------------------- | ------------------ | ------------------ | ------------------ | ------------------- | ------------------ | --------------------------------- | | A2C | :x: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | +| DDPG | :x: | :heavy_check_mark: | :x: | :x: | :x: | :x: | +| DQN | :x: | :x: | :heavy_check_mark: | :x: | :x: | :x: | | PPO | :x: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | | SAC | :x: | :heavy_check_mark: | :x: | :x: | :x: | :x: | | TD3 | :x: | :heavy_check_mark: | :x: | :x: | :x: | :x: | diff --git a/docs/guide/algos.rst b/docs/guide/algos.rst index 821fee9..1cc3e9d 100644 --- a/docs/guide/algos.rst +++ b/docs/guide/algos.rst @@ -9,10 +9,11 @@ along with some useful characteristics: support for discrete/continuous actions, Name ``Box`` ``Discrete`` ``MultiDiscrete`` ``MultiBinary`` Multi Processing ============ =========== ============ ================= =============== ================ A2C ✔️ ✔️ ✔️ ✔️ ✔️ +DDPG ✔️ ❌ ❌ ❌ ❌ +DQN ❌ ✔️ ❌ ❌ ❌ PPO ✔️ ✔️ ✔️ ✔️ ✔️ SAC ✔️ ❌ ❌ ❌ ❌ TD3 ✔️ ❌ ❌ ❌ ❌ -DQN ❌ ✔️ ❌ ❌ ❌ ============ =========== ============ ================= =============== ================ diff --git a/docs/guide/custom_policy.rst b/docs/guide/custom_policy.rst index 84782ee..2ce9202 100644 --- a/docs/guide/custom_policy.rst +++ b/docs/guide/custom_policy.rst @@ -37,8 +37,8 @@ You can also easily define a custom architecture for the policy (or value) netwo .. note:: Defining a custom policy class is equivalent to passing ``policy_kwargs``. - However, it lets you name the policy and so makes usually the code clearer. - ``policy_kwargs`` should be rather used when doing hyperparameter search. + However, it lets you name the policy and so usually makes the code clearer. + ``policy_kwargs`` is particularly useful when doing hyperparameter search. diff --git a/docs/index.rst b/docs/index.rst index 2eea1bf..5dcb89e 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -55,10 +55,11 @@ Main Features modules/base modules/a2c + modules/ddpg + modules/dqn modules/ppo modules/sac modules/td3 - modules/dqn .. toctree:: :maxdepth: 1 diff --git a/docs/misc/changelog.rst b/docs/misc/changelog.rst index 1c6af97..bb5ceca 100644 --- a/docs/misc/changelog.rst +++ b/docs/misc/changelog.rst @@ -3,7 +3,7 @@ Changelog ========== -Pre-Release 0.8.0a3 (WIP) +Pre-Release 0.8.0a4 (WIP) ------------------------------ Breaking Changes: @@ -12,6 +12,8 @@ Breaking Changes: - ``save_replay_buffer`` now receives as argument the file path instead of the folder path (@tirafesi) - Refactored ``Critic`` class for ``TD3`` and ``SAC``, it is now called ``ContinuousCritic`` and has an additional parameter ``n_critics`` +- ``SAC`` and ``TD3`` now accept an arbitrary number of critics (e.g. ``policy_kwargs=dict(n_critics=3)``) + instead of only 2 previously New Features: ^^^^^^^^^^^^^ @@ -21,6 +23,7 @@ New Features: when ``psutil`` is available - Saving models now automatically creates the necessary folders and raises appropriate warnings (@PartiallyTyped) - Refactored opening paths for saving and loading to use strings, pathlib or io.BufferedIOBase (@PartiallyTyped) +- Added ``DDPG`` algorithm as a special case of ``TD3``. - Introduced ``BaseModel`` abstract parent for ``BasePolicy``, which critics inherit from. Bug Fixes: @@ -38,6 +41,7 @@ Others: - Added ``_on_step()`` for off-policy base class - Optimized replay buffer size by removing the need of ``next_observations`` numpy array - Ignored errors from newer pytype version +- Added a check when using ``gSDE`` Documentation: ^^^^^^^^^^^^^^ diff --git a/docs/modules/ddpg.rst b/docs/modules/ddpg.rst new file mode 100644 index 0000000..dd74f3a --- /dev/null +++ b/docs/modules/ddpg.rst @@ -0,0 +1,104 @@ +.. _ddpg: + +.. automodule:: stable_baselines3.ddpg + + +DDPG +==== + +`Deep Deterministic Policy Gradient (DDPG) `_ combines the +trick for DQN with the deterministic policy gradient, to obtain an algorithm for continuous actions. + + +.. rubric:: Available Policies + +.. autosummary:: + :nosignatures: + + MlpPolicy + + +Notes +----- + +- Deterministic Policy Gradient: http://proceedings.mlr.press/v32/silver14.pdf +- DDPG Paper: https://arxiv.org/abs/1509.02971 +- OpenAI Spinning Guide for DDPG: https://spinningup.openai.com/en/latest/algorithms/ddpg.html + +.. note:: + + The default policy for DDPG uses a ReLU activation, to match the original paper, whereas most other algorithms' MlpPolicy uses a tanh activation. + to match the original paper + + +Can I use? +---------- + +- Recurrent policies: ❌ +- Multi processing: ❌ +- Gym spaces: + + +============= ====== =========== +Space Action Observation +============= ====== =========== +Discrete ❌ ✔️ +Box ✔️ ✔️ +MultiDiscrete ❌ ✔️ +MultiBinary ❌ ✔️ +============= ====== =========== + + +Example +------- + +.. code-block:: python + + import gym + import numpy as np + + from stable_baselines3 import DDPG + from stable_baselines3.common.noise import NormalActionNoise, OrnsteinUhlenbeckActionNoise + + env = gym.make('Pendulum-v0') + + # The noise objects for DDPG + n_actions = env.action_space.shape[-1] + action_noise = NormalActionNoise(mean=np.zeros(n_actions), sigma=0.1 * np.ones(n_actions)) + + model = DDPG('MlpPolicy', env, action_noise=action_noise, verbose=1) + model.learn(total_timesteps=10000, log_interval=10) + model.save("ddpg_pendulum") + env = model.get_env() + + del model # remove to demonstrate saving and loading + + model = DDPG.load("ddpg_pendulum") + + obs = env.reset() + while True: + action, _states = model.predict(obs) + obs, rewards, dones, info = env.step(action) + env.render() + + +Parameters +---------- + +.. autoclass:: DDPG + :members: + :inherited-members: + +.. _ddpg_policies: + +DDPG Policies +------------- + +.. autoclass:: MlpPolicy + :members: + :inherited-members: + + +.. .. autoclass:: CnnPolicy +.. :members: +.. :inherited-members: diff --git a/docs/spelling_wordlist.txt b/docs/spelling_wordlist.txt index b708581..42669bf 100644 --- a/docs/spelling_wordlist.txt +++ b/docs/spelling_wordlist.txt @@ -117,3 +117,5 @@ Deprecations forkserver cuda Polyak +gSDE +rollouts diff --git a/setup.cfg b/setup.cfg index c20a6ad..1d01700 100644 --- a/setup.cfg +++ b/setup.cfg @@ -27,6 +27,7 @@ per-file-ignores = ./stable_baselines3/__init__.py:F401 ./stable_baselines3/common/__init__.py:F401 ./stable_baselines3/a2c/__init__.py:F401 + ./stable_baselines3/ddpg/__init__.py:F401 ./stable_baselines3/dqn/__init__.py:F401 ./stable_baselines3/ppo/__init__.py:F401 ./stable_baselines3/sac/__init__.py:F401 diff --git a/stable_baselines3/__init__.py b/stable_baselines3/__init__.py index 47a9bed..1bdb2e9 100644 --- a/stable_baselines3/__init__.py +++ b/stable_baselines3/__init__.py @@ -1,10 +1,11 @@ import os from stable_baselines3.a2c import A2C +from stable_baselines3.ddpg import DDPG +from stable_baselines3.dqn import DQN from stable_baselines3.ppo import PPO from stable_baselines3.sac import SAC from stable_baselines3.td3 import TD3 -from stable_baselines3.dqn import DQN # Read version from file version_file = os.path.join(os.path.dirname(__file__), 'version.txt') diff --git a/stable_baselines3/common/base_class.py b/stable_baselines3/common/base_class.py index e2e8d98..56f9f96 100644 --- a/stable_baselines3/common/base_class.py +++ b/stable_baselines3/common/base_class.py @@ -150,6 +150,10 @@ class BaseAlgorithm(ABC): raise ValueError("Error: the model does not support multiple envs; it requires " "a single vectorized environment.") + if self.use_sde and not isinstance(self.observation_space, gym.spaces.Box): + raise ValueError("generalized State-Dependent Exploration (gSDE) can only " + "be used with continuous actions.") + def _wrap_env(self, env: GymEnv) -> VecEnv: if not isinstance(env, VecEnv): if self.verbose >= 1: diff --git a/stable_baselines3/common/noise.py b/stable_baselines3/common/noise.py index d18866c..63b4b4a 100644 --- a/stable_baselines3/common/noise.py +++ b/stable_baselines3/common/noise.py @@ -46,7 +46,7 @@ class NormalActionNoise(ActionNoise): class OrnsteinUhlenbeckActionNoise(ActionNoise): """ - An Ornstein Uhlenbeck action noise, this is designed to aproximate brownian motion with friction. + An Ornstein Uhlenbeck action noise, this is designed to approximate Brownian motion with friction. Based on http://math.stackexchange.com/questions/1287634/implementing-ornstein-uhlenbeck-in-matlab diff --git a/stable_baselines3/common/off_policy_algorithm.py b/stable_baselines3/common/off_policy_algorithm.py index b7d92ad..5a8d085 100644 --- a/stable_baselines3/common/off_policy_algorithm.py +++ b/stable_baselines3/common/off_policy_algorithm.py @@ -35,10 +35,13 @@ class OffPolicyAlgorithm(BaseAlgorithm): :param batch_size: (int) Minibatch size for each gradient update :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) :param gamma: (float) the discount factor - :param train_freq: (int) Update the model every ``train_freq`` steps. - :param gradient_steps: (int) How many gradient update after each step + :param train_freq: (int) Update the model every ``train_freq`` steps. Set to `-1` to disable. + :param gradient_steps: (int) How many gradient steps to do after each rollout + (see ``train_freq`` and ``n_episodes_rollout``) + Set to ``-1`` means to do as many gradient steps as steps done in the environment + during the rollout. :param n_episodes_rollout: (int) Update the model every ``n_episodes_rollout`` episodes. - Note that this cannot be used at the same time as ``train_freq`` + Note that this cannot be used at the same time as ``train_freq``. Set to `-1` to disable. :param action_noise: (ActionNoise) the action noise type (None by default), this can help for hard exploration problem. Cf common.noise for the different action noise type. :param optimize_memory_usage: (bool) Enable a memory efficient variant of the replay buffer diff --git a/stable_baselines3/ddpg/__init__.py b/stable_baselines3/ddpg/__init__.py new file mode 100644 index 0000000..11a890b --- /dev/null +++ b/stable_baselines3/ddpg/__init__.py @@ -0,0 +1,2 @@ +from stable_baselines3.ddpg.ddpg import DDPG +from stable_baselines3.ddpg.policies import MlpPolicy, CnnPolicy diff --git a/stable_baselines3/ddpg/ddpg.py b/stable_baselines3/ddpg/ddpg.py new file mode 100644 index 0000000..83325b4 --- /dev/null +++ b/stable_baselines3/ddpg/ddpg.py @@ -0,0 +1,116 @@ +import torch as th +from typing import Type, Union, Callable, Optional, Dict, Any + +from stable_baselines3.common.off_policy_algorithm import OffPolicyAlgorithm +from stable_baselines3.common.noise import ActionNoise +from stable_baselines3.common.type_aliases import GymEnv, MaybeCallback +from stable_baselines3.td3.td3 import TD3 +from stable_baselines3.td3.policies import TD3Policy + + +class DDPG(TD3): + """ + Deep Deterministic Policy Gradient (DDPG). + + Deterministic Policy Gradient: http://proceedings.mlr.press/v32/silver14.pdf + DDPG Paper: https://arxiv.org/abs/1509.02971 + Introduction to DDPG: https://spinningup.openai.com/en/latest/algorithms/ddpg.html + + Note: we treat DDPG as a special case of its successor TD3. + + :param policy: (DDPGPolicy or str) The policy model to use (MlpPolicy, CnnPolicy, ...) + :param env: (GymEnv or str) The environment to learn from (if registered in Gym, can be str) + :param learning_rate: (float or callable) learning rate for adam optimizer, + the same learning rate will be used for all networks (Q-Values, Actor and Value function) + it can be a function of the current progress remaining (from 1 to 0) + :param buffer_size: (int) size of the replay buffer + :param learning_starts: (int) how many steps of the model to collect transitions for before learning starts + :param batch_size: (int) Minibatch size for each gradient update + :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) + :param gamma: (float) the discount factor + :param train_freq: (int) Update the model every ``train_freq`` steps. Set to `-1` to disable. + :param gradient_steps: (int) How many gradient steps to do after each rollout + (see ``train_freq`` and ``n_episodes_rollout``) + Set to ``-1`` means to do as many gradient steps as steps done in the environment + during the rollout. + :param n_episodes_rollout: (int) Update the model every ``n_episodes_rollout`` episodes. + Note that this cannot be used at the same time as ``train_freq``. Set to `-1` to disable. + :param action_noise: (ActionNoise) the action noise type (None by default), this can help + for hard exploration problem. Cf common.noise for the different action noise type. + :param optimize_memory_usage: (bool) Enable a memory efficient variant of the replay buffer + at a cost of more complexity. + See https://github.com/DLR-RM/stable-baselines3/issues/37#issuecomment-637501195 + :param create_eval_env: (bool) Whether to create a second environment that will be + used for evaluating the agent periodically. (Only available when passing string for the environment) + :param policy_kwargs: (dict) additional arguments to be passed to the policy on creation + :param verbose: (int) the verbosity level: 0 no output, 1 info, 2 debug + :param seed: (int) Seed for the pseudo random generators + :param device: (str or th.device) Device (cpu, cuda, ...) on which the code should be run. + Setting it to auto, the code will be run on the GPU if possible. + :param _init_setup_model: (bool) Whether or not to build the network at the creation of the instance + """ + + def __init__(self, policy: Union[str, Type[TD3Policy]], + env: Union[GymEnv, str], + learning_rate: Union[float, Callable] = 1e-3, + buffer_size: int = int(1e6), + learning_starts: int = 100, + batch_size: int = 100, + tau: float = 0.005, + gamma: float = 0.99, + train_freq: int = -1, + gradient_steps: int = -1, + n_episodes_rollout: int = 1, + action_noise: Optional[ActionNoise] = None, + optimize_memory_usage: bool = False, + tensorboard_log: Optional[str] = None, + create_eval_env: bool = False, + policy_kwargs: Dict[str, Any] = None, + verbose: int = 0, + seed: Optional[int] = None, + device: Union[th.device, str] = 'auto', + _init_setup_model: bool = True): + + super(DDPG, self).__init__(policy=policy, + env=env, + learning_rate=learning_rate, + buffer_size=buffer_size, + learning_starts=learning_starts, + batch_size=batch_size, + tau=tau, gamma=gamma, + train_freq=train_freq, + gradient_steps=gradient_steps, + n_episodes_rollout=n_episodes_rollout, + action_noise=action_noise, + policy_kwargs=policy_kwargs, + tensorboard_log=tensorboard_log, + verbose=verbose, device=device, + create_eval_env=create_eval_env, seed=seed, + optimize_memory_usage=optimize_memory_usage, + # Remove all tricks from TD3 to obtain DDPG: + # we still need to specify target_policy_noise > 0 to avoid errors + policy_delay=1, target_noise_clip=0.0, target_policy_noise=0.1, + _init_setup_model=False) + + # Use only one critic + if 'n_critics' not in self.policy_kwargs: + self.policy_kwargs['n_critics'] = 1 + + if _init_setup_model: + self._setup_model() + + def learn(self, + total_timesteps: int, + callback: MaybeCallback = None, + log_interval: int = 4, + eval_env: Optional[GymEnv] = None, + eval_freq: int = -1, + n_eval_episodes: int = 5, + tb_log_name: str = "DDPG", + eval_log_path: Optional[str] = None, + reset_num_timesteps: bool = True) -> OffPolicyAlgorithm: + + return super(DDPG, self).learn(total_timesteps=total_timesteps, callback=callback, log_interval=log_interval, + eval_env=eval_env, eval_freq=eval_freq, n_eval_episodes=n_eval_episodes, + tb_log_name=tb_log_name, eval_log_path=eval_log_path, + reset_num_timesteps=reset_num_timesteps) diff --git a/stable_baselines3/ddpg/policies.py b/stable_baselines3/ddpg/policies.py new file mode 100644 index 0000000..b826c7a --- /dev/null +++ b/stable_baselines3/ddpg/policies.py @@ -0,0 +1,2 @@ +# DDPG can be view as a special case of TD3 +from stable_baselines3.td3.policies import MlpPolicy, CnnPolicy # noqa:F401 diff --git a/stable_baselines3/dqn/dqn.py b/stable_baselines3/dqn/dqn.py index 084824e..05e0a0a 100644 --- a/stable_baselines3/dqn/dqn.py +++ b/stable_baselines3/dqn/dqn.py @@ -28,10 +28,13 @@ class DQN(OffPolicyAlgorithm): :param batch_size: (int) Minibatch size for each gradient update :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) default 1 for hard update :param gamma: (float) the discount factor - :param train_freq: (int) Update the model every ``train_freq`` steps. - :param gradient_steps: (int) How many gradient update after each step + :param train_freq: (int) Update the model every ``train_freq`` steps. Set to `-1` to disable. + :param gradient_steps: (int) How many gradient steps to do after each rollout + (see ``train_freq`` and ``n_episodes_rollout``) + Set to ``-1`` means to do as many gradient steps as steps done in the environment + during the rollout. :param n_episodes_rollout: (int) Update the model every ``n_episodes_rollout`` episodes. - Note that this cannot be used at the same time as ``train_freq`` + Note that this cannot be used at the same time as ``train_freq``. Set to `-1` to disable. :param optimize_memory_usage: (bool) Enable a memory efficient variant of the replay buffer at a cost of more complexity. See https://github.com/DLR-RM/stable-baselines3/issues/37#issuecomment-637501195 diff --git a/stable_baselines3/sac/sac.py b/stable_baselines3/sac/sac.py index 04e20fa..c177df3 100644 --- a/stable_baselines3/sac/sac.py +++ b/stable_baselines3/sac/sac.py @@ -34,10 +34,13 @@ class SAC(OffPolicyAlgorithm): :param batch_size: (int) Minibatch size for each gradient update :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) :param gamma: (float) the discount factor - :param train_freq: (int) Update the model every ``train_freq`` steps. - :param gradient_steps: (int) How many gradient update after each step + :param train_freq: (int) Update the model every ``train_freq`` steps. Set to `-1` to disable. + :param gradient_steps: (int) How many gradient steps to do after each rollout + (see ``train_freq`` and ``n_episodes_rollout``) + Set to ``-1`` means to do as many gradient steps as steps done in the environment + during the rollout. :param n_episodes_rollout: (int) Update the model every ``n_episodes_rollout`` episodes. - Note that this cannot be used at the same time as ``train_freq`` + Note that this cannot be used at the same time as ``train_freq``. Set to `-1` to disable. :param action_noise: (ActionNoise) the action noise type (None by default), this can help for hard exploration problem. Cf common.noise for the different action noise type. :param optimize_memory_usage: (bool) Enable a memory efficient variant of the replay buffer @@ -118,7 +121,6 @@ class SAC(OffPolicyAlgorithm): def _setup_model(self) -> None: super(SAC, self)._setup_model() self._create_aliases() - assert self.critic.n_critics == 2, "SAC only supports `n_critics=2` for now" # Target entropy is used when learning the entropy coefficient if self.target_entropy == 'auto': # automatically set target entropy if needed @@ -200,18 +202,20 @@ class SAC(OffPolicyAlgorithm): with th.no_grad(): # Select action according to policy next_actions, next_log_prob = self.actor.action_log_prob(replay_data.next_observations) - # Compute the target Q value - target_q1, target_q2 = self.critic_target(replay_data.next_observations, next_actions) - target_q = th.min(target_q1, target_q2) - ent_coef * next_log_prob.reshape(-1, 1) + # Compute the target Q value: min over all critics targets + targets = th.cat(self.critic_target(replay_data.next_observations, next_actions), dim=1) + target_q, _ = th.min(targets, dim=1, keepdim=True) + # add entropy term + target_q = target_q - ent_coef * next_log_prob.reshape(-1, 1) # td error + entropy term q_backup = replay_data.rewards + (1 - replay_data.dones) * self.gamma * target_q - # Get current Q estimates + # Get current Q estimates for each critic network # using action from the replay buffer - current_q1, current_q2 = self.critic(replay_data.observations, replay_data.actions) + current_q_esimates = self.critic(replay_data.observations, replay_data.actions) # Compute critic loss - critic_loss = 0.5 * (F.mse_loss(current_q1, q_backup) + F.mse_loss(current_q2, q_backup)) + critic_loss = 0.5 * sum([F.mse_loss(current_q, q_backup) for current_q in current_q_esimates]) critic_losses.append(critic_loss.item()) # Optimize the critic @@ -221,8 +225,9 @@ class SAC(OffPolicyAlgorithm): # Compute actor loss # Alternative: actor_loss = th.mean(log_prob - qf1_pi) - qf1_pi, qf2_pi = self.critic.forward(replay_data.observations, actions_pi) - min_qf_pi = th.min(qf1_pi, qf2_pi) + # Mean over all critic networks + q_values_pi = th.cat(self.critic.forward(replay_data.observations, actions_pi), dim=1) + min_qf_pi, _ = th.min(q_values_pi, dim=1, keepdim=True) actor_loss = (ent_coef * log_prob - min_qf_pi).mean() actor_losses.append(actor_loss.item()) diff --git a/stable_baselines3/td3/td3.py b/stable_baselines3/td3/td3.py index 13bcc98..83b7447 100644 --- a/stable_baselines3/td3/td3.py +++ b/stable_baselines3/td3/td3.py @@ -28,10 +28,13 @@ class TD3(OffPolicyAlgorithm): :param batch_size: (int) Minibatch size for each gradient update :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) :param gamma: (float) the discount factor - :param train_freq: (int) Update the model every ``train_freq`` steps. - :param gradient_steps: (int) How many gradient update after each step + :param train_freq: (int) Update the model every ``train_freq`` steps. Set to `-1` to disable. + :param gradient_steps: (int) How many gradient steps to do after each rollout + (see ``train_freq`` and ``n_episodes_rollout``) + Set to ``-1`` means to do as many gradient steps as steps done in the environment + during the rollout. :param n_episodes_rollout: (int) Update the model every ``n_episodes_rollout`` episodes. - Note that this cannot be used at the same time as ``train_freq`` + Note that this cannot be used at the same time as ``train_freq``. Set to `-1` to disable. :param action_noise: (ActionNoise) the action noise type (None by default), this can help for hard exploration problem. Cf common.noise for the different action noise type. :param optimize_memory_usage: (bool) Enable a memory efficient variant of the replay buffer @@ -96,7 +99,6 @@ class TD3(OffPolicyAlgorithm): def _setup_model(self) -> None: super(TD3, self)._setup_model() self._create_aliases() - assert self.critic.n_critics == 2, "TD3 only supports `n_critics=2` for now" def _create_aliases(self) -> None: self.actor = self.policy.actor @@ -120,18 +122,18 @@ class TD3(OffPolicyAlgorithm): noise = noise.clamp(-self.target_noise_clip, self.target_noise_clip) next_actions = (self.actor_target(replay_data.next_observations) + noise).clamp(-1, 1) - # Compute the target Q value - target_q1, target_q2 = self.critic_target(replay_data.next_observations, next_actions) - target_q = th.min(target_q1, target_q2) + # Compute the target Q value: min over all critics targets + targets = th.cat(self.critic_target(replay_data.next_observations, next_actions), dim=1) + target_q, _ = th.min(targets, dim=1, keepdim=True) target_q = replay_data.rewards + (1 - replay_data.dones) * self.gamma * target_q - # Get current Q estimates - current_q1, current_q2 = self.critic(replay_data.observations, replay_data.actions) + # Get current Q estimates for each critic network + current_q_esimates = self.critic(replay_data.observations, replay_data.actions) # Compute critic loss - critic_loss = F.mse_loss(current_q1, target_q) + F.mse_loss(current_q2, target_q) + critic_loss = sum([F.mse_loss(current_q, target_q) for current_q in current_q_esimates]) - # Optimize the critic + # Optimize the critics self.critic.optimizer.zero_grad() critic_loss.backward() self.critic.optimizer.step() diff --git a/stable_baselines3/version.txt b/stable_baselines3/version.txt index 8369211..9678ab2 100644 --- a/stable_baselines3/version.txt +++ b/stable_baselines3/version.txt @@ -1 +1 @@ -0.8.0a3 +0.8.0a4 diff --git a/tests/test_callbacks.py b/tests/test_callbacks.py index 8395364..fd4abb5 100644 --- a/tests/test_callbacks.py +++ b/tests/test_callbacks.py @@ -4,12 +4,12 @@ import shutil import pytest import gym -from stable_baselines3 import A2C, PPO, SAC, TD3, DQN +from stable_baselines3 import A2C, PPO, SAC, TD3, DQN, DDPG from stable_baselines3.common.callbacks import (CallbackList, CheckpointCallback, EvalCallback, EveryNTimesteps, StopTrainingOnRewardThreshold) -@pytest.mark.parametrize("model_class", [A2C, PPO, SAC, TD3, DQN]) +@pytest.mark.parametrize("model_class", [A2C, PPO, SAC, TD3, DQN, DDPG]) def test_callbacks(tmp_path, model_class): log_folder = tmp_path / 'logs/callbacks/' diff --git a/tests/test_identity.py b/tests/test_identity.py index f56622a..72d8c41 100644 --- a/tests/test_identity.py +++ b/tests/test_identity.py @@ -1,7 +1,7 @@ import numpy as np import pytest -from stable_baselines3 import A2C, PPO, SAC, TD3, DQN +from stable_baselines3 import A2C, PPO, SAC, TD3, DQN, DDPG from stable_baselines3.common.identity_env import (IdentityEnvBox, IdentityEnv, IdentityEnvMultiBinary, IdentityEnvMultiDiscrete) @@ -33,7 +33,7 @@ def test_discrete(model_class, env): assert np.shape(model.predict(obs)[0]) == np.shape(obs) -@pytest.mark.parametrize("model_class", [A2C, PPO, SAC, TD3]) +@pytest.mark.parametrize("model_class", [A2C, PPO, SAC, DDPG, TD3]) def test_continuous(model_class): env = IdentityEnvBox(eps=0.5) @@ -41,7 +41,8 @@ def test_continuous(model_class): A2C: 3500, PPO: 3000, SAC: 700, - TD3: 500 + TD3: 500, + DDPG: 500 }[model_class] kwargs = dict( diff --git a/tests/test_run.py b/tests/test_run.py index 5aa936a..d22d346 100644 --- a/tests/test_run.py +++ b/tests/test_run.py @@ -1,16 +1,20 @@ import numpy as np import pytest -from stable_baselines3 import A2C, PPO, SAC, TD3, DQN +from stable_baselines3 import A2C, PPO, SAC, TD3, DQN, DDPG from stable_baselines3.common.noise import NormalActionNoise, OrnsteinUhlenbeckActionNoise normal_action_noise = NormalActionNoise(np.zeros(1), 0.1 * np.ones(1)) +@pytest.mark.parametrize('model_class', [TD3, DDPG]) @pytest.mark.parametrize('action_noise', [normal_action_noise, OrnsteinUhlenbeckActionNoise(np.zeros(1), 0.1 * np.ones(1))]) -def test_td3(action_noise): - model = TD3('MlpPolicy', 'Pendulum-v0', policy_kwargs=dict(net_arch=[64, 64]), - learning_starts=100, verbose=1, create_eval_env=True, action_noise=action_noise) +def test_deterministic_pg(model_class, action_noise): + """ + Test for DDPG and variants (TD3). + """ + model = model_class('MlpPolicy', 'Pendulum-v0', policy_kwargs=dict(net_arch=[64, 64]), + learning_starts=100, verbose=1, create_eval_env=True, action_noise=action_noise) model.learn(total_timesteps=1000, eval_freq=500) @@ -42,6 +46,14 @@ def test_sac(ent_coef): model.learn(total_timesteps=1000, eval_freq=500) +@pytest.mark.parametrize("n_critics", [1, 3]) +def test_n_critics(n_critics): + # Test SAC with different number of critics, for TD3, n_critics=1 corresponds to DDPG + model = SAC('MlpPolicy', 'Pendulum-v0', policy_kwargs=dict(net_arch=[64, 64], n_critics=n_critics), + learning_starts=100, verbose=1) + model.learn(total_timesteps=1000) + + def test_dqn(): model = DQN('MlpPolicy', 'CartPole-v1', policy_kwargs=dict(net_arch=[64, 64]), learning_starts=500, buffer_size=500, learning_rate=3e-4, verbose=1, create_eval_env=True) diff --git a/tests/test_save_load.py b/tests/test_save_load.py index 95f8c93..69df31a 100644 --- a/tests/test_save_load.py +++ b/tests/test_save_load.py @@ -9,7 +9,7 @@ import gym import numpy as np import torch as th -from stable_baselines3 import A2C, PPO, SAC, TD3, DQN +from stable_baselines3 import A2C, PPO, SAC, TD3, DQN, DDPG from stable_baselines3.common.base_class import BaseAlgorithm from stable_baselines3.common.identity_env import IdentityEnvBox, IdentityEnv from stable_baselines3.common.vec_env import DummyVecEnv @@ -26,6 +26,7 @@ MODEL_LIST = [ TD3, SAC, DQN, + DDPG ]