stable-baselines3/torchy_baselines/ppo/policies.py

238 lines
11 KiB
Python
Raw Normal View History

2020-03-24 09:10:37 +00:00
from typing import Optional, List, Tuple, Callable, Union, Dict, Type
2019-09-21 14:48:51 +00:00
from functools import partial
2020-03-10 17:09:45 +00:00
import gym
2019-09-18 11:10:27 +00:00
import torch as th
import torch.nn as nn
2019-09-21 14:48:51 +00:00
import numpy as np
2019-09-18 11:10:27 +00:00
from torchy_baselines.common.preprocessing import get_obs_dim
2020-03-10 17:09:45 +00:00
from torchy_baselines.common.policies import (BasePolicy, register_policy, MlpExtractor,
2020-03-23 16:15:30 +00:00
create_sde_features_extractor)
2020-03-10 17:09:45 +00:00
from torchy_baselines.common.distributions import (make_proba_distribution, Distribution,
2020-03-12 10:12:10 +00:00
DiagGaussianDistribution, CategoricalDistribution,
StateDependentNoiseDistribution)
2019-09-18 11:10:27 +00:00
2019-09-24 12:53:03 +00:00
2019-09-18 11:10:27 +00:00
class PPOPolicy(BasePolicy):
2019-11-22 16:24:47 +00:00
"""
Policy class (with both actor and critic) for A2C and derivates (PPO).
:param observation_space: (gym.spaces.Space) Observation space
2019-12-02 13:27:38 +00:00
:param action_space: (gym.spaces.Space) Action space
2020-03-16 13:05:21 +00:00
:param lr_schedule: (Callable) Learning rate schedule (could be constant)
2019-11-22 16:24:47 +00:00
:param net_arch: ([int or dict]) The specification of the policy and value networks.
:param device: (str or th.device) Device on which the code should run.
2020-03-24 09:10:37 +00:00
:param activation_fn: (Type[nn.Module]) Activation function
2019-11-22 16:24:47 +00:00
:param adam_epsilon: (float) Small values to avoid NaN in ADAM optimizer
:param ortho_init: (bool) Whether to use or not orthogonal initialization
:param use_sde: (bool) Whether to use State Dependent Exploration or not
:param log_std_init: (float) Initial value for the log standard deviation
2019-11-25 13:00:21 +00:00
:param full_std: (bool) Whether to use (n_features x n_actions) parameters
for the std instead of only (n_features,) when using SDE
:param sde_net_arch: ([int]) Network architecture for extracting features
when using SDE. If None, the latent features from the policy will be used.
Pass an empty list to use the states as features.
2020-03-12 14:34:35 +00:00
:param use_expln: (bool) Use ``expln()`` function instead of ``exp()`` to ensure
a positive standard deviation (cf paper). It allows to keep variance
2020-03-12 14:34:35 +00:00
above zero and prevent it from growing too fast. In practice, ``exp()`` is usually enough.
:param squash_output: (bool) Whether to squash the output using a tanh function,
this allows to ensure boundaries when using SDE.
2020-03-23 16:15:30 +00:00
:param normalize_images: (bool) Whether to normalize images or not,
dividing by 255.0 (True by default)
2019-11-22 16:24:47 +00:00
"""
2020-03-10 17:09:45 +00:00
def __init__(self,
observation_space: gym.spaces.Space,
action_space: gym.spaces.Space,
lr_schedule: Callable,
2020-03-10 17:09:45 +00:00
net_arch: Optional[List[Union[int, Dict[str, List[int]]]]] = None,
device: Union[th.device, str] = 'cpu',
2020-03-24 09:10:37 +00:00
activation_fn: Type[nn.Module] = nn.Tanh,
2020-03-10 17:09:45 +00:00
adam_epsilon: float = 1e-5,
ortho_init: bool = True,
use_sde: bool = False,
log_std_init: float = 0.0,
full_std: bool = True,
sde_net_arch: Optional[List[int]] = None,
use_expln: bool = False,
2020-03-23 16:15:30 +00:00
squash_output: bool = False,
normalize_images: bool = True):
super(PPOPolicy, self).__init__(observation_space, action_space, device, squash_output=squash_output)
2019-11-22 16:24:14 +00:00
# Default network architecture, from stable-baselines
2019-09-18 11:10:27 +00:00
if net_arch is None:
2019-11-22 16:24:14 +00:00
net_arch = [dict(pi=[64, 64], vf=[64, 64])]
2019-09-18 11:10:27 +00:00
self.net_arch = net_arch
self.activation_fn = activation_fn
2019-09-19 15:18:41 +00:00
self.adam_epsilon = adam_epsilon
2019-09-26 14:29:47 +00:00
self.ortho_init = ortho_init
2019-10-17 11:32:25 +00:00
# In the future, feature_extractor will be replaced with a CNN
self.features_extractor = nn.Flatten()
2020-03-23 16:15:30 +00:00
self.features_dim = get_obs_dim(self.observation_space)
self.normalize_images = normalize_images
2019-10-29 17:43:16 +00:00
self.log_std_init = log_std_init
2019-11-25 13:00:21 +00:00
dist_kwargs = None
# Keyword arguments for SDE distribution
if use_sde:
dist_kwargs = {
'full_std': full_std,
'squash_output': squash_output,
'use_expln': use_expln,
'learn_features': sde_net_arch is not None
2019-11-25 13:00:21 +00:00
}
2020-03-23 16:15:30 +00:00
self.sde_features_extractor = None
self.sde_net_arch = sde_net_arch
2020-01-22 16:17:12 +00:00
self.use_sde = use_sde
2019-10-28 17:24:13 +00:00
# Action distribution
2019-11-25 13:00:21 +00:00
self.action_dist = make_proba_distribution(action_space, use_sde=use_sde, dist_kwargs=dist_kwargs)
2019-10-28 17:24:13 +00:00
self._build(lr_schedule)
2019-09-18 11:10:27 +00:00
2020-03-10 17:09:45 +00:00
def reset_noise(self, n_envs: int = 1) -> None:
2019-11-25 12:19:33 +00:00
"""
Sample new weights for the exploration matrix.
:param n_envs: (int)
2019-11-25 12:19:33 +00:00
"""
2020-01-22 16:17:12 +00:00
assert isinstance(self.action_dist, StateDependentNoiseDistribution), 'reset_noise() is only available when using SDE'
self.action_dist.sample_weights(self.log_std, batch_size=n_envs)
2019-10-28 17:24:13 +00:00
def _build(self, lr_schedule: Callable) -> None:
2020-03-23 16:15:30 +00:00
"""
Create the networks and the optimizer.
:param lr_schedule: (Callable) Learning rate schedule
lr_schedule(1) is the initial learning rate
"""
2019-10-17 11:32:25 +00:00
self.mlp_extractor = MlpExtractor(self.features_dim, net_arch=self.net_arch,
activation_fn=self.activation_fn, device=self.device)
2019-09-21 16:12:06 +00:00
2019-12-02 13:27:38 +00:00
latent_dim_pi = self.mlp_extractor.latent_dim_pi
# Separate feature extractor for SDE
if self.sde_net_arch is not None:
2020-03-23 16:15:30 +00:00
self.sde_features_extractor, latent_sde_dim = create_sde_features_extractor(self.features_dim,
self.sde_net_arch,
self.activation_fn)
2019-11-25 14:02:10 +00:00
if isinstance(self.action_dist, DiagGaussianDistribution):
2019-12-02 13:27:38 +00:00
self.action_net, self.log_std = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi,
2019-10-29 17:43:16 +00:00
log_std_init=self.log_std_init)
elif isinstance(self.action_dist, StateDependentNoiseDistribution):
2019-12-02 13:27:38 +00:00
latent_sde_dim = latent_dim_pi if self.sde_net_arch is None else latent_sde_dim
self.action_net, self.log_std = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi,
latent_sde_dim=latent_sde_dim,
log_std_init=self.log_std_init)
elif isinstance(self.action_dist, CategoricalDistribution):
2019-12-02 13:27:38 +00:00
self.action_net = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi)
2019-10-17 11:32:25 +00:00
self.value_net = nn.Linear(self.mlp_extractor.latent_dim_vf, 1)
2019-09-21 14:48:51 +00:00
# Init weights: use orthogonal initialization
2019-09-26 14:29:47 +00:00
# with small initial weight for the output
if self.ortho_init:
2019-10-17 11:32:25 +00:00
for module in [self.mlp_extractor, self.action_net, self.value_net]:
2019-11-25 12:19:33 +00:00
# Values from stable-baselines, TODO: check why
2019-09-26 14:29:47 +00:00
gain = {
2019-10-17 11:32:25 +00:00
self.mlp_extractor: np.sqrt(2),
2019-09-26 14:29:47 +00:00
self.action_net: 0.01,
self.value_net: 1
}[module]
module.apply(partial(self.init_weights, gain=gain))
2020-03-23 16:15:30 +00:00
# Setup optimizer with initial learning rate
self.optimizer = th.optim.Adam(self.parameters(), lr=lr_schedule(1), eps=self.adam_epsilon)
2019-09-18 11:10:27 +00:00
2020-03-23 16:15:30 +00:00
def forward(self, obs: th.Tensor,
deterministic: bool = False) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
"""
Forward pass in all the networks (actor and critic)
:param obs: (th.Tensor) Observation
:param deterministic: (bool) Whether to sample or use deterministic actions
:return: (Tuple[th.Tensor, th.Tensor, th.Tensor]) action, value and log probability of the action
"""
latent_pi, latent_vf, latent_sde = self._get_latent(obs)
2020-03-23 16:15:30 +00:00
# Evaluate the values for the given observations
2019-09-21 16:12:06 +00:00
value = self.value_net(latent_vf)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde=latent_sde)
action = distribution.get_action(deterministic=deterministic)
log_prob = distribution.log_prob(action)
2019-09-18 11:10:27 +00:00
return action, value, log_prob
2020-03-10 17:09:45 +00:00
def _get_latent(self, obs: th.Tensor) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
2020-03-23 16:15:30 +00:00
"""
Get the latent code (i.e., activations of the last layer of each network)
for the different networks.
:param obs: (th.Tensor) Observation
:return: (Tuple[th.Tensor, th.Tensor, th.Tensor]) Latent codes
for the actor, the value function and for SDE function
"""
# Preprocess the observation if needed
features = self.extract_features(obs)
latent_pi, latent_vf = self.mlp_extractor(features)
# Features for sde
latent_sde = latent_pi
2020-03-23 16:15:30 +00:00
if self.sde_features_extractor is not None:
latent_sde = self.sde_features_extractor(features)
return latent_pi, latent_vf, latent_sde
2020-03-10 17:09:45 +00:00
def _get_action_dist_from_latent(self, latent_pi: th.Tensor,
latent_sde: Optional[th.Tensor] = None) -> Distribution:
2020-03-23 16:15:30 +00:00
"""
Retrieve action distribution given the latent codes.
2020-03-23 16:15:30 +00:00
:param latent_pi: (th.Tensor) Latent code for the actor
:param latent_sde: (Optional[th.Tensor]) Latent code for the SDE exploration function
:return: (Distribution) Action distribution
2020-03-23 16:15:30 +00:00
"""
2019-10-31 10:44:27 +00:00
mean_actions = self.action_net(latent_pi)
2019-10-29 17:43:16 +00:00
if isinstance(self.action_dist, DiagGaussianDistribution):
return self.action_dist.proba_distribution(mean_actions, self.log_std)
2019-10-29 17:43:16 +00:00
elif isinstance(self.action_dist, CategoricalDistribution):
2019-12-02 13:14:48 +00:00
# Here mean_actions are the logits before the softmax
return self.action_dist.proba_distribution(action_logits=mean_actions)
2019-10-29 17:43:16 +00:00
2019-10-28 17:24:13 +00:00
elif isinstance(self.action_dist, StateDependentNoiseDistribution):
return self.action_dist.proba_distribution(mean_actions, self.log_std, latent_sde)
2020-02-13 12:46:22 +00:00
else:
raise ValueError('Invalid action distribution')
2019-09-19 09:43:15 +00:00
def _predict(self, observation: th.Tensor, deterministic: bool = False) -> th.Tensor:
2020-03-23 16:15:30 +00:00
"""
Get the action according to the policy for a given observation.
:param observation: (th.Tensor)
:param deterministic: (bool) Whether to use stochastic or deterministic actions
:return: (th.Tensor) Taken action according to the policy
"""
2020-02-12 14:25:05 +00:00
latent_pi, _, latent_sde = self._get_latent(observation)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
return distribution.get_action(deterministic=deterministic)
2019-09-19 09:43:15 +00:00
2020-03-12 10:12:10 +00:00
def evaluate_actions(self, obs: th.Tensor,
actions: th.Tensor) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
2019-11-22 16:24:47 +00:00
"""
Evaluate actions according to the current policy,
given the observations.
:param obs: (th.Tensor)
2020-03-12 10:12:10 +00:00
:param actions: (th.Tensor)
2019-11-22 16:24:47 +00:00
:return: (th.Tensor, th.Tensor, th.Tensor) estimated value, log likelihood of taking those actions
and entropy of the action distribution.
"""
latent_pi, latent_vf, latent_sde = self._get_latent(obs)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
log_prob = distribution.log_prob(actions)
2020-03-10 17:09:45 +00:00
values = self.value_net(latent_vf)
return values, log_prob, distribution.entropy()
2019-09-18 11:10:27 +00:00
2019-09-21 15:17:09 +00:00
2019-09-18 11:10:27 +00:00
MlpPolicy = PPOPolicy
register_policy("MlpPolicy", MlpPolicy)