stable-baselines3/stable_baselines3/ppo/policies.py

363 lines
18 KiB
Python
Raw Normal View History

from typing import Optional, List, Tuple, Callable, Union, Dict, Type, Any
2019-09-21 14:48:51 +00:00
from functools import partial
2020-03-10 17:09:45 +00:00
import gym
2019-09-18 11:10:27 +00:00
import torch as th
import torch.nn as nn
2019-09-21 14:48:51 +00:00
import numpy as np
2019-09-18 11:10:27 +00:00
2020-05-05 13:02:35 +00:00
from stable_baselines3.common.policies import (BasePolicy, register_policy, MlpExtractor,
2020-04-21 14:22:46 +00:00
create_sde_features_extractor, NatureCNN,
BaseFeaturesExtractor, FlattenExtractor)
2020-05-05 13:02:35 +00:00
from stable_baselines3.common.distributions import (make_proba_distribution, Distribution,
2020-03-12 10:12:10 +00:00
DiagGaussianDistribution, CategoricalDistribution,
StateDependentNoiseDistribution)
2019-09-18 11:10:27 +00:00
2019-09-24 12:53:03 +00:00
2019-09-18 11:10:27 +00:00
class PPOPolicy(BasePolicy):
2019-11-22 16:24:47 +00:00
"""
Policy class (with both actor and critic) for A2C and derivates (PPO).
:param observation_space: (gym.spaces.Space) Observation space
2019-12-02 13:27:38 +00:00
:param action_space: (gym.spaces.Space) Action space
2020-03-16 13:05:21 +00:00
:param lr_schedule: (Callable) Learning rate schedule (could be constant)
2019-11-22 16:24:47 +00:00
:param net_arch: ([int or dict]) The specification of the policy and value networks.
:param device: (str or th.device) Device on which the code should run.
2020-03-24 09:10:37 +00:00
:param activation_fn: (Type[nn.Module]) Activation function
2019-11-22 16:24:47 +00:00
:param ortho_init: (bool) Whether to use or not orthogonal initialization
:param use_sde: (bool) Whether to use State Dependent Exploration or not
:param log_std_init: (float) Initial value for the log standard deviation
2019-11-25 13:00:21 +00:00
:param full_std: (bool) Whether to use (n_features x n_actions) parameters
for the std instead of only (n_features,) when using SDE
:param sde_net_arch: ([int]) Network architecture for extracting features
when using SDE. If None, the latent features from the policy will be used.
Pass an empty list to use the states as features.
2020-03-12 14:34:35 +00:00
:param use_expln: (bool) Use ``expln()`` function instead of ``exp()`` to ensure
a positive standard deviation (cf paper). It allows to keep variance
2020-03-12 14:34:35 +00:00
above zero and prevent it from growing too fast. In practice, ``exp()`` is usually enough.
:param squash_output: (bool) Whether to squash the output using a tanh function,
this allows to ensure boundaries when using SDE.
2020-04-21 14:22:46 +00:00
:param features_extractor_class: (Type[BaseFeaturesExtractor]) Features extractor to use.
2020-04-22 11:14:22 +00:00
:param features_extractor_kwargs: (Optional[Dict[str, Any]]) Keyword arguments
to pass to the feature extractor.
2020-03-23 16:15:30 +00:00
:param normalize_images: (bool) Whether to normalize images or not,
dividing by 255.0 (True by default)
2020-04-22 11:14:22 +00:00
:param optimizer_class: (Type[th.optim.Optimizer]) The optimizer to use,
``th.optim.Adam`` by default
:param optimizer_kwargs: (Optional[Dict[str, Any]]) Additional keyword arguments,
excluding the learning rate, to pass to the optimizer
2019-11-22 16:24:47 +00:00
"""
2020-03-10 17:09:45 +00:00
def __init__(self,
observation_space: gym.spaces.Space,
action_space: gym.spaces.Space,
lr_schedule: Callable,
2020-03-10 17:09:45 +00:00
net_arch: Optional[List[Union[int, Dict[str, List[int]]]]] = None,
device: Union[th.device, str] = 'auto',
2020-03-24 09:10:37 +00:00
activation_fn: Type[nn.Module] = nn.Tanh,
2020-03-10 17:09:45 +00:00
ortho_init: bool = True,
use_sde: bool = False,
log_std_init: float = 0.0,
full_std: bool = True,
sde_net_arch: Optional[List[int]] = None,
use_expln: bool = False,
2020-03-23 16:15:30 +00:00
squash_output: bool = False,
2020-04-21 14:22:46 +00:00
features_extractor_class: Type[BaseFeaturesExtractor] = FlattenExtractor,
2020-04-22 11:14:22 +00:00
features_extractor_kwargs: Optional[Dict[str, Any]] = None,
normalize_images: bool = True,
2020-04-22 11:14:22 +00:00
optimizer_class: Type[th.optim.Optimizer] = th.optim.Adam,
optimizer_kwargs: Optional[Dict[str, Any]] = None):
2020-04-22 11:14:22 +00:00
if optimizer_kwargs is None:
optimizer_kwargs = {}
# Small values to avoid NaN in ADAM optimizer
if optimizer_class == th.optim.Adam:
optimizer_kwargs['eps'] = 1e-5
super(PPOPolicy, self).__init__(observation_space, action_space,
device,
features_extractor_class,
features_extractor_kwargs,
optimizer_class=optimizer_class,
optimizer_kwargs=optimizer_kwargs,
squash_output=squash_output)
2019-11-22 16:24:14 +00:00
# Default network architecture, from stable-baselines
2019-09-18 11:10:27 +00:00
if net_arch is None:
2020-04-21 14:22:46 +00:00
if features_extractor_class == FlattenExtractor:
net_arch = [dict(pi=[64, 64], vf=[64, 64])]
else:
net_arch = []
2019-11-22 16:24:14 +00:00
2019-09-18 11:10:27 +00:00
self.net_arch = net_arch
self.activation_fn = activation_fn
2019-09-26 14:29:47 +00:00
self.ortho_init = ortho_init
2020-04-21 14:22:46 +00:00
2020-04-22 11:14:22 +00:00
self.features_extractor = features_extractor_class(self.observation_space,
**self.features_extractor_kwargs)
2020-04-21 14:22:46 +00:00
self.features_dim = self.features_extractor.features_dim
2020-03-23 16:15:30 +00:00
self.normalize_images = normalize_images
2019-10-29 17:43:16 +00:00
self.log_std_init = log_std_init
2019-11-25 13:00:21 +00:00
dist_kwargs = None
# Keyword arguments for SDE distribution
if use_sde:
dist_kwargs = {
'full_std': full_std,
'squash_output': squash_output,
'use_expln': use_expln,
'learn_features': sde_net_arch is not None
2019-11-25 13:00:21 +00:00
}
2020-03-23 16:15:30 +00:00
self.sde_features_extractor = None
self.sde_net_arch = sde_net_arch
2020-01-22 16:17:12 +00:00
self.use_sde = use_sde
self.dist_kwargs = dist_kwargs
2019-10-28 17:24:13 +00:00
# Action distribution
2019-11-25 13:00:21 +00:00
self.action_dist = make_proba_distribution(action_space, use_sde=use_sde, dist_kwargs=dist_kwargs)
2019-10-28 17:24:13 +00:00
self._build(lr_schedule)
2019-09-18 11:10:27 +00:00
def _get_data(self) -> Dict[str, Any]:
data = super()._get_data()
data.update(dict(
net_arch=self.net_arch,
activation_fn=self.activation_fn,
use_sde=self.use_sde,
log_std_init=self.log_std_init,
squash_output=self.dist_kwargs['squash_output'] if self.dist_kwargs else None,
full_std=self.dist_kwargs['full_std'] if self.dist_kwargs else None,
sde_net_arch=self.dist_kwargs['sde_net_arch'] if self.dist_kwargs else None,
use_expln=self.dist_kwargs['use_expln'] if self.dist_kwargs else None,
lr_schedule=self._dummy_schedule, # dummy lr schedule, not needed for loading policy alone
2020-04-21 14:22:46 +00:00
ortho_init=self.ortho_init,
2020-04-22 11:14:22 +00:00
optimizer_class=self.optimizer_class,
optimizer_kwargs=self.optimizer_kwargs,
features_extractor_class=self.features_extractor_class,
features_extractor_kwargs=self.features_extractor_kwargs
))
return data
2020-03-10 17:09:45 +00:00
def reset_noise(self, n_envs: int = 1) -> None:
2019-11-25 12:19:33 +00:00
"""
Sample new weights for the exploration matrix.
:param n_envs: (int)
2019-11-25 12:19:33 +00:00
"""
2020-01-22 16:17:12 +00:00
assert isinstance(self.action_dist, StateDependentNoiseDistribution), 'reset_noise() is only available when using SDE'
self.action_dist.sample_weights(self.log_std, batch_size=n_envs)
2019-10-28 17:24:13 +00:00
def _build(self, lr_schedule: Callable) -> None:
2020-03-23 16:15:30 +00:00
"""
Create the networks and the optimizer.
:param lr_schedule: (Callable) Learning rate schedule
lr_schedule(1) is the initial learning rate
"""
2019-10-17 11:32:25 +00:00
self.mlp_extractor = MlpExtractor(self.features_dim, net_arch=self.net_arch,
activation_fn=self.activation_fn, device=self.device)
2019-09-21 16:12:06 +00:00
2019-12-02 13:27:38 +00:00
latent_dim_pi = self.mlp_extractor.latent_dim_pi
# Separate feature extractor for SDE
if self.sde_net_arch is not None:
2020-03-23 16:15:30 +00:00
self.sde_features_extractor, latent_sde_dim = create_sde_features_extractor(self.features_dim,
self.sde_net_arch,
self.activation_fn)
2019-11-25 14:02:10 +00:00
if isinstance(self.action_dist, DiagGaussianDistribution):
2019-12-02 13:27:38 +00:00
self.action_net, self.log_std = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi,
2019-10-29 17:43:16 +00:00
log_std_init=self.log_std_init)
elif isinstance(self.action_dist, StateDependentNoiseDistribution):
2019-12-02 13:27:38 +00:00
latent_sde_dim = latent_dim_pi if self.sde_net_arch is None else latent_sde_dim
self.action_net, self.log_std = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi,
latent_sde_dim=latent_sde_dim,
log_std_init=self.log_std_init)
elif isinstance(self.action_dist, CategoricalDistribution):
2019-12-02 13:27:38 +00:00
self.action_net = self.action_dist.proba_distribution_net(latent_dim=latent_dim_pi)
2019-10-17 11:32:25 +00:00
self.value_net = nn.Linear(self.mlp_extractor.latent_dim_vf, 1)
2019-09-21 14:48:51 +00:00
# Init weights: use orthogonal initialization
2019-09-26 14:29:47 +00:00
# with small initial weight for the output
if self.ortho_init:
2020-04-22 11:14:22 +00:00
# TODO: check for features_extractor
for module in [self.features_extractor, self.mlp_extractor,
self.action_net, self.value_net]:
2019-11-25 12:19:33 +00:00
# Values from stable-baselines, TODO: check why
2019-09-26 14:29:47 +00:00
gain = {
2020-04-22 11:14:22 +00:00
self.features_extractor: np.sqrt(2),
2019-10-17 11:32:25 +00:00
self.mlp_extractor: np.sqrt(2),
2019-09-26 14:29:47 +00:00
self.action_net: 0.01,
self.value_net: 1
}[module]
module.apply(partial(self.init_weights, gain=gain))
2020-03-23 16:15:30 +00:00
# Setup optimizer with initial learning rate
self.optimizer = self.optimizer_class(self.parameters(), lr=lr_schedule(1), **self.optimizer_kwargs)
2019-09-18 11:10:27 +00:00
2020-03-23 16:15:30 +00:00
def forward(self, obs: th.Tensor,
deterministic: bool = False) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
"""
Forward pass in all the networks (actor and critic)
:param obs: (th.Tensor) Observation
:param deterministic: (bool) Whether to sample or use deterministic actions
:return: (Tuple[th.Tensor, th.Tensor, th.Tensor]) action, value and log probability of the action
"""
latent_pi, latent_vf, latent_sde = self._get_latent(obs)
2020-03-23 16:15:30 +00:00
# Evaluate the values for the given observations
values = self.value_net(latent_vf)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde=latent_sde)
actions = distribution.get_actions(deterministic=deterministic)
log_prob = distribution.log_prob(actions)
return actions, values, log_prob
2019-09-18 11:10:27 +00:00
2020-03-10 17:09:45 +00:00
def _get_latent(self, obs: th.Tensor) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
2020-03-23 16:15:30 +00:00
"""
Get the latent code (i.e., activations of the last layer of each network)
for the different networks.
:param obs: (th.Tensor) Observation
:return: (Tuple[th.Tensor, th.Tensor, th.Tensor]) Latent codes
for the actor, the value function and for SDE function
"""
# Preprocess the observation if needed
features = self.extract_features(obs)
latent_pi, latent_vf = self.mlp_extractor(features)
# Features for sde
latent_sde = latent_pi
2020-03-23 16:15:30 +00:00
if self.sde_features_extractor is not None:
latent_sde = self.sde_features_extractor(features)
return latent_pi, latent_vf, latent_sde
2020-03-10 17:09:45 +00:00
def _get_action_dist_from_latent(self, latent_pi: th.Tensor,
latent_sde: Optional[th.Tensor] = None) -> Distribution:
2020-03-23 16:15:30 +00:00
"""
Retrieve action distribution given the latent codes.
2020-03-23 16:15:30 +00:00
:param latent_pi: (th.Tensor) Latent code for the actor
:param latent_sde: (Optional[th.Tensor]) Latent code for the SDE exploration function
:return: (Distribution) Action distribution
2020-03-23 16:15:30 +00:00
"""
2019-10-31 10:44:27 +00:00
mean_actions = self.action_net(latent_pi)
2019-10-29 17:43:16 +00:00
if isinstance(self.action_dist, DiagGaussianDistribution):
return self.action_dist.proba_distribution(mean_actions, self.log_std)
2019-10-29 17:43:16 +00:00
elif isinstance(self.action_dist, CategoricalDistribution):
2019-12-02 13:14:48 +00:00
# Here mean_actions are the logits before the softmax
return self.action_dist.proba_distribution(action_logits=mean_actions)
2019-10-29 17:43:16 +00:00
2019-10-28 17:24:13 +00:00
elif isinstance(self.action_dist, StateDependentNoiseDistribution):
return self.action_dist.proba_distribution(mean_actions, self.log_std, latent_sde)
2020-02-13 12:46:22 +00:00
else:
raise ValueError('Invalid action distribution')
2019-09-19 09:43:15 +00:00
def _predict(self, observation: th.Tensor, deterministic: bool = False) -> th.Tensor:
2020-03-23 16:15:30 +00:00
"""
Get the action according to the policy for a given observation.
:param observation: (th.Tensor)
:param deterministic: (bool) Whether to use stochastic or deterministic actions
:return: (th.Tensor) Taken action according to the policy
"""
2020-02-12 14:25:05 +00:00
latent_pi, _, latent_sde = self._get_latent(observation)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
return distribution.get_actions(deterministic=deterministic)
2019-09-19 09:43:15 +00:00
2020-03-12 10:12:10 +00:00
def evaluate_actions(self, obs: th.Tensor,
actions: th.Tensor) -> Tuple[th.Tensor, th.Tensor, th.Tensor]:
2019-11-22 16:24:47 +00:00
"""
Evaluate actions according to the current policy,
given the observations.
:param obs: (th.Tensor)
2020-03-12 10:12:10 +00:00
:param actions: (th.Tensor)
2019-11-22 16:24:47 +00:00
:return: (th.Tensor, th.Tensor, th.Tensor) estimated value, log likelihood of taking those actions
and entropy of the action distribution.
"""
latent_pi, latent_vf, latent_sde = self._get_latent(obs)
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
log_prob = distribution.log_prob(actions)
2020-03-10 17:09:45 +00:00
values = self.value_net(latent_vf)
return values, log_prob, distribution.entropy()
2019-09-18 11:10:27 +00:00
2019-09-21 15:17:09 +00:00
2019-09-18 11:10:27 +00:00
MlpPolicy = PPOPolicy
2020-04-21 14:22:46 +00:00
class CnnPolicy(PPOPolicy):
"""
CnnPolicy class (with both actor and critic) for A2C and derivates (PPO).
:param observation_space: (gym.spaces.Space) Observation space
:param action_space: (gym.spaces.Space) Action space
:param lr_schedule: (Callable) Learning rate schedule (could be constant)
:param net_arch: ([int or dict]) The specification of the policy and value networks.
:param device: (str or th.device) Device on which the code should run.
:param activation_fn: (Type[nn.Module]) Activation function
:param ortho_init: (bool) Whether to use or not orthogonal initialization
:param use_sde: (bool) Whether to use State Dependent Exploration or not
:param log_std_init: (float) Initial value for the log standard deviation
:param full_std: (bool) Whether to use (n_features x n_actions) parameters
for the std instead of only (n_features,) when using SDE
:param sde_net_arch: ([int]) Network architecture for extracting features
when using SDE. If None, the latent features from the policy will be used.
Pass an empty list to use the states as features.
:param use_expln: (bool) Use ``expln()`` function instead of ``exp()`` to ensure
a positive standard deviation (cf paper). It allows to keep variance
above zero and prevent it from growing too fast. In practice, ``exp()`` is usually enough.
:param squash_output: (bool) Whether to squash the output using a tanh function,
this allows to ensure boundaries when using SDE.
:param features_extractor_class: (Type[BaseFeaturesExtractor]) Features extractor to use.
2020-04-22 11:14:22 +00:00
:param features_extractor_kwargs: (Optional[Dict[str, Any]]) Keyword arguments
to pass to the feature extractor.
2020-04-21 14:22:46 +00:00
:param normalize_images: (bool) Whether to normalize images or not,
dividing by 255.0 (True by default)
2020-04-22 11:14:22 +00:00
:param optimizer_class: (Type[th.optim.Optimizer]) The optimizer to use,
2020-04-21 14:22:46 +00:00
``th.optim.Adam`` by default
:param optimizer_kwargs: (Optional[Dict[str, Any]]) Additional keyword arguments,
excluding the learning rate, to pass to the optimizer
"""
def __init__(self,
observation_space: gym.spaces.Space,
action_space: gym.spaces.Space,
lr_schedule: Callable,
net_arch: Optional[List[Union[int, Dict[str, List[int]]]]] = None,
device: Union[th.device, str] = 'auto',
activation_fn: Type[nn.Module] = nn.Tanh,
ortho_init: bool = True,
use_sde: bool = False,
log_std_init: float = 0.0,
full_std: bool = True,
sde_net_arch: Optional[List[int]] = None,
use_expln: bool = False,
squash_output: bool = False,
features_extractor_class: Type[BaseFeaturesExtractor] = NatureCNN,
2020-04-22 11:14:22 +00:00
features_extractor_kwargs: Optional[Dict[str, Any]] = None,
2020-04-21 14:22:46 +00:00
normalize_images: bool = True,
2020-04-22 11:14:22 +00:00
optimizer_class: Type[th.optim.Optimizer] = th.optim.Adam,
2020-04-21 14:22:46 +00:00
optimizer_kwargs: Optional[Dict[str, Any]] = None):
super(CnnPolicy, self).__init__(observation_space,
action_space,
lr_schedule,
net_arch,
device,
activation_fn,
ortho_init,
use_sde,
log_std_init,
full_std,
sde_net_arch,
use_expln,
squash_output,
features_extractor_class,
2020-04-22 11:14:22 +00:00
features_extractor_kwargs,
2020-04-21 14:22:46 +00:00
normalize_images,
2020-04-22 11:14:22 +00:00
optimizer_class,
2020-04-21 14:22:46 +00:00
optimizer_kwargs)
2019-09-18 11:10:27 +00:00
register_policy("MlpPolicy", MlpPolicy)
2020-04-21 14:22:46 +00:00
register_policy("CnnPolicy", CnnPolicy)