diff --git a/docs/guide/install.rst b/docs/guide/install.rst index f7a444e..61553ec 100644 --- a/docs/guide/install.rst +++ b/docs/guide/install.rst @@ -44,8 +44,8 @@ Bleeding-edge version pip install git+https://github.com/DLR-RM/stable-baselines3 -Development verion ------------------- +Development version +------------------- To contribute to Stable-Baselines3, with support for running tests and building the documentation. diff --git a/docs/spelling_wordlist.txt b/docs/spelling_wordlist.txt index 6ac1c76..b708581 100644 --- a/docs/spelling_wordlist.txt +++ b/docs/spelling_wordlist.txt @@ -10,7 +10,10 @@ VecEnv pretrain petrained tf +th +nn np +str mujoco cpu ndarray diff --git a/stable_baselines3/common/distributions.py b/stable_baselines3/common/distributions.py index e3ff2a7..fc601ef 100644 --- a/stable_baselines3/common/distributions.py +++ b/stable_baselines3/common/distributions.py @@ -24,7 +24,7 @@ class Distribution(object): def entropy(self) -> Optional[th.Tensor]: """ - Returns shannon's entropy of the probability + Returns Shannon's entropy of the probability :return: (Optional[th.Tensor]) the entropy, return None if no analytical form is known @@ -33,7 +33,7 @@ class Distribution(object): def sample(self) -> th.Tensor: """ - Returns a sample from the probabilty distribution + Returns a sample from the probability distribution :return: (th.Tensor) the stochastic action """ @@ -42,7 +42,7 @@ class Distribution(object): def mode(self) -> th.Tensor: """ Returns the most likely action (deterministic output) - from the probabilty distribution + from the probability distribution :return: (th.Tensor) the stochastic action """ @@ -50,7 +50,7 @@ class Distribution(object): def get_actions(self, deterministic: bool = False) -> th.Tensor: """ - Return actions according to the probabilty distribution. + Return actions according to the probability distribution. :param deterministic: (bool) :return: (th.Tensor) @@ -61,7 +61,7 @@ class Distribution(object): def actions_from_params(self, *args, **kwargs) -> th.Tensor: """ - Returns samples from the probabilty distribution + Returns samples from the probability distribution given its parameters. :return: (th.Tensor) actions @@ -70,8 +70,8 @@ class Distribution(object): def log_prob_from_params(self, *args, **kwargs) -> Tuple[th.Tensor, th.Tensor]: """ - Returns samples and the associated log probabilties - from the probabilty distribution given its parameters. + Returns samples and the associated log probabilities + from the probability distribution given its parameters. :return: (th.Tuple[th.Tensor, th.Tensor]) actions and log prob """ @@ -113,7 +113,7 @@ class DiagGaussianDistribution(Distribution): log_std_init: float = 0.0) -> Tuple[nn.Module, nn.Parameter]: """ Create the layers and parameter that represent the distribution: - one output will be the mean of the gaussian, the other parameter will be the + one output will be the mean of the Gaussian, the other parameter will be the standard deviation (log std in fact to allow negative values) :param latent_dim: (int) Dimension og the last layer of the policy (before the action layer) @@ -158,7 +158,7 @@ class DiagGaussianDistribution(Distribution): def log_prob_from_params(self, mean_actions: th.Tensor, log_std: th.Tensor) -> Tuple[th.Tensor, th.Tensor]: """ - Compute the log probabilty of taking an action + Compute the log probability of taking an action given the distribution parameters. :param mean_actions: (th.Tensor) @@ -171,7 +171,7 @@ class DiagGaussianDistribution(Distribution): def log_prob(self, actions: th.Tensor) -> th.Tensor: """ - Get the log probabilties of actions according to the distribution. + Get the log probabilities of actions according to the distribution. Note that you must call ``proba_distribution()`` method before. :param actions: (th.Tensor) @@ -255,7 +255,7 @@ class CategoricalDistribution(Distribution): """ Create the layer that represents the distribution: it will be the logits of the Categorical distribution. - You can then get probabilties using a softmax. + You can then get probabilities using a softmax. :param latent_dim: (int) Dimension of the last layer of the policy network (before the action layer) @@ -296,7 +296,7 @@ class StateDependentNoiseDistribution(Distribution): """ Distribution class for using State Dependent Exploration (SDE). It is used to create the noise exploration matrix and - compute the log probabilty of an action with that noise. + compute the log probability of an action with that noise. :param action_dim: (int) Dimension of the action space. :param full_std: (bool) Whether to use (n_features x n_actions) parameters @@ -485,7 +485,7 @@ class StateDependentNoiseDistribution(Distribution): class TanhBijector(object): """ - Bijective transformation of a probabilty distribution + Bijective transformation of a probability distribution using a squashing function (tanh) TODO: use Pyro instead (https://pyro.ai/) @@ -536,8 +536,8 @@ def make_proba_distribution(action_space: gym.spaces.Space, :param action_space: (gym.spaces.Space) the input action space :param use_sde: (bool) Force the use of StateDependentNoiseDistribution instead of DiagGaussianDistribution - :param dist_kwargs: (Optional[Dict[str, Any]]) Keyword arguments to pass to the probabilty distribution - :return: (Distribution) the approriate Distribution object + :param dist_kwargs: (Optional[Dict[str, Any]]) Keyword arguments to pass to the probability distribution + :return: (Distribution) the appropriate Distribution object """ if dist_kwargs is None: dist_kwargs = {} diff --git a/stable_baselines3/sac/sac.py b/stable_baselines3/sac/sac.py index 85aba9e..34e7d42 100644 --- a/stable_baselines3/sac/sac.py +++ b/stable_baselines3/sac/sac.py @@ -33,7 +33,7 @@ class SAC(OffPolicyRLModel): :param buffer_size: (int) size of the replay buffer :param learning_starts: (int) how many steps of the model to collect transitions for before learning starts :param batch_size: (int) Minibatch size for each gradient update - :param tau: (float) the soft update coefficient ("polyak update", between 0 and 1) + :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) :param gamma: (float) the discount factor :param train_freq: (int) Update the model every ``train_freq`` steps. :param gradient_steps: (int) How many gradient update after each step diff --git a/stable_baselines3/td3/td3.py b/stable_baselines3/td3/td3.py index 18361f2..015727d 100644 --- a/stable_baselines3/td3/td3.py +++ b/stable_baselines3/td3/td3.py @@ -26,7 +26,7 @@ class TD3(OffPolicyRLModel): :param buffer_size: (int) size of the replay buffer :param learning_starts: (int) how many steps of the model to collect transitions for before learning starts :param batch_size: (int) Minibatch size for each gradient update - :param tau: (float) the soft update coefficient ("polyak update", between 0 and 1) + :param tau: (float) the soft update coefficient ("Polyak update", between 0 and 1) :param gamma: (float) the discount factor :param train_freq: (int) Update the model every ``train_freq`` steps. :param gradient_steps: (int) How many gradient update after each step