PyTorch torch.nn.Tanh Function
PyTorch torch.nn Reference Manual
torch.nn.TanhIt is the hyperbolic tangent activation function in PyTorch.
It maps input values to between -1 and 1, the output is zero-centered, and it is commonly used in recurrent neural networks.
Function Definition
torch.nn.Tanh()
Mathematical Principle
Tanh(x) = (e^x - e^(-x)) / (e^x + e^(-x))
Usage Examples
Example 1: Basic Usage
Example
import torch
import torch.nn as nn
tanh = nn.Tanh()
x = torch.tensor([-2.0, -1.0, 0.0, 1.0, 2.0])
output = tanh(x)
print("Input:", x.tolist())
print("Output:", output.tolist())
print("Features: Output range [-1, 1], zero-centered")
import torch.nn as nn
tanh = nn.Tanh()
x = torch.tensor([-2.0, -1.0, 0.0, 1.0, 2.0])
output = tanh(x)
print("Input:", x.tolist())
print("Output:", output.tolist())
print("Features: Output range [-1, 1], zero-centered")
Example 2: Use in LSTM
Example
import torch
import torch.nn as nn
class SimpleLSTMCell(nn.Module):
def __init__(self, input_size, hidden_size):
super(SimpleLSTMCell, self).__init__()
self.hidden_size = hidden_size
# The gating mechanism uses Tanh and Sigmoid
self.tanh = nn.Tanh()
self.sigmoid = nn.Sigmoid()
def forward(self, x, hidden):
h, c = hidden
# Simplified gating computation
gates = self.sigmoid(x @ torch.randn(x.shape[1], self.hidden_size * 4))
# Tanh is used for candidate memory
candidate = self.tanh(x @ torch.randn(x.shape[1], self.hidden_size))
return candidate, torch.zeros_like(h)
# Test
cell = SimpleLSTMCell(10, 20)
x = torch.randn(1, 10)
h = torch.randn(1, 20)
c = torch.randn(1, 20)
new_h, new_c = cell(x, (h, c))
print("Input shape:", x.shape)
print("Hidden state shape:", new_h.shape)
import torch.nn as nn
class SimpleLSTMCell(nn.Module):
def __init__(self, input_size, hidden_size):
super(SimpleLSTMCell, self).__init__()
self.hidden_size = hidden_size
# The gating mechanism uses Tanh and Sigmoid
self.tanh = nn.Tanh()
self.sigmoid = nn.Sigmoid()
def forward(self, x, hidden):
h, c = hidden
# Simplified gating computation
gates = self.sigmoid(x @ torch.randn(x.shape[1], self.hidden_size * 4))
# Tanh is used for candidate memory
candidate = self.tanh(x @ torch.randn(x.shape[1], self.hidden_size))
return candidate, torch.zeros_like(h)
# Test
cell = SimpleLSTMCell(10, 20)
x = torch.randn(1, 10)
h = torch.randn(1, 20)
c = torch.randn(1, 20)
new_h, new_c = cell(x, (h, c))
print("Input shape:", x.shape)
print("Hidden state shape:", new_h.shape)
Example 3: Compare with Sigmoid
Example
import torch
import torch.nn as nn
import numpy as np
x = np.linspace(-3, 3, 11)
x_tensor = torch.tensor(x, dtype=torch.float32)
sigmoid = nn.Sigmoid()
tanh = nn.Tanh()
print("x Sigmoid Tanh")
print("-" * 35)
for i in range(0, 11, 2):
xi = x_tensor[i:i+2]
print(f"{xi[0].item():5.1f} {sigmoid(xi)[0].item():9.4f} {tanh(xi)[0].item():9.4f}")
import torch.nn as nn
import numpy as np
x = np.linspace(-3, 3, 11)
x_tensor = torch.tensor(x, dtype=torch.float32)
sigmoid = nn.Sigmoid()
tanh = nn.Tanh()
print("x Sigmoid Tanh")
print("-" * 35)
for i in range(0, 11, 2):
xi = x_tensor[i:i+2]
print(f"{xi[0].item():5.1f} {sigmoid(xi)[0].item():9.4f} {tanh(xi)[0].item():9.4f}")
FAQ
Q1: Difference between Tanh and Sigmoid?
Tanh outputs in range [-1,1] (zero-centered), Sigmoid outputs [0,1].
Q2: Why does LSTM use Tanh?
Tanh's zero-centered property makes gradient flow more stable.
Use Cases
- Recurrent neural networks: Default activation for LSTM, GRU
- Generative models: Generator of GAN
- Gating mechanism: Control information range
Other extensions