chore: import upstream snapshot with attribution

This commit is contained in:
wehub-resource-sync
2026-07-13 13:39:55 +08:00
commit 7ee4420c10
87 changed files with 15222 additions and 0 deletions
+4
View File
@@ -0,0 +1,4 @@
# coding:utf-8
from .basic import *
from .convnet import *
from .normalization import *
+183
View File
@@ -0,0 +1,183 @@
# coding:utf-8
import autograd.numpy as np
from autograd import elementwise_grad
from mla.neuralnet.activations import get_activation
from mla.neuralnet.parameters import Parameters
np.random.seed(9999)
class Layer(object):
def setup(self, X_shape):
"""Allocates initial weights."""
pass
def forward_pass(self, x):
raise NotImplementedError()
def backward_pass(self, delta):
raise NotImplementedError()
def shape(self, x_shape):
"""Returns shape of the current layer."""
raise NotImplementedError()
class ParamMixin(object):
@property
def parameters(self):
return self._params
class PhaseMixin(object):
_train = False
@property
def is_training(self):
return self._train
@is_training.setter
def is_training(self, is_train=True):
self._train = is_train
@property
def is_testing(self):
return not self._train
@is_testing.setter
def is_testing(self, is_test=True):
self._train = not is_test
class Dense(Layer, ParamMixin):
def __init__(self, output_dim, parameters=None):
"""A fully connected layer.
Parameters
----------
output_dim : int
"""
self._params = parameters
self.output_dim = output_dim
self.last_input = None
if parameters is None:
self._params = Parameters()
def setup(self, x_shape):
self._params.setup_weights((x_shape[1], self.output_dim))
def forward_pass(self, X):
self.last_input = X
return self.weight(X)
def weight(self, X):
W = np.dot(X, self._params["W"])
return W + self._params["b"]
def backward_pass(self, delta):
dW = np.dot(self.last_input.T, delta)
db = np.sum(delta, axis=0)
# Update gradient values
self._params.update_grad("W", dW)
self._params.update_grad("b", db)
return np.dot(delta, self._params["W"].T)
def shape(self, x_shape):
return x_shape[0], self.output_dim
class Activation(Layer):
def __init__(self, name):
self.last_input = None
self.activation = get_activation(name)
# Derivative of activation function
self.activation_d = elementwise_grad(self.activation)
def forward_pass(self, X):
self.last_input = X
return self.activation(X)
def backward_pass(self, delta):
return self.activation_d(self.last_input) * delta
def shape(self, x_shape):
return x_shape
class Dropout(Layer, PhaseMixin):
"""Randomly set a fraction of `p` inputs to 0 at each training update."""
def __init__(self, p=0.1):
self.p = p
self._mask = None
def forward_pass(self, X):
assert self.p > 0
if self.is_training:
self._mask = np.random.uniform(size=X.shape) > self.p
y = X * self._mask
else:
y = X * (1.0 - self.p)
return y
def backward_pass(self, delta):
return delta * self._mask
def shape(self, x_shape):
return x_shape
class TimeStepSlicer(Layer):
"""Take a specific time step from 3D tensor."""
def __init__(self, step=-1):
self.step = step
def forward_pass(self, x):
return x[:, self.step, :]
def backward_pass(self, delta):
return np.repeat(delta[:, np.newaxis, :], 2, 1)
def shape(self, x_shape):
return x_shape[0], x_shape[2]
class TimeDistributedDense(Layer):
"""Apply regular Dense layer to every timestep."""
def __init__(self, output_dim):
self.output_dim = output_dim
self.n_timesteps = None
self.dense = None
self.input_dim = None
def setup(self, X_shape):
self.dense = Dense(self.output_dim)
self.dense.setup((X_shape[0], X_shape[2]))
self.input_dim = X_shape[2]
def forward_pass(self, X):
n_timesteps = X.shape[1]
X = X.reshape(-1, X.shape[-1])
y = self.dense.forward_pass(X)
y = y.reshape((-1, n_timesteps, self.output_dim))
return y
def backward_pass(self, delta):
n_timesteps = delta.shape[1]
X = delta.reshape(-1, delta.shape[-1])
y = self.dense.backward_pass(X)
y = y.reshape((-1, n_timesteps, self.input_dim))
return y
@property
def parameters(self):
return self.dense._params
def shape(self, x_shape):
return x_shape[0], x_shape[1], self.output_dim
+231
View File
@@ -0,0 +1,231 @@
# coding:utf-8
import autograd.numpy as np
from mla.neuralnet.layers import Layer, ParamMixin
from mla.neuralnet.parameters import Parameters
class Convolution(Layer, ParamMixin):
def __init__(
self,
n_filters=8,
filter_shape=(3, 3),
padding=(0, 0),
stride=(1, 1),
parameters=None,
):
"""A 2D convolutional layer.
Input shape: (n_images, n_channels, height, width)
Parameters
----------
n_filters : int, default 8
The number of filters (kernels).
filter_shape : tuple(int, int), default (3, 3)
The shape of the filters. (height, width)
parameters : Parameters instance, default None
stride : tuple(int, int), default (1, 1)
The step of the convolution. (height, width).
padding : tuple(int, int), default (0, 0)
The number of pixel to add to each side of the input. (height, weight)
"""
self.padding = padding
self._params = parameters
self.stride = stride
self.filter_shape = filter_shape
self.n_filters = n_filters
if self._params is None:
self._params = Parameters()
def setup(self, X_shape):
n_channels, self.height, self.width = X_shape[1:]
W_shape = (self.n_filters, n_channels) + self.filter_shape
b_shape = self.n_filters
self._params.setup_weights(W_shape, b_shape)
def forward_pass(self, X):
n_images, n_channels, height, width = self.shape(X.shape)
self.last_input = X
self.col = image_to_column(X, self.filter_shape, self.stride, self.padding)
self.col_W = self._params["W"].reshape(self.n_filters, -1).T
out = np.dot(self.col, self.col_W) + self._params["b"]
out = out.reshape(n_images, height, width, -1).transpose(0, 3, 1, 2)
return out
def backward_pass(self, delta):
delta = delta.transpose(0, 2, 3, 1).reshape(-1, self.n_filters)
d_W = np.dot(self.col.T, delta).transpose(1, 0).reshape(self._params["W"].shape)
d_b = np.sum(delta, axis=0)
self._params.update_grad("b", d_b)
self._params.update_grad("W", d_W)
d_c = np.dot(delta, self.col_W.T)
return column_to_image(
d_c, self.last_input.shape, self.filter_shape, self.stride, self.padding
)
def shape(self, x_shape):
height, width = convoltuion_shape(
self.height, self.width, self.filter_shape, self.stride, self.padding
)
return x_shape[0], self.n_filters, height, width
class MaxPooling(Layer):
def __init__(self, pool_shape=(2, 2), stride=(1, 1), padding=(0, 0)):
"""Max pooling layer.
Input shape: (n_images, n_channels, height, width)
Parameters
----------
pool_shape : tuple(int, int), default (2, 2)
stride : tuple(int, int), default (1,1)
padding : tuple(int, int), default (0,0)
"""
self.pool_shape = pool_shape
self.stride = stride
self.padding = padding
def forward_pass(self, X):
self.last_input = X
out_height, out_width = pooling_shape(self.pool_shape, X.shape, self.stride)
n_images, n_channels, _, _ = X.shape
col = image_to_column(X, self.pool_shape, self.stride, self.padding)
col = col.reshape(-1, self.pool_shape[0] * self.pool_shape[1])
arg_max = np.argmax(col, axis=1)
out = np.max(col, axis=1)
self.arg_max = arg_max
return out.reshape(n_images, out_height, out_width, n_channels).transpose(
0, 3, 1, 2
)
def backward_pass(self, delta):
delta = delta.transpose(0, 2, 3, 1)
pool_size = self.pool_shape[0] * self.pool_shape[1]
y_max = np.zeros((delta.size, pool_size))
y_max[np.arange(self.arg_max.size), self.arg_max.flatten()] = delta.flatten()
y_max = y_max.reshape(delta.shape + (pool_size,))
dcol = y_max.reshape(y_max.shape[0] * y_max.shape[1] * y_max.shape[2], -1)
return column_to_image(
dcol, self.last_input.shape, self.pool_shape, self.stride, self.padding
)
def shape(self, x_shape):
h, w = convoltuion_shape(
x_shape[2], x_shape[3], self.pool_shape, self.stride, self.padding
)
return x_shape[0], x_shape[1], h, w
class Flatten(Layer):
"""Flattens multidimensional input into 2D matrix."""
def forward_pass(self, X):
self.last_input_shape = X.shape
return X.reshape((X.shape[0], -1))
def backward_pass(self, delta):
return delta.reshape(self.last_input_shape)
def shape(self, x_shape):
return x_shape[0], np.prod(x_shape[1:])
def image_to_column(images, filter_shape, stride, padding):
"""Rearrange image blocks into columns.
Parameters
----------
filter_shape : tuple(height, width)
images : np.array, shape (n_images, n_channels, height, width)
padding: tuple(height, width)
stride : tuple (height, width)
"""
n_images, n_channels, height, width = images.shape
f_height, f_width = filter_shape
out_height, out_width = convoltuion_shape(
height, width, (f_height, f_width), stride, padding
)
images = np.pad(images, ((0, 0), (0, 0), padding, padding), mode="constant")
col = np.zeros((n_images, n_channels, f_height, f_width, out_height, out_width))
for y in range(f_height):
y_bound = y + stride[0] * out_height
for x in range(f_width):
x_bound = x + stride[1] * out_width
col[:, :, y, x, :, :] = images[
:, :, y : y_bound : stride[0], x : x_bound : stride[1]
]
col = col.transpose(0, 4, 5, 1, 2, 3).reshape(n_images * out_height * out_width, -1)
return col
def column_to_image(columns, images_shape, filter_shape, stride, padding):
"""Rearrange columns into image blocks.
Parameters
----------
columns
images_shape : tuple(n_images, n_channels, height, width)
filter_shape : tuple(height, _width)
stride : tuple(height, width)
padding : tuple(height, width)
"""
n_images, n_channels, height, width = images_shape
f_height, f_width = filter_shape
out_height, out_width = convoltuion_shape(
height, width, (f_height, f_width), stride, padding
)
columns = columns.reshape(
n_images, out_height, out_width, n_channels, f_height, f_width
).transpose(0, 3, 4, 5, 1, 2)
img_h = height + 2 * padding[0] + stride[0] - 1
img_w = width + 2 * padding[1] + stride[1] - 1
img = np.zeros((n_images, n_channels, img_h, img_w))
for y in range(f_height):
y_bound = y + stride[0] * out_height
for x in range(f_width):
x_bound = x + stride[1] * out_width
img[:, :, y : y_bound : stride[0], x : x_bound : stride[1]] += columns[
:, :, y, x, :, :
]
return img[:, :, padding[0] : height + padding[0], padding[1] : width + padding[1]]
def convoltuion_shape(img_height, img_width, filter_shape, stride, padding):
"""Calculate output shape for convolution layer."""
height = (img_height + 2 * padding[0] - filter_shape[0]) / float(stride[0]) + 1
width = (img_width + 2 * padding[1] - filter_shape[1]) / float(stride[1]) + 1
assert height % 1 == 0
assert width % 1 == 0
return int(height), int(width)
def pooling_shape(pool_shape, image_shape, stride):
"""Calculate output shape for pooling layer."""
n_images, n_channels, height, width = image_shape
height = (height - pool_shape[0]) / float(stride[0]) + 1
width = (width - pool_shape[1]) / float(stride[1]) + 1
assert height % 1 == 0
assert width % 1 == 0
return int(height), int(width)
+158
View File
@@ -0,0 +1,158 @@
# coding:utf-8
import numpy as np
from mla.neuralnet.layers import Layer, PhaseMixin, ParamMixin
from mla.neuralnet.parameters import Parameters
"""
References:
https://kratzert.github.io/2016/02/12/understanding-the-gradient-flow-through-the-batch-normalization-layer.html
"""
class BatchNormalization(Layer, ParamMixin, PhaseMixin):
def __init__(self, momentum=0.9, eps=1e-5, parameters=None):
super().__init__()
self._params = parameters
if self._params is None:
self._params = Parameters()
self.momentum = momentum
self.eps = eps
self.ema_mean = None
self.ema_var = None
def setup(self, x_shape):
self._params.setup_weights((1, x_shape[1]))
def _forward_pass(self, X):
gamma = self._params["W"]
beta = self._params["b"]
if self.is_testing:
mu = self.ema_mean
xmu = X - mu
var = self.ema_var
sqrtvar = np.sqrt(var + self.eps)
ivar = 1.0 / sqrtvar
xhat = xmu * ivar
gammax = gamma * xhat
return gammax + beta
N, D = X.shape
# step1: calculate mean
mu = 1.0 / N * np.sum(X, axis=0)
# step2: subtract mean vector of every trainings example
xmu = X - mu
# step3: following the lower branch - calculation denominator
sq = xmu**2
# step4: calculate variance
var = 1.0 / N * np.sum(sq, axis=0)
# step5: add eps for numerical stability, then sqrt
sqrtvar = np.sqrt(var + self.eps)
# step6: invert sqrtwar
ivar = 1.0 / sqrtvar
# step7: execute normalization
xhat = xmu * ivar
# step8: Nor the two transformation steps
gammax = gamma * xhat
# step9
out = gammax + beta
# store running averages of mean and variance during training for use during testing
if self.ema_mean is None or self.ema_var is None:
self.ema_mean = mu
self.ema_var = var
else:
self.ema_mean = self.momentum * self.ema_mean + (1 - self.momentum) * mu
self.ema_var = self.momentum * self.ema_var + (1 - self.momentum) * var
# store intermediate
self.cache = (xhat, gamma, xmu, ivar, sqrtvar, var)
return out
def forward_pass(self, X):
if len(X.shape) == 2:
# input is a regular layer
return self._forward_pass(X)
elif len(X.shape) == 4:
# input is a convolution layer
N, C, H, W = X.shape
x_flat = X.transpose(0, 2, 3, 1).reshape(-1, C)
out_flat = self._forward_pass(x_flat)
return out_flat.reshape(N, H, W, C).transpose(0, 3, 1, 2)
else:
raise NotImplementedError(
"Unknown model with dimensions = {}".format(len(X.shape))
)
def _backward_pass(self, delta):
# unfold the variables stored in cache
xhat, gamma, xmu, ivar, sqrtvar, var = self.cache
# get the dimensions of the input/output
N, D = delta.shape
# step9
dbeta = np.sum(delta, axis=0)
dgammax = delta # not necessary, but more understandable
# step8
dgamma = np.sum(dgammax * xhat, axis=0)
dxhat = dgammax * gamma
# step7
divar = np.sum(dxhat * xmu, axis=0)
dxmu1 = dxhat * ivar
# step6
dsqrtvar = -1.0 / (sqrtvar**2) * divar
# step5
dvar = 0.5 * 1.0 / np.sqrt(var + self.eps) * dsqrtvar
# step4
dsq = 1.0 / N * np.ones((N, D)) * dvar
# step3
dxmu2 = 2 * xmu * dsq
# step2
dx1 = dxmu1 + dxmu2
dmu = -1 * np.sum(dxmu1 + dxmu2, axis=0)
# step1
dx2 = 1.0 / N * np.ones((N, D)) * dmu
# step0
dx = dx1 + dx2
# Update gradient values
self._params.update_grad("W", dgamma)
self._params.update_grad("b", dbeta)
return dx
def backward_pass(self, X):
if len(X.shape) == 2:
# input is a regular layer
return self._backward_pass(X)
elif len(X.shape) == 4:
# input is a convolution layer
N, C, H, W = X.shape
x_flat = X.transpose(0, 2, 3, 1).reshape(-1, C)
out_flat = self._backward_pass(x_flat)
return out_flat.reshape(N, H, W, C).transpose(0, 3, 1, 2)
else:
raise NotImplementedError("Unknown model shape: {}".format(X.shape))
def shape(self, x_shape):
return x_shape
@@ -0,0 +1,3 @@
# coding:utf-8
from .lstm import *
from .rnn import *
+195
View File
@@ -0,0 +1,195 @@
# coding:utf-8
import autograd.numpy as np
from autograd import elementwise_grad
from mla.neuralnet.activations import sigmoid
from mla.neuralnet.initializations import get_initializer
from mla.neuralnet.layers import Layer, get_activation, ParamMixin
from mla.neuralnet.parameters import Parameters
"""
References:
Understanding LSTM Networks http://colah.github.io/posts/2015-08-Understanding-LSTMs/
A Critical Review of Recurrent Neural Networks for Sequence Learning http://arxiv.org/pdf/1506.00019v4.pdf
"""
class LSTM(Layer, ParamMixin):
def __init__(
self,
hidden_dim,
activation="tanh",
inner_init="orthogonal",
parameters=None,
return_sequences=True,
):
self.return_sequences = return_sequences
self.hidden_dim = hidden_dim
self.inner_init = get_initializer(inner_init)
self.activation = get_activation(activation)
self.activation_d = elementwise_grad(self.activation)
self.sigmoid_d = elementwise_grad(sigmoid)
if parameters is None:
self._params = Parameters()
else:
self._params = parameters
self.last_input = None
self.states = None
self.outputs = None
self.gates = None
self.hprev = None
self.input_dim = None
self.W = None
self.U = None
def setup(self, x_shape):
"""
Naming convention:
i : input gate
f : forget gate
c : cell
o : output gate
Parameters
----------
x_shape : np.array(batch size, time steps, input shape)
"""
self.input_dim = x_shape[2]
# Input -> Hidden
W_params = ["W_i", "W_f", "W_o", "W_c"]
# Hidden -> Hidden
U_params = ["U_i", "U_f", "U_o", "U_c"]
# Bias terms
b_params = ["b_i", "b_f", "b_o", "b_c"]
# Initialize params
for param in W_params:
self._params[param] = self._params.init((self.input_dim, self.hidden_dim))
for param in U_params:
self._params[param] = self.inner_init((self.hidden_dim, self.hidden_dim))
for param in b_params:
self._params[param] = np.full((self.hidden_dim,), self._params.initial_bias)
# Combine weights for simplicity
self.W = [self._params[param] for param in W_params]
self.U = [self._params[param] for param in U_params]
# Init gradient arrays for all weights
self._params.init_grad()
self.hprev = np.zeros((x_shape[0], self.hidden_dim))
self.oprev = np.zeros((x_shape[0], self.hidden_dim))
def forward_pass(self, X):
n_samples, n_timesteps, input_shape = X.shape
p = self._params
self.last_input = X
self.states = np.zeros((n_samples, n_timesteps + 1, self.hidden_dim))
self.outputs = np.zeros((n_samples, n_timesteps + 1, self.hidden_dim))
self.gates = {
k: np.zeros((n_samples, n_timesteps, self.hidden_dim))
for k in ["i", "f", "o", "c"]
}
self.states[:, -1, :] = self.hprev
self.outputs[:, -1, :] = self.oprev
for i in range(n_timesteps):
t_gates = np.dot(X[:, i, :], self.W) + np.dot(
self.outputs[:, i - 1, :], self.U
)
# Input
self.gates["i"][:, i, :] = sigmoid(t_gates[:, 0, :] + p["b_i"])
# Forget
self.gates["f"][:, i, :] = sigmoid(t_gates[:, 1, :] + p["b_f"])
# Output
self.gates["o"][:, i, :] = sigmoid(t_gates[:, 2, :] + p["b_o"])
# Cell
self.gates["c"][:, i, :] = self.activation(t_gates[:, 3, :] + p["b_c"])
# (previous state * forget) + input + cell
self.states[:, i, :] = (
self.states[:, i - 1, :] * self.gates["f"][:, i, :]
+ self.gates["i"][:, i, :] * self.gates["c"][:, i, :]
)
self.outputs[:, i, :] = self.gates["o"][:, i, :] * self.activation(
self.states[:, i, :]
)
self.hprev = self.states[:, n_timesteps - 1, :].copy()
self.oprev = self.outputs[:, n_timesteps - 1, :].copy()
if self.return_sequences:
return self.outputs[:, 0:-1, :]
else:
return self.outputs[:, -2, :]
def backward_pass(self, delta):
if len(delta.shape) == 2:
delta = delta[:, np.newaxis, :]
n_samples, n_timesteps, input_shape = delta.shape
# Temporal gradient arrays
grad = {k: np.zeros_like(self._params[k]) for k in self._params.keys()}
dh_next = np.zeros((n_samples, input_shape))
output = np.zeros((n_samples, n_timesteps, self.input_dim))
# Backpropagation through time
for i in reversed(range(n_timesteps)):
dhi = (
delta[:, i, :]
* self.gates["o"][:, i, :]
* self.activation_d(self.states[:, i, :])
+ dh_next
)
og = delta[:, i, :] * self.activation(self.states[:, i, :])
de_o = og * self.sigmoid_d(self.gates["o"][:, i, :])
grad["W_o"] += np.dot(self.last_input[:, i, :].T, de_o)
grad["U_o"] += np.dot(self.outputs[:, i - 1, :].T, de_o)
grad["b_o"] += de_o.sum(axis=0)
de_f = (dhi * self.states[:, i - 1, :]) * self.sigmoid_d(
self.gates["f"][:, i, :]
)
grad["W_f"] += np.dot(self.last_input[:, i, :].T, de_f)
grad["U_f"] += np.dot(self.outputs[:, i - 1, :].T, de_f)
grad["b_f"] += de_f.sum(axis=0)
de_i = (dhi * self.gates["c"][:, i, :]) * self.sigmoid_d(
self.gates["i"][:, i, :]
)
grad["W_i"] += np.dot(self.last_input[:, i, :].T, de_i)
grad["U_i"] += np.dot(self.outputs[:, i - 1, :].T, de_i)
grad["b_i"] += de_i.sum(axis=0)
de_c = (dhi * self.gates["i"][:, i, :]) * self.activation_d(
self.gates["c"][:, i, :]
)
grad["W_c"] += np.dot(self.last_input[:, i, :].T, de_c)
grad["U_c"] += np.dot(self.outputs[:, i - 1, :].T, de_c)
grad["b_c"] += de_c.sum(axis=0)
dh_next = dhi * self.gates["f"][:, i, :]
# TODO: propagate error to the next layer
# Change actual gradient arrays
for k in grad.keys():
self._params.update_grad(k, grad[k])
return output
def shape(self, x_shape):
if self.return_sequences:
return x_shape[0], x_shape[1], self.hidden_dim
else:
return x_shape[0], self.hidden_dim
+110
View File
@@ -0,0 +1,110 @@
# coding:utf-8
import autograd.numpy as np
from autograd import elementwise_grad
from mla.neuralnet.initializations import get_initializer
from mla.neuralnet.layers import Layer, get_activation, ParamMixin
from mla.neuralnet.parameters import Parameters
class RNN(Layer, ParamMixin):
"""Vanilla RNN."""
def __init__(
self,
hidden_dim,
activation="tanh",
inner_init="orthogonal",
parameters=None,
return_sequences=True,
):
self.return_sequences = return_sequences
self.hidden_dim = hidden_dim
self.inner_init = get_initializer(inner_init)
self.activation = get_activation(activation)
self.activation_d = elementwise_grad(self.activation)
if parameters is None:
self._params = Parameters()
else:
self._params = parameters
self.last_input = None
self.states = None
self.hprev = None
self.input_dim = None
def setup(self, x_shape):
"""
Parameters
----------
x_shape : np.array(batch size, time steps, input shape)
"""
self.input_dim = x_shape[2]
# Input -> Hidden
self._params["W"] = self._params.init((self.input_dim, self.hidden_dim))
# Bias
self._params["b"] = np.full((self.hidden_dim,), self._params.initial_bias)
# Hidden -> Hidden layer
self._params["U"] = self.inner_init((self.hidden_dim, self.hidden_dim))
# Init gradient arrays
self._params.init_grad()
self.hprev = np.zeros((x_shape[0], self.hidden_dim))
def forward_pass(self, X):
self.last_input = X
n_samples, n_timesteps, input_shape = X.shape
states = np.zeros((n_samples, n_timesteps + 1, self.hidden_dim))
states[:, -1, :] = self.hprev.copy()
p = self._params
for i in range(n_timesteps):
states[:, i, :] = np.tanh(
np.dot(X[:, i, :], p["W"])
+ np.dot(states[:, i - 1, :], p["U"])
+ p["b"]
)
self.states = states
self.hprev = states[:, n_timesteps - 1, :].copy()
if self.return_sequences:
return states[:, 0:-1, :]
else:
return states[:, -2, :]
def backward_pass(self, delta):
if len(delta.shape) == 2:
delta = delta[:, np.newaxis, :]
n_samples, n_timesteps, input_shape = delta.shape
p = self._params
# Temporal gradient arrays
grad = {k: np.zeros_like(p[k]) for k in p.keys()}
dh_next = np.zeros((n_samples, input_shape))
output = np.zeros((n_samples, n_timesteps, self.input_dim))
# Backpropagation through time
for i in reversed(range(n_timesteps)):
dhi = self.activation_d(self.states[:, i, :]) * (delta[:, i, :] + dh_next)
grad["W"] += np.dot(self.last_input[:, i, :].T, dhi)
grad["b"] += delta[:, i, :].sum(axis=0)
grad["U"] += np.dot(self.states[:, i - 1, :].T, dhi)
dh_next = np.dot(dhi, p["U"].T)
d = np.dot(delta[:, i, :], p["U"].T)
output[:, i, :] = np.dot(d, p["W"].T)
# Change actual gradient arrays
for k in grad.keys():
self._params.update_grad(k, grad[k])
return output
def shape(self, x_shape):
if self.return_sequences:
return x_shape[0], x_shape[1], self.hidden_dim
else:
return x_shape[0], self.hidden_dim