diff --git a/mindspore/nn/layer/__init__.py b/mindspore/nn/layer/__init__.py index 65d4e8589ef..5a0836fe035 100644 --- a/mindspore/nn/layer/__init__.py +++ b/mindspore/nn/layer/__init__.py @@ -17,14 +17,14 @@ Layer. The high-level components(Cells) used to construct the neural network. """ -from . import activation, normalization, container, conv, lstm, basic, embedding, pooling, image, quant, math, \ - combined, timedistributed, thor_layer, rnns +from . import activation, normalization, container, conv, basic, embedding, pooling, image, quant, math, \ + combined, timedistributed, thor_layer, rnns, rnn_cells from .activation import * from .normalization import * from .container import * from .conv import * -from .lstm import * from .rnns import * +from .rnn_cells import * from .basic import * from .embedding import * from .pooling import * @@ -40,7 +40,7 @@ __all__.extend(activation.__all__) __all__.extend(normalization.__all__) __all__.extend(container.__all__) __all__.extend(conv.__all__) -__all__.extend(lstm.__all__) +__all__.extend(rnn_cells.__all__) __all__.extend(rnns.__all__) __all__.extend(basic.__all__) __all__.extend(embedding.__all__) diff --git a/mindspore/nn/layer/lstm.py b/mindspore/nn/layer/lstm.py deleted file mode 100755 index f394686e5a8..00000000000 --- a/mindspore/nn/layer/lstm.py +++ /dev/null @@ -1,400 +0,0 @@ -# Copyright 2020-2021 Huawei Technologies Co., Ltd -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# ============================================================================ -"""lstm""" -import math -import numpy as np -import mindspore.context as context -import mindspore.common.dtype as mstype -from mindspore.ops.primitive import constexpr -from mindspore._checkparam import Validator as validator -from mindspore.common.initializer import initializer -from mindspore.common.parameter import Parameter, ParameterTuple -from mindspore.common.tensor import Tensor -from mindspore.nn.cell import Cell -from mindspore import nn -from mindspore.ops import operations as P -from mindspore.ops import functional as F - - -__all__ = ['LSTM', 'LSTMCell'] - - -@constexpr -def _create_sequence_length(shape): - num_step, batch_size, _ = shape - sequence_length = Tensor(np.ones(batch_size, np.int32) * num_step, mstype.int32) - return sequence_length - - -@constexpr -def _check_input_dtype(input_dtype, param_name, allow_dtypes, cls_name): - validator.check_type_name(param_name, input_dtype, allow_dtypes, cls_name) - - -@constexpr -def _check_input_3d(input_shape, param_name, func_name): - if len(input_shape) != 3: - raise ValueError(f"For '{func_name}', the '{param_name}' should be 3d, but got the length of input_shape:" - f" {len(input_shape)}.") - - -class LSTM(Cell): - r""" - Stacked LSTM (Long Short-Term Memory) layers. - - Apply LSTM layer to the input. - - There are two pipelines connecting two consecutive cells in a LSTM model; one is cell state pipeline - and the other is hidden state pipeline. Denote two consecutive time nodes as :math:`t-1` and :math:`t`. - Given an input :math:`x_t` at time :math:`t`, a hidden state :math:`h_{t-1}` and a cell - state :math:`c_{t-1}` of the layer at time :math:`{t-1}`, the cell state and hidden state at - time :math:`t` is computed using a gating mechanism. Input gate :math:`i_t` is designed to protect the cell - from perturbation by irrelevant inputs. Forget gate :math:`f_t` affords protection of the cell by forgetting - some information in the past, which is stored in :math:`h_{t-1}`. Output gate :math:`o_t` protects other - units from perturbation by currently irrelevant memory contents. Candidate cell state :math:`\tilde{c}_t` is - calculated with the current input, on which the input gate will be applied. Finally, current cell state - :math:`c_{t}` and hidden state :math:`h_{t}` are computed with the calculated gates and cell states. The complete - formulation is as follows. - - .. math:: - \begin{array}{ll} \\ - i_t = \sigma(W_{ix} x_t + b_{ix} + W_{ih} h_{(t-1)} + b_{ih}) \\ - f_t = \sigma(W_{fx} x_t + b_{fx} + W_{fh} h_{(t-1)} + b_{fh}) \\ - \tilde{c}_t = \tanh(W_{cx} x_t + b_{cx} + W_{ch} h_{(t-1)} + b_{ch}) \\ - o_t = \sigma(W_{ox} x_t + b_{ox} + W_{oh} h_{(t-1)} + b_{oh}) \\ - c_t = f_t \odot c_{(t-1)} + i_t \odot \tilde{c}_t \\ - h_t = o_t \odot \tanh(c_t) \\ - \end{array} - - Here :math:`\sigma` is the sigmoid function, and :math:`*` is the Hadamard product. :math:`W, b` - are learnable weights between the output and the input in the formula. For instance, - :math:`W_{ix}, b_{ix}` are the weight and bias used to transform from input :math:`x` to :math:`i`. - Details can be found in paper `LONG SHORT-TERM MEMORY - `_ and - `Long Short-Term Memory Recurrent Neural Network Architectures for Large Scale Acoustic Modeling - `_. - - Args: - input_size (int): Number of features of input. - hidden_size (int): Number of features of hidden layer. - num_layers (int): Number of layers of stacked LSTM . Default: 1. - has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. - batch_first (bool): Specifies whether the first dimension of input `x` is batch_size. Default: False. - dropout (float, int): If not 0, append `Dropout` layer on the outputs of each - LSTM layer except the last layer. Default 0. The range of dropout is [0.0, 1.0]. - bidirectional (bool): Specifies whether it is a bidirectional LSTM. Default: False. - - Inputs: - - **x** (Tensor) - Tensor of shape (seq_len, batch_size, `input_size`) or - (batch_size, seq_len, `input_size`). - - **hx** (tuple) - A tuple of two Tensors (h_0, c_0) both of data type mindspore.float32 or - mindspore.float16 and shape (num_directions * `num_layers`, batch_size, `hidden_size`). - The data type of `hx` must be the same as `x`. - - Outputs: - Tuple, a tuple contains (`output`, (`h_n`, `c_n`)). - - - **output** (Tensor) - Tensor of shape (seq_len, batch_size, num_directions * `hidden_size`). - - **hx_n** (tuple) - A tuple of two Tensor (h_n, c_n) both of shape - (num_directions * `num_layers`, batch_size, `hidden_size`). - - Raises: - TypeError: If `input_size`, `hidden_size` or `num_layers` is not an int. - TypeError: If `has_bias`, `batch_first` or `bidirectional` is not a bool. - TypeError: If `dropout` is neither a float nor an int. - ValueError: If `dropout` is not in range [0.0, 1.0]. - - Supported Platforms: - ``Ascend`` ``GPU`` - - Examples: - >>> net = nn.LSTM(10, 16, 2, has_bias=True, batch_first=True, bidirectional=False) - >>> x = Tensor(np.ones([3, 5, 10]).astype(np.float32)) - >>> h0 = Tensor(np.ones([1 * 2, 3, 16]).astype(np.float32)) - >>> c0 = Tensor(np.ones([1 * 2, 3, 16]).astype(np.float32)) - >>> output, (hn, cn) = net(x, (h0, c0)) - >>> print(output.shape) - (3, 5, 16) - """ - - def __init__(self, - input_size, - hidden_size, - num_layers=1, - has_bias=True, - batch_first=False, - dropout=0, - bidirectional=False): - """Initialize LSTM.""" - super(LSTM, self).__init__() - validator.check_value_type("batch_first", batch_first, [bool], self.cls_name) - validator.check_positive_int(hidden_size, "hidden_size", self.cls_name) - validator.check_positive_int(num_layers, "num_layers", self.cls_name) - self.is_ascend = context.get_context("device_target") == "Ascend" - - self.batch_first = batch_first - self.transpose = P.Transpose() - self.num_layers = num_layers - self.bidirectional = bidirectional - self.dropout = dropout - self.lstm = P.LSTM(input_size=input_size, - hidden_size=hidden_size, - num_layers=num_layers, - has_bias=has_bias, - bidirectional=bidirectional, - dropout=float(dropout)) - - weight_size = 0 - gate_size = 4 * hidden_size - stdv = 1 / math.sqrt(hidden_size) - num_directions = 2 if bidirectional else 1 - if self.is_ascend: - self.reverse_seq = P.ReverseSequence(batch_dim=1, seq_dim=0) - self.concat = P.Concat(axis=0) - self.concat_2dim = P.Concat(axis=2) - self.cast = P.Cast() - self.shape = P.Shape() - if dropout < 0 or dropout > 1: - raise ValueError(f"For '{self.cls_name}', the 'dropout' must be a number in range [0, 1], " - f"but got {dropout}.") - if dropout == 1: - self.dropout_op = P.ZerosLike() - else: - self.dropout_op = nn.Dropout(float(1 - dropout)) - b0 = np.zeros(gate_size, dtype=np.float16) - self.w_list = [] - self.b_list = [] - self.rnns_fw = P.DynamicRNN(forget_bias=0.0) - self.rnns_bw = P.DynamicRNN(forget_bias=0.0) - - for layer in range(num_layers): - w_shape = input_size if layer == 0 else (num_directions * hidden_size) - w_np = np.random.uniform(-stdv, stdv, (w_shape + hidden_size, gate_size)).astype(np.float16) - self.w_list.append(Parameter( - initializer(Tensor(w_np), [w_shape + hidden_size, gate_size]), name='weight_fw' + str(layer))) - if has_bias: - b_np = np.random.uniform(-stdv, stdv, gate_size).astype(np.float16) - self.b_list.append(Parameter(initializer(Tensor(b_np), [gate_size]), name='bias_fw' + str(layer))) - else: - self.b_list.append(Parameter(initializer(Tensor(b0), [gate_size]), name='bias_fw' + str(layer))) - if bidirectional: - w_bw_np = np.random.uniform(-stdv, stdv, (w_shape + hidden_size, gate_size)).astype(np.float16) - self.w_list.append(Parameter(initializer(Tensor(w_bw_np), [w_shape + hidden_size, gate_size]), - name='weight_bw' + str(layer))) - b_bw_np = np.random.uniform(-stdv, stdv, (4 * hidden_size)).astype(np.float16) if has_bias else b0 - self.b_list.append(Parameter(initializer(Tensor(b_bw_np), [gate_size]), - name='bias_bw' + str(layer))) - self.w_list = ParameterTuple(self.w_list) - self.b_list = ParameterTuple(self.b_list) - else: - for layer in range(num_layers): - input_layer_size = input_size if layer == 0 else hidden_size * num_directions - increment_size = gate_size * input_layer_size - increment_size += gate_size * hidden_size - if has_bias: - increment_size += 2 * gate_size - weight_size += increment_size * num_directions - w_np = np.random.uniform(-stdv, stdv, (weight_size, 1, 1)).astype(np.float32) - self.weight = Parameter(initializer(Tensor(w_np), [weight_size, 1, 1]), name='weight') - - def _stacked_bi_dynamic_rnn(self, x, init_h, init_c, weight, bias): - """stacked bidirectional dynamic_rnn""" - x_shape = self.shape(x) - sequence_length = _create_sequence_length(x_shape) - pre_layer = x - hn = () - cn = () - output = x - for i in range(self.num_layers): - offset = i * 2 - weight_fw, weight_bw = weight[offset], weight[offset + 1] - bias_fw, bias_bw = bias[offset], bias[offset + 1] - init_h_fw, init_h_bw = init_h[offset:offset + 1, :, :], init_h[offset + 1:offset + 2, :, :] - init_c_fw, init_c_bw = init_c[offset:offset + 1, :, :], init_c[offset + 1:offset + 2, :, :] - bw_x = self.reverse_seq(pre_layer, sequence_length) - y, h, c, _, _, _, _, _ = self.rnns_fw(pre_layer, weight_fw, bias_fw, None, init_h_fw, init_c_fw) - y_bw, h_bw, c_bw, _, _, _, _, _ = self.rnns_bw(bw_x, weight_bw, bias_bw, None, init_h_bw, init_c_bw) - y_bw = self.reverse_seq(y_bw, sequence_length) - output = self.concat_2dim((y, y_bw)) - pre_layer = self.dropout_op(output) if self.dropout else output - hn += (h[-1:, :, :],) - hn += (h_bw[-1:, :, :],) - cn += (c[-1:, :, :],) - cn += (c_bw[-1:, :, :],) - status_h = self.concat(hn) - status_c = self.concat(cn) - return output, status_h, status_c - - def _stacked_dynamic_rnn(self, x, init_h, init_c, weight, bias): - """stacked mutil_layer dynamic_rnn""" - pre_layer = x - hn = () - cn = () - y = 0 - for i in range(self.num_layers): - weight_fw, bias_bw = weight[i], bias[i] - init_h_fw, init_c_bw = init_h[i:i + 1, :, :], init_c[i:i + 1, :, :] - y, h, c, _, _, _, _, _ = self.rnns_fw(pre_layer, weight_fw, bias_bw, None, init_h_fw, init_c_bw) - pre_layer = self.dropout_op(y) if self.dropout else y - hn += (h[-1:, :, :],) - cn += (c[-1:, :, :],) - status_h = self.concat(hn) - status_c = self.concat(cn) - return y, status_h, status_c - - def construct(self, x, hx): - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - h, c = hx - if self.is_ascend: - x_dtype = F.dtype(x) - h_dtype = F.dtype(h) - c_dtype = F.dtype(c) - _check_input_3d(F.shape(h), "h of hx", self.cls_name) - _check_input_3d(F.shape(c), "c of hx", self.cls_name) - _check_input_dtype(x_dtype, "x", [mstype.float32, mstype.float16], self.cls_name) - _check_input_dtype(h_dtype, "h", [mstype.float32, mstype.float16], self.cls_name) - _check_input_dtype(c_dtype, "c", [mstype.float32, mstype.float16], self.cls_name) - x = self.cast(x, mstype.float16) - h = self.cast(h, mstype.float16) - c = self.cast(c, mstype.float16) - if self.bidirectional: - x, h, c = self._stacked_bi_dynamic_rnn(x, h, c, self.w_list, self.b_list) - else: - x, h, c = self._stacked_dynamic_rnn(x, h, c, self.w_list, self.b_list) - x = self.cast(x, x_dtype) - h = self.cast(h, h_dtype) - c = self.cast(c, c_dtype) - else: - x, h, c, _, _ = self.lstm(x, h, c, self.weight) - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - return x, (h, c) - - -class LSTMCell(Cell): - r""" - LSTM (Long Short-Term Memory) layer. - - Apply LSTM layer to the input. - - There are two pipelines connecting two consecutive cells in a LSTM model; one is cell state pipeline - and the other is hidden state pipeline. Denote two consecutive time nodes as :math:`t-1` and :math:`t`. - Given an input :math:`x_t` at time :math:`t`, a hidden state :math:`h_{t-1}` and a cell - state :math:`c_{t-1}` of the layer at time :math:`{t-1}`, the cell state and hidden state at - time :math:`t` is computed using a gating mechanism. Input gate :math:`i_t` is designed to protect the cell - from perturbation by irrelevant inputs. Forget gate :math:`f_t` affords protection of the cell by forgetting - some information in the past, which is stored in :math:`h_{t-1}`. Output gate :math:`o_t` protects other - units from perturbation by currently irrelevant memory contents. Candidate cell state :math:`\tilde{c}_t` is - calculated with the current input, on which the input gate will be applied. Finally, current cell state - :math:`c_{t}` and hidden state :math:`h_{t}` are computed with the calculated gates and cell states. The complete - formulation is as follows. - - .. math:: - \begin{array}{ll} \\ - i_t = \sigma(W_{ix} x_t + b_{ix} + W_{ih} h_{(t-1)} + b_{ih}) \\ - f_t = \sigma(W_{fx} x_t + b_{fx} + W_{fh} h_{(t-1)} + b_{fh}) \\ - \tilde{c}_t = \tanh(W_{cx} x_t + b_{cx} + W_{ch} h_{(t-1)} + b_{ch}) \\ - o_t = \sigma(W_{ox} x_t + b_{ox} + W_{oh} h_{(t-1)} + b_{oh}) \\ - c_t = f_t * c_{(t-1)} + i_t * \tilde{c}_t \\ - h_t = o_t * \tanh(c_t) \\ - \end{array} - - Here :math:`\sigma` is the sigmoid function, and :math:`*` is the Hadamard product. :math:`W, b` - are learnable weights between the output and the input in the formula. For instance, - :math:`W_{ix}, b_{ix}` are the weight and bias used to transform from input :math:`x` to :math:`i`. - Details can be found in paper `LONG SHORT-TERM MEMORY - `_ and - `Long Short-Term Memory Recurrent Neural Network Architectures for Large Scale Acoustic Modeling - `_. - - Note: - LSTMCell is a single-layer RNN, you can achieve multi-layer RNN by stacking LSTMCell. - - Args: - input_size (int): Number of features of input. - hidden_size (int): Number of features of hidden layer. - has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. - batch_first (bool): Specifies whether the first dimension of input `x` is batch_size. Default: False. - dropout (float, int): If not 0, append `Dropout` layer on the outputs of each - LSTM layer except the last layer. Default 0. The range of dropout is [0.0, 1.0]. - bidirectional (bool): Specifies whether this is a bidirectional LSTM. If set True, - number of directions will be 2 otherwise number of directions is 1. Default: False. - - Inputs: - - **x** (Tensor) - Tensor of shape (seq_len, batch_size, `input_size`). - - **h** - data type mindspore.float32 or - mindspore.float16 and shape (num_directions, batch_size, `hidden_size`). - - **c** - data type mindspore.float32 or - mindspore.float16 and shape (num_directions, batch_size, `hidden_size`). - The data type of `h` and `c` must be the same of `x`. - - **w** - data type mindspore.float32 or - mindspore.float16 and shape (`weight_size`, 1, 1). - The value of `weight_size` depends on `input_size`, `hidden_size` and `bidirectional` - - Outputs: - `output`, `h_n`, `c_n`, 'reserve', 'state'. - - - **output** (Tensor) - Tensor of shape (seq_len, batch_size, num_directions * `hidden_size`). - - **h** - A Tensor with shape (num_directions, batch_size, `hidden_size`). - - **c** - A Tensor with shape (num_directions, batch_size, `hidden_size`). - - **reserve** - reserved - - **state** - reserved - - Raises: - TypeError: If `input_size` or `hidden_size` or `num_layers` is not an int. - TypeError: If `has_bias` or `batch_first` or `bidirectional` is not a bool. - TypeError: If `dropout` is neither a float nor an int. - ValueError: If `dropout` is not in range [0.0, 1.0]. - - Supported Platforms: - ``GPU`` ``CPU`` - - Examples: - >>> net = nn.LSTMCell(10, 12, has_bias=True, batch_first=True, bidirectional=False) - >>> x = Tensor(np.ones([3, 5, 10]).astype(np.float32)) - >>> h = Tensor(np.ones([1, 3, 12]).astype(np.float32)) - >>> c = Tensor(np.ones([1, 3, 12]).astype(np.float32)) - >>> w = Tensor(np.ones([1152, 1, 1]).astype(np.float32)) - >>> output, h, c, _, _ = net(x, h, c, w) - >>> print(output.shape) - (3, 5, 12) - """ - - def __init__(self, - input_size, - hidden_size, - has_bias=True, - batch_first=False, - dropout=0, - bidirectional=False): - """Initialize LSTMCell.""" - super(LSTMCell, self).__init__() - self.batch_first = validator.check_value_type("batch_first", batch_first, [bool], self.cls_name) - self.transpose = P.Transpose() - self.lstm = P.LSTM(input_size=input_size, - hidden_size=hidden_size, - num_layers=1, - has_bias=has_bias, - bidirectional=bidirectional, - dropout=float(dropout)) - - def construct(self, x, h, c, w): - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - x, h, c, _, _ = self.lstm(x, h, c, w) - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - return x, h, c, _, _ diff --git a/mindspore/nn/layer/rnn_cells.py b/mindspore/nn/layer/rnn_cells.py new file mode 100644 index 00000000000..25a4950eadc --- /dev/null +++ b/mindspore/nn/layer/rnn_cells.py @@ -0,0 +1,341 @@ +# Copyright 2021 Huawei Technologies Co., Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================ +'''RNN Cells module, include RNNCell, GRUCell, LSTMCell''' +import math +import numpy as np +import mindspore.ops as P +import mindspore.common.dtype as mstype +from mindspore.common.tensor import Tensor +from mindspore.common.parameter import Parameter +from mindspore.common.initializer import initializer, Uniform +from mindspore.ops.primitive import constexpr +from mindspore.nn.cell import Cell +from mindspore._checkparam import Validator as validator + +__all__ = ['LSTMCell', 'GRUCell', 'RNNCell'] + + +@constexpr +def _check_input_dtype(input_dtype, param_name, allow_dtypes, cls_name): + validator.check_type_name(param_name, input_dtype, allow_dtypes, cls_name) + + +@constexpr +def _check_is_tensor(param_name, input_data, cls_name): + """Internal function, used to check whether the input data is Tensor.""" + if input_data is not None and not isinstance(P.typeof(input_data), mstype.tensor_type): + raise TypeError(f"For '{cls_name}', the '{param_name}' should be '{mstype.tensor_type}', " + f"but got '{P.typeof(input_data)}'") + +@constexpr +def _check_is_tuple(param_name, input_data, cls_name): + """Internal function, used to check whether the input data is Tensor.""" + if input_data is not None and not isinstance({P.typeof(input_data)}, mstype.Tuple): + raise TypeError(f"For '{cls_name}', the '{param_name}' should be '{mstype.Tuple}', " + f"but got '{P.typeof(input_data)}'") + +@constexpr +def _check_tuple_length(param_name, input_data, length, cls_name): + """Internal function, used to check whether the input data is Tensor.""" + if input_data is not None and len(input_data) != length: + raise TypeError(f"For '{cls_name}', the length of '{param_name}' should be '{length}', " + f"but got '{len(input_data)}'") + +@constexpr +def _check_batch_size_equal(batch_size_x, batch_size_hx, cls_name): + if batch_size_x != batch_size_hx: + raise ValueError(f"For '{cls_name}' batch size of x and hx should be equal, but got {batch_size_x} of x " + f"and {batch_size_hx} of hx.") + +def _rnn_tanh_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): + '''RNN cell function with tanh activation''' + if b_ih is None: + igates = P.MatMul(False, True)(inputs, w_ih) + hgates = P.MatMul(False, True)(hidden, w_hh) + else: + igates = P.MatMul(False, True)(inputs, w_ih) + b_ih + hgates = P.MatMul(False, True)(hidden, w_hh) + b_hh + return P.Tanh()(igates + hgates) + +def _rnn_relu_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): + '''RNN cell function with relu activation''' + if b_ih is None: + igates = P.MatMul(False, True)(inputs, w_ih) + hgates = P.MatMul(False, True)(hidden, w_hh) + else: + igates = P.MatMul(False, True)(inputs, w_ih) + b_ih + hgates = P.MatMul(False, True)(hidden, w_hh) + b_hh + return P.ReLU()(igates + hgates) + +def _lstm_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): + '''LSTM cell function''' + hx, cx = hidden + if b_ih is None: + gates = P.MatMul(False, True)(inputs, w_ih) + P.MatMul(False, True)(hx, w_hh) + else: + gates = P.MatMul(False, True)(inputs, w_ih) + P.MatMul(False, True)(hx, w_hh) + b_ih + b_hh + ingate, forgetgate, cellgate, outgate = P.Split(1, 4)(gates) + + ingate = P.Sigmoid()(ingate) + forgetgate = P.Sigmoid()(forgetgate) + cellgate = P.Tanh()(cellgate) + outgate = P.Sigmoid()(outgate) + + cy = (forgetgate * cx) + (ingate * cellgate) + hy = outgate * P.Tanh()(cy) + + return hy, cy + +def _gru_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): + '''GRU cell function''' + if b_ih is None: + gi = P.MatMul(False, True)(inputs, w_ih) + gh = P.MatMul(False, True)(hidden, w_hh) + else: + gi = P.MatMul(False, True)(inputs, w_ih) + b_ih + gh = P.MatMul(False, True)(hidden, w_hh) + b_hh + i_r, i_i, i_n = P.Split(1, 3)(gi) + h_r, h_i, h_n = P.Split(1, 3)(gh) + + resetgate = P.Sigmoid()(i_r + h_r) + inputgate = P.Sigmoid()(i_i + h_i) + newgate = P.Tanh()(i_n + resetgate * h_n) + hy = newgate + inputgate * (hidden - newgate) + + return hy + +class RNNCellBase(Cell): + '''Basic class for RNN Cells''' + def __init__(self, input_size: int, hidden_size: int, has_bias: bool, num_chunks: int): + super().__init__() + validator.check_value_type("has_bias", has_bias, [bool], self.cls_name) + validator.check_positive_int(hidden_size, "hidden_size", self.cls_name) + validator.check_positive_int(input_size, "input_size", self.cls_name) + self.input_size = input_size + self.hidden_size = hidden_size + self.has_bias = has_bias + self.weight_ih = Parameter(Tensor(np.random.randn(num_chunks * hidden_size, input_size).astype(np.float32))) + self.weight_hh = Parameter(Tensor(np.random.randn(num_chunks * hidden_size, hidden_size).astype(np.float32))) + if has_bias: + self.bias_ih = Parameter(Tensor(np.random.randn(num_chunks * hidden_size).astype(np.float32))) + self.bias_hh = Parameter(Tensor(np.random.randn(num_chunks * hidden_size).astype(np.float32))) + else: + self.bias_ih = None + self.bias_hh = None + self.reset_parameters() + + def reset_parameters(self): + stdv = 1 / math.sqrt(self.hidden_size) + for weight in self.get_parameters(): + weight.set_data(initializer(Uniform(stdv), weight.shape)) + +class RNNCell(RNNCellBase): + r""" + An Elman RNN cell with tanh or ReLU non-linearity. + + .. math:: + h_t = \tanh(W_{ih} x_t + b_{ih} + W_{hh} h_{(t-1)} + b_{hh}) + + Here :math:`h_t` is the hidden state at time `t`, :math:`x_t` is + the input at time `t`, and :math:`h_{(t-1)}` is the hidden state of the + previous layer at time `t-1` or the initial hidden state at time `0`. + If `nonlinearity` is `relu`, then `relu` is used instead of `tanh`. + + Args: + input_size (int): Number of features of input. + hidden_size (int): Number of features of hidden layer. + has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. + nonlinearity (str): The non-linearity to use. Can be either `tanh` or `relu`. Default: `tanh`. + + Inputs: + - **x** (Tensor) - Tensor of shape (batch_size, `input_size`). + - **hx** (Tensor) - Tensor of data type mindspore.float32 and shape (batch_size, `hidden_size`). + Data type of `hx` must be the same as `x`. + + Outputs: + - **hx'** (Tensor) - Tensor of shape (batch_size, `hidden_size`). + + Raises: + TypeError: If `input_size` or `hidden_size` is not an int or not greater than 0. + TypeError: If `has_bias` is not a bool. + ValueError: If `nonlinearity` is not in ['tanh', 'relu']. + + Supported Platforms: + ``Ascend`` ``GPU`` ``CPU`` + + Examples: + >>> net = nn.RNNCell(10, 16) + >>> x = Tensor(np.ones([5, 3, 10]).astype(np.float32)) + >>> hx = Tensor(np.ones([3, 16]).astype(np.float32)) + >>> output = [] + >>> for i in range(5): + ... hx = net(x[i], hx) + ... output.append(hx) + >>> print(output[0].shape) + (3, 16) + """ + _non_linearity = ['tanh', 'relu'] + + def __init__(self, input_size: int, hidden_size: int, bias: bool = True, nonlinearity: str = "tanh"): + super().__init__(input_size, hidden_size, bias, num_chunks=1) + if nonlinearity not in self._non_linearity: + raise ValueError("Unknown nonlinearity: {}".format(nonlinearity)) + self.nonlinearity = nonlinearity + + def construct(self, x, hx): + _check_is_tensor('x', x, self.cls_name) + _check_is_tensor('hx', hx, self.cls_name) + _check_input_dtype(x.dtype, "x", [mstype.float32, mstype.float16], self.cls_name) + _check_input_dtype(hx.dtype, "hx", [mstype.float32, mstype.float16], self.cls_name) + _check_batch_size_equal(x.shape[0], hx.shape[0], self.cls_name) + + if self.nonlinearity == "tanh": + ret = _rnn_tanh_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) + else: + ret = _rnn_relu_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) + return ret + +class LSTMCell(RNNCellBase): + r""" + A LSTM (Long Short-Term Memory) cell. + + .. math:: + \begin{array}{ll} \\ + i_t = \sigma(W_{ix} x_t + b_{ix} + W_{ih} h_{(t-1)} + b_{ih}) \\ + f_t = \sigma(W_{fx} x_t + b_{fx} + W_{fh} h_{(t-1)} + b_{fh}) \\ + \tilde{c}_t = \tanh(W_{cx} x_t + b_{cx} + W_{ch} h_{(t-1)} + b_{ch}) \\ + o_t = \sigma(W_{ox} x_t + b_{ox} + W_{oh} h_{(t-1)} + b_{oh}) \\ + c_t = f_t * c_{(t-1)} + i_t * \tilde{c}_t \\ + h_t = o_t * \tanh(c_t) \\ + \end{array} + + Here :math:`\sigma` is the sigmoid function, and :math:`*` is the Hadamard product. :math:`W, b` + are learnable weights between the output and the input in the formula. For instance, + :math:`W_{ix}, b_{ix}` are the weight and bias used to transform from input :math:`x` to :math:`i`. + Details can be found in paper `LONG SHORT-TERM MEMORY + `_ and + `Long Short-Term Memory Recurrent Neural Network Architectures for Large Scale Acoustic Modeling + `_. + + Args: + input_size (int): Number of features of input. + hidden_size (int): Number of features of hidden layer. + has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. + + Inputs: + - **x** (Tensor) - Tensor of shape (batch_size, `input_size`). + - **hx** (tuple) - A tuple of two Tensors (h_0, c_0) both of data type mindspore.float32 + and shape (batch_size, `hidden_size`). The data type of `hx` must be the same as `x`. + + Outputs: + - **hx'** (Tensor) - A tuple of two Tensors (h', c') both of data shape (batch_size, `hidden_size`). + + Raises: + TypeError: If `input_size`, `hidden_size` is not an int. + TypeError: If `has_bias` is not a bool. + + Supported Platforms: + ``Ascend`` ``GPU`` ``CPU`` + + Examples: + >>> net = nn.LSTMCell(10, 16) + >>> x = Tensor(np.ones([5, 3, 10]).astype(np.float32)) + >>> h = Tensor(np.ones([3, 16]).astype(np.float32)) + >>> c = Tensor(np.ones([3, 16]).astype(np.float32)) + >>> output = [] + >>> for i in range(5): + ... hx = net(x[i], (h, c)) + ... output.append(hx) + >>> print(output[0][0].shape) + (3, 16) + """ + def __init__(self, input_size: int, hidden_size: int, bias: bool = True): + super().__init__(input_size, hidden_size, bias, num_chunks=4) + self.support_non_tensor_inputs = True + + def construct(self, x, hx): + _check_is_tensor('x', x, self.cls_name) + _check_is_tuple('hx', hx, self.cls_name) + _check_tuple_length('hx', hx, 2, self.cls_name) + _check_is_tensor('hx[0]', hx[0], self.cls_name) + _check_is_tensor('hx[1]', hx[1], self.cls_name) + _check_input_dtype(x.dtype, "x", [mstype.float32, mstype.float16], self.cls_name) + _check_input_dtype(hx[0].dtype, "hx[0]", [mstype.float32, mstype.float16], self.cls_name) + _check_input_dtype(hx[1].dtype, "hx[1]", [mstype.float32, mstype.float16], self.cls_name) + _check_batch_size_equal(x.shape[0], hx[0].shape[0], self.cls_name) + _check_batch_size_equal(x.shape[0], hx[1].shape[0], self.cls_name) + return _lstm_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) + +class GRUCell(RNNCellBase): + r""" + A GRU(Gated Recurrent Unit) cell. + + .. math:: + + \begin{array}{ll} + r = \sigma(W_{ir} x + b_{ir} + W_{hr} h + b_{hr}) \\ + z = \sigma(W_{iz} x + b_{iz} + W_{hz} h + b_{hz}) \\ + n = \tanh(W_{in} x + b_{in} + r * (W_{hn} h + b_{hn})) \\ + h' = (1 - z) * n + z * h + \end{array} + + Here :math:`\sigma` is the sigmoid function, and :math:`*` is the Hadamard product. :math:`W, b` + are learnable weights between the output and the input in the formula. For instance, + :math:`W_{ir}, b_{ir}` are the weight and bias used to transform from input :math:`x` to :math:`r`. + Details can be found in paper + `Learning Phrase Representations using RNN Encoder–Decoder for Statistical Machine Translation + `_. + + Args: + input_size (int): Number of features of input. + hidden_size (int): Number of features of hidden layer. + has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. + + Inputs: + - **x** (Tensor) - Tensor of shape (batch_size, `input_size`). + - **hx** (Tensor) - Tensor of data type mindspore.float32 and shape (batch_size, `hidden_size`). + Data type of `hx` must be the same as `x`. + + Outputs: + - **hx'** (Tensor) - Tensor of shape (batch_size, `hidden_size`). + + Raises: + TypeError: If `input_size`, `hidden_size` is not an int. + TypeError: If `has_bias` is not a bool. + + Supported Platforms: + ``Ascend`` ``GPU`` ``CPU`` + + Examples: + >>> net = nn.GRUCell(10, 16) + >>> x = Tensor(np.ones([5, 3, 10]).astype(np.float32)) + >>> hx = Tensor(np.ones([3, 16]).astype(np.float32)) + >>> output = [] + >>> for i in range(5): + ... hx = net(x[i], hx) + ... output.append(hx) + >>> print(output[0].shape) + (3, 16) + """ + def __init__(self, input_size: int, hidden_size: int, bias: bool = True): + super().__init__(input_size, hidden_size, bias, num_chunks=3) + + def construct(self, x, hx): + _check_is_tensor('x', x, self.cls_name) + _check_is_tensor('hx', hx, self.cls_name) + _check_input_dtype(x.dtype, "x", [mstype.float32, mstype.float16], self.cls_name) + _check_input_dtype(hx.dtype, "hx", [mstype.float32, mstype.float16], self.cls_name) + _check_batch_size_equal(x.shape[0], hx.shape[0], self.cls_name) + return _gru_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) diff --git a/mindspore/nn/layer/rnn_utils.py b/mindspore/nn/layer/rnn_utils.py new file mode 100644 index 00000000000..533cb99debb --- /dev/null +++ b/mindspore/nn/layer/rnn_utils.py @@ -0,0 +1,81 @@ +# Copyright 2021 Huawei Technologies Co., Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================ +'''Utils for RNNs CPU version, like Reverse operators''' +import numpy as np +import mindspore.common.dtype as mstype +import mindspore.ops as P +from mindspore.ops.primitive import constexpr +from mindspore.nn.cell import Cell +from mindspore.common.tensor import Tensor + +@constexpr +def arange(start, stop, step): + return Tensor(np.arange(start, stop, step), mstype.int32) + +class _Reverse(Cell): + """Reverse operator, like Reverse in mindspore""" + def __init__(self, dim): + super().__init__() + self.dim = dim + + def construct(self, input_x): + dim_size = input_x.shape[self.dim] + reversed_indexes = arange(dim_size-1, -1, -1) + output = P.Gather()(input_x, reversed_indexes, self.dim) + return output + +class _ReverseSequence(Cell): + """Reverse sequence operator, like ReverseSequenceV2 in mindspore""" + def __init__(self, seq_dim, batch_dim=0): + super().__init__() + self.seq_dim = seq_dim + self.batch_dim = batch_dim + + def construct(self, x, seq_lengths): + """Defines the ReverseSequence operator computation performed.""" + batch_size = x.shape[self.batch_dim] + max_seq_len = x.shape[self.seq_dim] + seq_lens_type = seq_lengths.dtype + + back = P.Sub()(seq_lengths, P.OnesLike()(seq_lengths)) + + batch_idx = self.make_shape((batch_size, max_seq_len), seq_lens_type, 0) + forward_idx = self.make_shape((batch_size, max_seq_len), seq_lens_type, 1) + + back = back.view(-1, 1) + reverse_idx = P.Sub()(back, forward_idx) + + condition = P.Less()(reverse_idx, P.ZerosLike()(reverse_idx)) + reverse_idx = P.Select()(condition, forward_idx, reverse_idx) + + reverse_idx = P.ExpandDims()(reverse_idx, 2) + batch_idx = P.ExpandDims()(batch_idx, 2) + + if self.batch_dim > self.seq_dim: + batch_idx = P.Transpose()(batch_idx, (1, 0, 2)) + reverse_idx = P.Transpose()(reverse_idx, (1, 0, 2)) + x = P.Transpose()(x, (1, 0, 2)) + start_indices = P.Concat(2)((batch_idx, reverse_idx)) + + output = P.GatherNd()(x, start_indices) + + return output + + def make_shape(self, shape, dtype, range_dim): + output = P.Ones()(shape, mstype.float32) + output = P.CumSum()(output, range_dim) + output = P.Cast()(output, dtype) + output = output - 1 + return output diff --git a/mindspore/nn/layer/rnns.py b/mindspore/nn/layer/rnns.py index 23e24cedc10..d9069b55fd3 100644 --- a/mindspore/nn/layer/rnns.py +++ b/mindspore/nn/layer/rnns.py @@ -15,18 +15,25 @@ '''RNN operators module, include RNN, GRU''' import math import numpy as np +import mindspore.nn as nn import mindspore.ops as P +import mindspore.context as context import mindspore.common.dtype as mstype from mindspore.ops.primitive import constexpr -from mindspore.common.initializer import initializer, Uniform from mindspore.common.tensor import Tensor from mindspore.common.parameter import ParameterTuple, Parameter from mindspore.nn.cell import Cell -from mindspore import nn from mindspore import log as logger from mindspore._checkparam import Validator as validator +from .rnn_cells import _rnn_relu_cell, _rnn_tanh_cell, _gru_cell, _lstm_cell +from .rnn_utils import _Reverse, _ReverseSequence -__all__ = ['GRU', 'RNN', 'GRUCell', 'RNNCell'] +__all__ = ['LSTM', 'GRU', 'RNN'] + + +@constexpr +def arange(start, stop, step): + return Tensor(np.arange(start, stop, step), mstype.int32) @constexpr @@ -43,12 +50,6 @@ def _check_input_dtype(input_dtype, param_name, allow_dtypes, cls_name): validator.check_type_name(param_name, input_dtype, allow_dtypes, cls_name) -@constexpr -def _check_batch_size_equal(batch_size_x, batch_size_hx, cls_name): - if batch_size_x != batch_size_hx: - raise ValueError(f"For '{cls_name}' batch size of x and hx should be equal, but got {batch_size_x} of x " - f"and {batch_size_hx} of hx.") - @constexpr def _check_is_tensor(param_name, input_data, cls_name): """Internal function, used to check whether the input data is Tensor.""" @@ -56,69 +57,38 @@ def _check_is_tensor(param_name, input_data, cls_name): raise TypeError(f"For '{cls_name}', the '{param_name}' should be '{mstype.tensor_type}', " f"but got '{P.typeof(input_data)}'") +@constexpr +def _check_is_tuple(param_name, input_data, cls_name): + """Internal function, used to check whether the input data is Tensor.""" + if input_data is not None and not isinstance(P.typeof(input_data), mstype.Tuple): + raise TypeError(f"For '{cls_name}', the '{param_name}' should be '{mstype.Tuple}', " + f"but got '{P.typeof(input_data)}'") -def _rnn_tanh_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): - '''RNN cell function with tanh activation''' - if b_ih is None: - igates = P.MatMul(False, True)(inputs, w_ih) - hgates = P.MatMul(False, True)(hidden, w_hh) - else: - igates = P.MatMul(False, True)(inputs, w_ih) + b_ih - hgates = P.MatMul(False, True)(hidden, w_hh) + b_hh - return P.Tanh()(igates + hgates) +@constexpr +def _check_tuple_length(param_name, input_data, length, cls_name): + """Internal function, used to check whether the input data is Tensor.""" + if input_data is not None and len(input_data) != length: + raise TypeError(f"For '{cls_name}', the length of '{param_name}' should be '{length}', " + f"but got '{len(input_data)}'") +def sequence_mask(lengths, maxlen): + """generate mask matrix by seq_length""" + range_vector = arange(0, maxlen, 1) + result = range_vector < lengths.view(lengths.shape + (1,)) + return result.astype(mstype.int32) -def _rnn_relu_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): - '''RNN cell function with relu activation''' - if b_ih is None: - igates = P.MatMul(False, True)(inputs, w_ih) - hgates = P.MatMul(False, True)(hidden, w_hh) - else: - igates = P.MatMul(False, True)(inputs, w_ih) + b_ih - hgates = P.MatMul(False, True)(hidden, w_hh) + b_hh - return P.ReLU()(igates + hgates) +def select_by_mask(inputs, mask): + """mask hiddens by mask matrix""" + return mask.view(mask.shape + (1,)).swapaxes(0, 1) \ + .expand_as(inputs).astype(mstype.bool_) * inputs +def get_hidden(output, seq_length): + """get hidden state by seq_length""" + batch_index = arange(0, seq_length.shape[0], 1) + indices = P.Concat(1)((seq_length.view(-1, 1) - 1, batch_index.view(-1, 1))) + return P.GatherNd()(output, indices) -def _lstm_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): - '''LSTM cell function''' - hx, cx = hidden - if b_ih is None: - gates = P.MatMul(False, True)(inputs, w_ih) + P.MatMul(False, True)(hx, w_hh) - else: - gates = P.MatMul(False, True)(inputs, w_ih) + P.MatMul(False, True)(hx, w_hh) + b_ih + b_hh - ingate, forgetgate, cellgate, outgate = P.Split(1, 4)(gates) - - ingate = P.Sigmoid()(ingate) - forgetgate = P.Sigmoid()(forgetgate) - cellgate = P.Tanh()(cellgate) - outgate = P.Sigmoid()(outgate) - - cy = (forgetgate * cx) + (ingate * cellgate) - hy = outgate * P.Tanh()(cy) - - return hy, cy - - -def _gru_cell(inputs, hidden, w_ih, w_hh, b_ih, b_hh): - '''GRU cell function''' - if b_ih is None: - gi = P.MatMul(False, True)(inputs, w_ih) - gh = P.MatMul(False, True)(hidden, w_hh) - else: - gi = P.MatMul(False, True)(inputs, w_ih) + b_ih - gh = P.MatMul(False, True)(hidden, w_hh) + b_hh - i_r, i_i, i_n = P.Split(1, 3)(gi) - h_r, h_i, h_n = P.Split(1, 3)(gh) - - resetgate = P.Sigmoid()(i_r + h_r) - inputgate = P.Sigmoid()(i_i + h_i) - newgate = P.Tanh()(i_n + resetgate * h_n) - hy = newgate + inputgate * (hidden - newgate) - - return hy - - -class _DynamicRNN(Cell): +class _DynamicRNNBase(Cell): '''Dynamic RNN module to compute RNN cell by timesteps''' def __init__(self, mode): super().__init__() @@ -131,8 +101,7 @@ class _DynamicRNN(Cell): elif mode == "GRU": cell = _gru_cell else: - raise ValueError(f"For '{self.cls_name}', the 'mode' should be in ['RNN_RELU', 'RNN_TANH', 'LSTM', 'GRU'], " - f"but got {mode}.") + raise ValueError("Unrecognized RNN mode: " + mode) self.cell = cell self.is_lstm = mode == "LSTM" @@ -195,11 +164,132 @@ class _DynamicRNN(Cell): return self.recurrent(x, h, w_ih, w_hh, b_ih, b_hh) return self.variable_recurrent(x, h, seq_length, w_ih, w_hh, b_ih, b_hh) +class _DynamicRNNRelu(_DynamicRNNBase): + '''Dynamic RNN module with Relu activation''' + def __init__(self): + mode = 'RNN_RELU' + super().__init__(mode) + +class _DynamicRNNTanh(_DynamicRNNBase): + '''Dynamic RNN module with Tanh activation''' + def __init__(self): + mode = 'RNN_TANH' + super().__init__(mode) + +class _DynamicGRUCPUGPU(_DynamicRNNBase): + '''Dynamic GRU module on CPU and GPU''' + def __init__(self): + mode = 'GRU' + super().__init__(mode) + +class _DynamicGRUAscend(Cell): + '''Dynamic GRU module on Ascend''' + def __init__(self): + super().__init__() + self.gru = P.DynamicGRUV2(gate_order='rzh') + self.transpose = P.Transpose() + self.dtype = mstype.float16 + + def construct(self, x, h_0, seq_length, w_ih, w_hh, b_ih, b_hh): + if b_ih is None: + b_ih = P.Zeros()(w_ih.shape[0], w_ih.dtype) + b_hh = P.Zeros()(w_ih.shape[0], w_ih.dtype) + outputs, _, _, _, _, _ = self.gru(self.cast(x, self.dtype), \ + self.cast(self.transpose(w_ih, (1, 0)), self.dtype), \ + self.cast(self.transpose(w_hh, (1, 0)), self.dtype), \ + self.cast(b_ih, self.dtype), \ + self.cast(b_hh, self.dtype), \ + None, self.cast(h_0, self.dtype)) + if seq_length is not None: + h = get_hidden(outputs, seq_length) + mask = sequence_mask(seq_length, x.shape[0]) + outputs = select_by_mask(outputs, mask) + else: + h = outputs[-1] + return outputs, h + +class _DynamicLSTMCPUGPU(Cell): + '''Dynamic LSTM module on CPU and GPU''' + def __init__(self): + super().__init__() + self.concat = P.Concat() + self.is_gpu = context.get_context("device_target") == "GPU" + + def construct(self, x, h_0, seq_length, w_ih, w_hh, b_ih, b_hh): + gate_size, input_size = w_ih.shape + hidden_size = gate_size // 4 + if self.is_gpu and seq_length is None: + if b_ih is None: + weights = self.concat(( + w_ih.view(-1, 1, 1), + w_hh.view(-1, 1, 1) + )) + has_bias = False + else: + weights = self.concat(( + w_ih.view(-1, 1, 1), + w_hh.view(-1, 1, 1), + b_ih.view(-1, 1, 1), + b_hh.view(-1, 1, 1) + )) + has_bias = True + output, h_n, c_n, _, _ = P.LSTM(input_size, hidden_size, 1, has_bias, False, 0.0)( + x, + h_0[0].view(1, *h_0[0].shape), + h_0[1].view(1, *h_0[1].shape), + weights + ) + else: + output, (h_n, c_n) = _DynamicRNNBase('LSTM')(x, h_0, seq_length, w_ih, w_hh, b_ih, b_hh) + return output, (h_n, c_n) + +class _DynamicLSTMAscend(Cell): + '''Dynamic LSTM module on Ascend''' + def __init__(self): + super().__init__() + self.lstm = P.DynamicRNN() + self.concat_dim1 = P.Concat(axis=1) + self.concat_dim0 = P.Concat(axis=0) + self.transpose = P.Transpose() + self.cast = P.Cast() + self.split = P.Split(axis=0, output_num=4) + self.dtype = mstype.float16 + + def construct(self, x, h_0, seq_length, w_ih, w_hh, b_ih, b_hh): + w_ih_i, w_ih_f, w_ih_g, w_ih_o = self.split(w_ih) + w_hh_i, w_hh_f, w_hh_g, w_hh_o = self.split(w_hh) + w_ih = self.concat_dim0((w_ih_i, w_ih_g, w_ih_f, w_ih_o)) + w_hh = self.concat_dim0((w_hh_i, w_hh_g, w_hh_f, w_hh_o)) + weight = self.concat_dim1((w_ih, w_hh)) + if b_ih is None: + bias = P.Zeros()(w_ih.shape[0], w_ih.dtype) + else: + b_ih_i, b_ih_f, b_ih_g, b_ih_o = self.split(b_ih) + b_hh_i, b_hh_f, b_hh_g, b_hh_o = self.split(b_hh) + bias = self.concat_dim0((b_ih_i + b_hh_i, \ + b_ih_g + b_hh_g, \ + b_ih_f + b_hh_f, \ + b_ih_o + b_hh_o)) + + outputs, h, c, _, _, _, _, _ = self.lstm(self.cast(x, self.dtype), \ + self.cast(self.transpose(weight, (1, 0)), self.dtype), \ + self.cast(bias, self.dtype), None, \ + self.cast(h_0[0].view(1, *h_0[0].shape), self.dtype), \ + self.cast(h_0[1].view(1, *h_0[1].shape), self.dtype)) + if seq_length is not None: + h = get_hidden(h, seq_length) + c = get_hidden(c, seq_length) + mask = sequence_mask(seq_length, x.shape[0]) + outputs = select_by_mask(outputs, mask) + else: + h = h[-1] + c = c[-1] + return outputs, (h, c) class _RNNBase(Cell): '''Basic class for RNN operators''' def __init__(self, mode, input_size, hidden_size, num_layers=1, has_bias=True, - batch_first=False, dropout=0.0, bidirectional=False): + batch_first=False, dropout=0., bidirectional=False): super().__init__() validator.check_positive_int(hidden_size, "hidden_size", self.cls_name) validator.check_positive_int(input_size, "input_size", self.cls_name) @@ -218,20 +308,30 @@ class _RNNBase(Cell): "recurrent layer, so non-zero dropout expects " "num_layers greater than 1, but got dropout={} and " "num_layers={}".format(dropout, num_layers)) + + is_ascend = context.get_context("device_target") == "Ascend" if mode == "LSTM": gate_size = 4 * hidden_size + self.rnn = _DynamicLSTMAscend() if is_ascend else _DynamicLSTMCPUGPU() elif mode == "GRU": gate_size = 3 * hidden_size + self.rnn = _DynamicGRUAscend() if is_ascend else _DynamicGRUCPUGPU() elif mode == "RNN_TANH": gate_size = hidden_size + self.rnn = _DynamicRNNTanh() elif mode == "RNN_RELU": gate_size = hidden_size + self.rnn = _DynamicRNNRelu() else: raise ValueError(f"For '{self.cls_name}', the 'mode' should be in ['RNN_RELU', 'RNN_TANH', 'LSTM', 'GRU'], " f"but got {mode}.") - self.reverse = P.ReverseV2([0]) - self.reverse_sequence = P.ReverseSequence(0, 1) + if context.get_context("device_target") == "CPU": + self.reverse = _Reverse(0) + self.reverse_sequence = _ReverseSequence(0, 1) + else: + self.reverse = P.ReverseV2([0]) + self.reverse_sequence = P.ReverseSequence(0, 1) self.hidden_size = hidden_size self.batch_first = batch_first self.num_layers = num_layers @@ -239,7 +339,6 @@ class _RNNBase(Cell): self.dropout_op = nn.Dropout(float(1 - dropout)) self.bidirectional = bidirectional self.has_bias = has_bias - self.rnn = _DynamicRNN(mode) num_directions = 2 if bidirectional else 1 self.is_lstm = mode == "LSTM" @@ -356,15 +455,26 @@ class _RNNBase(Cell): def construct(self, x, hx=None, seq_length=None): '''Defines the RNN like operators performed''' + _check_is_tensor("x", x, self.cls_name) + _check_input_dtype(x.dtype, "x", [mstype.float32], self.cls_name) + if hx is not None: + if not self.is_lstm: + _check_is_tensor("h", hx, self.cls_name) + _check_input_dtype(hx.dtype, "hx", [mstype.float32], self.cls_name) + else: + _check_is_tuple('hx', hx, self.cls_name) + _check_tuple_length('hx', hx, 2, self.cls_name) + _check_is_tensor('hx[0]', hx[0], self.cls_name) + _check_is_tensor('hx[1]', hx[1], self.cls_name) + _check_input_dtype(hx[0].dtype, "hx[0]", [mstype.float32], self.cls_name) + _check_input_dtype(hx[1].dtype, "hx[1]", [mstype.float32], self.cls_name) + if seq_length is not None: + _check_input_dtype(seq_length.dtype, "seq_length", [mstype.int32, mstype.int64], self.cls_name) max_batch_size = x.shape[0] if self.batch_first else x.shape[1] num_directions = 2 if self.bidirectional else 1 if hx is None: - hx = _init_state((self.num_layers * num_directions, max_batch_size, self.hidden_size), - x.dtype, self.is_lstm) - _check_input_dtype(x.dtype, "x", [mstype.float32], self.cls_name) - _check_input_dtype(hx.dtype, "hx", [mstype.float32], self.cls_name) - if seq_length is not None: - _check_input_dtype(seq_length.dtype, "seq_length", [mstype.int32, mstype.int64], self.cls_name) + hx = _init_state((self.num_layers * num_directions, max_batch_size, self.hidden_size), \ + x.dtype, self.is_lstm) if self.batch_first: x = P.Transpose()(x, (1, 0, 2)) if self.bidirectional: @@ -375,7 +485,6 @@ class _RNNBase(Cell): x = P.Transpose()(x, (1, 0, 2)) return x, h - class RNN(_RNNBase): r""" Stacked Elman RNN layers. @@ -455,7 +564,6 @@ class RNN(_RNNBase): super(RNN, self).__init__(mode, *args, **kwargs) - class GRU(_RNNBase): r""" Stacked GRU (Gated Recurrent Unit) layers. @@ -524,7 +632,7 @@ class GRU(_RNNBase): ValueError: If `dropout` is not in range [0.0, 1.0). Supported Platforms: - ``Ascend`` ``GPU`` + ``Ascend`` ``GPU`` ``CPU`` Examples: >>> net = nn.GRU(10, 16, 2, has_bias=True, batch_first=True, bidirectional=False) @@ -538,157 +646,89 @@ class GRU(_RNNBase): mode = 'GRU' super(GRU, self).__init__(mode, *args, **kwargs) - -class _RNNCellBase(Cell): - '''Basic class for RNN Cells''' - def __init__(self, input_size: int, hidden_size: int, has_bias: bool, num_chunks: int): - super().__init__() - validator.check_value_type("has_bias", has_bias, [bool], self.cls_name) - validator.check_positive_int(hidden_size, "hidden_size", self.cls_name) - validator.check_positive_int(input_size, "input_size", self.cls_name) - self.input_size = input_size - self.hidden_size = hidden_size - self.weight_ih = Parameter(Tensor(np.random.randn(num_chunks * hidden_size, input_size).astype(np.float32))) - self.weight_hh = Parameter(Tensor(np.random.randn(num_chunks * hidden_size, hidden_size).astype(np.float32))) - if has_bias: - self.bias_ih = Parameter(Tensor(np.random.randn(num_chunks * hidden_size).astype(np.float32))) - self.bias_hh = Parameter(Tensor(np.random.randn(num_chunks * hidden_size).astype(np.float32))) - else: - self.bias_ih = None - self.bias_hh = None - self.reset_parameters() - - def reset_parameters(self): - stdv = 1 / math.sqrt(self.hidden_size) - for weight in self.get_parameters(): - weight.set_data(initializer(Uniform(stdv), weight.shape)) - - -class RNNCell(_RNNCellBase): +class LSTM(_RNNBase): r""" - An Elman RNN cell with tanh or ReLU non-linearity. + Stacked LSTM (Long Short-Term Memory) layers. + + Apply LSTM layer to the input. + + There are two pipelines connecting two consecutive cells in a LSTM model; one is cell state pipeline + and the other is hidden state pipeline. Denote two consecutive time nodes as :math:`t-1` and :math:`t`. + Given an input :math:`x_t` at time :math:`t`, an hidden state :math:`h_{t-1}` and an cell + state :math:`c_{t-1}` of the layer at time :math:`{t-1}`, the cell state and hidden state at + time :math:`t` is computed using an gating mechanism. Input gate :math:`i_t` is designed to protect the cell + from perturbation by irrelevant inputs. Forget gate :math:`f_t` affords protection of the cell by forgetting + some information in the past, which is stored in :math:`h_{t-1}`. Output gate :math:`o_t` protects other + units from perturbation by currently irrelevant memory contents. Candidate cell state :math:`\tilde{c}_t` is + calculated with the current input, on which the input gate will be applied. Finally, current cell state + :math:`c_{t}` and hidden state :math:`h_{t}` are computed with the calculated gates and cell states. The complete + formulation is as follows. .. math:: - h_t = \tanh(W_{ih} x_t + b_{ih} + W_{hh} h_{(t-1)} + b_{hh}) - - Here :math:`h_t` is the hidden state at time `t`, :math:`x_t` is - the input at time `t`, and :math:`h_{(t-1)}` is the hidden state of the - previous layer at time `t-1` or the initial hidden state at time `0`. - If `nonlinearity` is `relu`, then `relu` is used instead of `tanh`. - - Args: - input_size (int): Number of features of input. - hidden_size (int): Number of features of hidden layer. - has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. - nonlinearity (str): The non-linearity to use. Can be either `tanh` or `relu`. Default: `tanh`. - - Inputs: - - **x** (Tensor) - Tensor of shape (batch_size, `input_size`). - - **hx** (Tensor) - Tensor of data type mindspore.float32 and shape (batch_size, `hidden_size`). - Data type of `hx` must be the same as `x`. - - Outputs: - - **h'** (Tensor) - Tensor of shape (batch_size, `hidden_size`). - - Raises: - TypeError: If `input_size` or `hidden_size` is not an int or not greater than 0. - TypeError: If `has_bias` is not a bool. - ValueError: If `nonlinearity` is not in ['tanh', 'relu']. - - Supported Platforms: - ``Ascend`` ``GPU`` - - Examples: - >>> net = nn.RNNCell(10, 16) - >>> x = Tensor(np.ones([5, 3, 10]).astype(np.float32)) - >>> hx = Tensor(np.ones([3, 16]).astype(np.float32)) - >>> output = [] - >>> for i in range(5): - ... hx = net(x[i], hx) - ... output.append(hx) - >>> print(output[0].shape) - (3, 16) - """ - _non_linearity = ['tanh', 'relu'] - def __init__(self, input_size: int, hidden_size: int, has_bias: bool = True, nonlinearity: str = "tanh"): - super().__init__(input_size, hidden_size, has_bias, num_chunks=1) - validator.check_value_type("nonlinearity", nonlinearity, [str], self.cls_name) - validator.check_string(nonlinearity, self._non_linearity, "nonlinearity", self.cls_name) - self.nonlinearity = nonlinearity - - def construct(self, x, hx): - _check_is_tensor('x', x, self.cls_name) - _check_is_tensor('hx', hx, self.cls_name) - _check_input_dtype(x.dtype, "x", [mstype.float32], self.cls_name) - _check_input_dtype(hx.dtype, "hx", [mstype.float32], self.cls_name) - _check_batch_size_equal(x.shape[0], hx.shape[0], self.cls_name) - - if self.nonlinearity == "tanh": - ret = _rnn_tanh_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) - else: - ret = _rnn_relu_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) - return ret - - -class GRUCell(_RNNCellBase): - r""" - A GRU(Gated Recurrent Unit) cell. - - .. math:: - - \begin{array}{ll} - r = \sigma(W_{ir} x + b_{ir} + W_{hr} h + b_{hr}) \\ - z = \sigma(W_{iz} x + b_{iz} + W_{hz} h + b_{hz}) \\ - n = \tanh(W_{in} x + b_{in} + r * (W_{hn} h + b_{hn})) \\ - h' = (1 - z) * n + z * h + \begin{array}{ll} \\ + i_t = \sigma(W_{ix} x_t + b_{ix} + W_{ih} h_{(t-1)} + b_{ih}) \\ + f_t = \sigma(W_{fx} x_t + b_{fx} + W_{fh} h_{(t-1)} + b_{fh}) \\ + \tilde{c}_t = \tanh(W_{cx} x_t + b_{cx} + W_{ch} h_{(t-1)} + b_{ch}) \\ + o_t = \sigma(W_{ox} x_t + b_{ox} + W_{oh} h_{(t-1)} + b_{oh}) \\ + c_t = f_t * c_{(t-1)} + i_t * \tilde{c}_t \\ + h_t = o_t * \tanh(c_t) \\ \end{array} Here :math:`\sigma` is the sigmoid function, and :math:`*` is the Hadamard product. :math:`W, b` are learnable weights between the output and the input in the formula. For instance, - :math:`W_{ir}, b_{ir}` are the weight and bias used to transform from input :math:`x` to :math:`r`. - Details can be found in paper - `Learning Phrase Representations using RNN Encoder–Decoder for Statistical Machine Translation - `_. + :math:`W_{ix}, b_{ix}` are the weight and bias used to transform from input :math:`x` to :math:`i`. + Details can be found in paper `LONG SHORT-TERM MEMORY + `_ and + `Long Short-Term Memory Recurrent Neural Network Architectures for Large Scale Acoustic Modeling + `_. Args: input_size (int): Number of features of input. hidden_size (int): Number of features of hidden layer. + num_layers (int): Number of layers of stacked LSTM . Default: 1. has_bias (bool): Whether the cell has bias `b_ih` and `b_hh`. Default: True. + batch_first (bool): Specifies whether the first dimension of input `x` is batch_size. Default: False. + dropout (float, int): If not 0, append `Dropout` layer on the outputs of each + LSTM layer except the last layer. Default 0. The range of dropout is [0.0, 1.0]. + bidirectional (bool): Specifies whether it is a bidirectional LSTM. Default: False. Inputs: - - **x** (Tensor) - Tensor of shape (batch_size, `input_size`). - - **hx** (Tensor) - Tensor of data type mindspore.float32 and shape (batch_size, `hidden_size`). + - **x** (Tensor) - Tensor of shape (seq_len, batch_size, `input_size`) or + (batch_size, seq_len, `input_size`). + - **hx** (tuple) - A tuple of two Tensors (h_0, c_0) both of data type mindspore.float32 or + mindspore.float16 and shape (num_directions * `num_layers`, batch_size, `hidden_size`). Data type of `hx` must be the same as `x`. + - **seq_length** (Tensor) - The length of each sequence in a input batch. + Tensor of shape :math:`(\text{batch_size})`. Default: None. + This input indicates the real sequence length before padding to avoid padded elements + have been used to compute hidden state and affect the final output. It is recommend to + use this input when **x** has padding elements. Outputs: - - **h'** (Tensor) - Tensor of shape (batch_size, `hidden_size`). + Tuple, a tuple contains (`output`, (`h_n`, `c_n`)). + + - **output** (Tensor) - Tensor of shape (seq_len, batch_size, num_directions * `hidden_size`). + - **hx_n** (tuple) - A tuple of two Tensor (h_n, c_n) both of shape + (num_directions * `num_layers`, batch_size, `hidden_size`). Raises: - TypeError: If `input_size`, `hidden_size` is not an int. - TypeError: If `has_bias` is not a bool. + TypeError: If `input_size`, `hidden_size` or `num_layers` is not an int. + TypeError: If `has_bias`, `batch_first` or `bidirectional` is not a bool. + TypeError: If `dropout` is neither a float nor an int. + ValueError: If `dropout` is not in range [0.0, 1.0]. Supported Platforms: - ``Ascend`` ``GPU`` + ``Ascend`` ``GPU`` ``CPU`` Examples: - >>> net = nn.GRUCell(10, 16) - >>> x = Tensor(np.ones([5, 3, 10]).astype(np.float32)) - >>> hx = Tensor(np.ones([3, 16]).astype(np.float32)) - >>> output = [] - >>> for i in range(5): - ... hx = net(x[i], hx) - ... output.append(hx) - >>> print(output[0].shape) - (3, 16) + >>> net = nn.LSTM(10, 16, 2, has_bias=True, batch_first=True, bidirectional=False) + >>> x = Tensor(np.ones([3, 5, 10]).astype(np.float32)) + >>> h0 = Tensor(np.ones([1 * 2, 3, 16]).astype(np.float32)) + >>> c0 = Tensor(np.ones([1 * 2, 3, 16]).astype(np.float32)) + >>> output, (hn, cn) = net(x, (h0, c0)) + >>> print(output.shape) + (3, 5, 16) """ - def __init__(self, input_size: int, hidden_size: int, has_bias: bool = True): - super().__init__(input_size, hidden_size, has_bias, num_chunks=3) - - def construct(self, x, hx): - _check_is_tensor('x', x, self.cls_name) - _check_is_tensor('hx', hx, self.cls_name) - _check_input_dtype(x.dtype, "x", [mstype.float32], self.cls_name) - _check_input_dtype(hx.dtype, "hx", [mstype.float32], self.cls_name) - _check_batch_size_equal(x.shape[0], hx.shape[0], self.cls_name) - - return _gru_cell(x, hx, self.weight_ih, self.weight_hh, self.bias_ih, self.bias_hh) + def __init__(self, *args, **kwargs): + mode = 'LSTM' + super(LSTM, self).__init__(mode, *args, **kwargs) diff --git a/tests/st/ops/ascend/test_gru_op.py b/tests/st/ops/ascend/test_gru_op.py index a8ae7f15581..a25e58720d4 100644 --- a/tests/st/ops/ascend/test_gru_op.py +++ b/tests/st/ops/ascend/test_gru_op.py @@ -177,5 +177,5 @@ def test_sit_gru_grad_input_3_32_32_is_32_hs_16(): x_grad_pynative = out_grad_pynative[0].asnumpy() h_grad_pynative = out_grad_pynative[1].asnumpy() - assert np.allclose(x_grad, x_grad_pynative, 0.0001, 0.0001) - assert np.allclose(h_grad, h_grad_pynative, 0.0001, 0.0001) + assert np.allclose(x_grad, x_grad_pynative, 0.001, 0.001) + assert np.allclose(h_grad, h_grad_pynative, 0.001, 0.001) diff --git a/tests/st/ops/ascend/test_lstm_op.py b/tests/st/ops/ascend/test_lstm_op.py index 2645940e80a..686e65331e1 100644 --- a/tests/st/ops/ascend/test_lstm_op.py +++ b/tests/st/ops/ascend/test_lstm_op.py @@ -19,7 +19,6 @@ import numpy as np from mindspore import context from mindspore import nn from mindspore import Tensor -from mindspore.common.initializer import initializer from mindspore.common.parameter import ParameterTuple from mindspore.common.parameter import Parameter from mindspore.ops import composite as c @@ -48,44 +47,45 @@ class LSTM(nn.Cell): class LSTMWeightBias(): - def __init__(self, num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional): + def __init__(self, num_layers, has_bias, input_size, num_directions, hidden_size, bidirectional): self.num_layers = num_layers self.has_bias = has_bias - self.input_s = input_s + self.input_size = input_size self.num_directions = num_directions - self.hidden_s = hidden_s + self.hidden_size = hidden_size self.bidirectional = bidirectional def get_weight_bias(self): - stdv = 1 / math.sqrt(self.hidden_s) - gate_size = 4 * self.hidden_s - w_list_value = [] - b_list_value = [] + gate_size = 4 * self.hidden_size - for i in range(self.num_layers): - b0 = np.zeros(gate_size, dtype=np.float16) - w_shape = self.input_s if i == 0 else (self.num_directions * self.hidden_s) - w_np = np.random.uniform(-stdv, stdv, (w_shape + self.hidden_s, gate_size)).astype(np.float16) - w_list_value.append(Parameter(initializer(Tensor(w_np), [w_shape + self.hidden_s, gate_size]), - name="weight_fw" + str(i))) - - if self.has_bias: - b_np = np.random.uniform(-stdv, stdv, gate_size).astype(np.float16) - b_list_value.append(Parameter(initializer(Tensor(b_np), [gate_size]), name="bias_fw" + str(i))) - else: - b_list_value.append(Parameter(initializer(Tensor(b0), [gate_size]), name="bias_fw" + str(i))) - - if self.bidirectional: - w_bw_np = np.random.uniform(-stdv, stdv, (w_shape + self.hidden_s, gate_size)).astype(np.float16) - b_list_value.append(Parameter(initializer(Tensor(w_bw_np), [w_shape + self.hidden_s, gate_size]), - name="weight_bw" + str(i))) - b_bw_np = np.random.uniform(-stdv, stdv, (4 * self.hidden_s)).astype( - np.float16) if self.has_bias else b0 - b_list_value.append(Parameter(initializer(Tensor(b_bw_np), [gate_size]), name="bias_bw" + str(i))) - w_list_value = ParameterTuple(w_list_value) - b_list_value = ParameterTuple(b_list_value) - return w_list_value, b_list_value + w_ih_list = [] + w_hh_list = [] + b_ih_list = [] + b_hh_list = [] + stdv = 1 / math.sqrt(self.hidden_size) + for layer in range(self.num_layers): + for direction in range(self.num_directions): + layer_input_size = self.input_size if layer == 0 else self.hidden_size * self.num_directions + suffix = '_reverse' if direction == 1 else '' + w_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, layer_input_size)).astype(np.float32)), + name='weight_ih_l{}{}'.format(layer, suffix))) + w_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, self.hidden_size)).astype(np.float32)), + name='weight_hh_l{}{}'.format(layer, suffix))) + if self.has_bias: + b_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_ih_l{}{}'.format(layer, suffix))) + b_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_hh_l{}{}'.format(layer, suffix))) + w_ih_list = ParameterTuple(w_ih_list) + w_hh_list = ParameterTuple(w_hh_list) + b_ih_list = ParameterTuple(b_ih_list) + b_hh_list = ParameterTuple(b_hh_list) + return w_ih_list, w_hh_list, b_ih_list, b_hh_list @pytest.mark.level0 @pytest.mark.platform_arm_ascend_training @@ -100,7 +100,7 @@ def test_sit_lstm_forward_input_3_32_32_is_32_hs_16(): num_directions = 1 fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) - w_list_value, b_list_value = fact.get_weight_bias() + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) @@ -110,23 +110,27 @@ def test_sit_lstm_forward_input_3_32_32_is_32_hs_16(): context.set_context(mode=context.GRAPH_MODE) net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, bidirectional=bidirectional, dropout=0.0) - net.lstm.w_list = w_list_value - net.lstm.b_list = b_list_value + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list out, (hy, cy) = net(input_ms, h0, c0) # pynative mode context.set_context(mode=context.PYNATIVE_MODE) net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, bidirectional=bidirectional, dropout=0.0) - net_pynative.lstm.w_list = w_list_value - net_pynative.lstm.b_list = b_list_value + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list out_pynative, (hy_pynative, cy_pynative) = net_pynative(input_ms, h0, c0) + context.set_context(mode=context.GRAPH_MODE) assert np.allclose(out.asnumpy(), out_pynative.asnumpy(), 0.0001, 0.0001) assert np.allclose(hy.asnumpy(), hy_pynative.asnumpy(), 0.0001, 0.0001) assert np.allclose(cy.asnumpy(), cy_pynative.asnumpy(), 0.0001, 0.0001) - @pytest.mark.level0 @pytest.mark.platform_arm_ascend_training @pytest.mark.platform_x86_ascend_training @@ -140,7 +144,7 @@ def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): num_directions = 1 fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) - w_list_value, b_list_value = fact.get_weight_bias() + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) @@ -150,8 +154,10 @@ def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): context.set_context(mode=context.GRAPH_MODE) net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, bidirectional=bidirectional, dropout=0.0) - net.lstm.w_list = w_list_value - net.lstm.b_list = b_list_value + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list grad_net_inp = GradOfAllInputsAndParams(net, sens_param=False) grad_net_inp.set_train() @@ -164,8 +170,10 @@ def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): context.set_context(mode=context.PYNATIVE_MODE) net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, bidirectional=bidirectional, dropout=0.0) - net_pynative.lstm.w_list = w_list_value - net_pynative.lstm.b_list = b_list_value + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list grad_net_inp_pynative = GradOfAllInputsAndParams(net_pynative, sens_param=False) grad_net_inp_pynative.set_train() @@ -173,7 +181,8 @@ def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): x_grad_pynative = out_grad_pynative[0].asnumpy() h_grad_pynative = out_grad_pynative[1].asnumpy() c_grad_pynative = out_grad_pynative[2].asnumpy() + context.set_context(mode=context.GRAPH_MODE) - assert np.allclose(x_grad, x_grad_pynative, 0.0001, 0.0001) - assert np.allclose(h_grad, h_grad_pynative, 0.0001, 0.0001) - assert np.allclose(c_grad, c_grad_pynative, 0.0001, 0.0001) + assert np.allclose(x_grad, x_grad_pynative, 0.001, 0.001) + assert np.allclose(h_grad, h_grad_pynative, 0.001, 0.001) + assert np.allclose(c_grad, c_grad_pynative, 0.001, 0.001) diff --git a/tests/st/ops/cpu/test_lstm_op.py b/tests/st/ops/cpu/test_lstm_op.py index e7687e4f3e4..eabef438330 100644 --- a/tests/st/ops/cpu/test_lstm_op.py +++ b/tests/st/ops/cpu/test_lstm_op.py @@ -1,4 +1,4 @@ -# Copyright 2020 Huawei Technologies Co., Ltd +# Copyright 2021 Huawei Technologies Co., Ltd # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -11,373 +11,186 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -# ============================================================================ -import math +# ============================================================================== +import math import pytest import numpy as np -import mindspore.nn as nn -import mindspore.context as context -from mindspore.common.api import ms_function -from mindspore.common.initializer import initializer -from mindspore.ops import composite as C -from mindspore.ops import operations as P -from mindspore.common.tensor import Tensor -from mindspore.common.parameter import ParameterTuple, Parameter - -context.set_context(mode=context.GRAPH_MODE, device_target='CPU') +from mindspore import context +from mindspore import nn +from mindspore import Tensor +from mindspore.common.parameter import ParameterTuple +from mindspore.common.parameter import Parameter +from mindspore.ops import composite as c -class StackLSTM(nn.Cell): - """ - Stack multi-layers LSTM together. - """ +class GradOfAllInputsAndParams(nn.Cell): + def __init__(self, network, sens_param): + super().__init__() + self.grad = c.GradOperation(get_all=True, get_by_list=True, sens_param=sens_param) + self.network = network + self.params = ParameterTuple(self.network.trainable_params()) - def __init__(self, - input_size, - hidden_size, - num_layers=1, - has_bias=True, - batch_first=False, - dropout=0.0, - bidirectional=False): - super(StackLSTM, self).__init__() + def construct(self, *inputs): + gout = self.grad(self.network, self.params)(*inputs) + return gout + + +class LSTM(nn.Cell): + def __init__(self, input_s, hidden_s, num_layers, has_bias, batch_first, bidirectional, dropout): + super().__init__() + self.lstm = nn.LSTM(input_size=input_s, hidden_size=hidden_s, num_layers=num_layers, has_bias=has_bias, + batch_first=batch_first, bidirectional=bidirectional, dropout=dropout) + + def construct(self, inp, h0, c0): + return self.lstm(inp, (h0, c0)) + + +class LSTMWeightBias(): + def __init__(self, num_layers, has_bias, input_size, num_directions, hidden_size, bidirectional): self.num_layers = num_layers - self.batch_first = batch_first - self.transpose = P.Transpose() + self.has_bias = has_bias + self.input_size = input_size + self.num_directions = num_directions + self.hidden_size = hidden_size + self.bidirectional = bidirectional - # direction number - num_directions = 2 if bidirectional else 1 + def get_weight_bias(self): + gate_size = 4 * self.hidden_size - # input_size list - input_size_list = [input_size] - for i in range(num_layers - 1): - input_size_list.append(hidden_size * num_directions) - - # layers - layers = [] - for i in range(num_layers): - layers.append(nn.LSTMCell(input_size=input_size_list[i], - hidden_size=hidden_size, - has_bias=has_bias, - batch_first=batch_first, - bidirectional=bidirectional, - dropout=dropout)) - - # weights - weights = [] - for i in range(num_layers): - # weight size - weight_size = (input_size_list[i] + hidden_size) * num_directions * hidden_size * 4 - if has_bias: - bias_size = num_directions * hidden_size * 4 - weight_size = weight_size + bias_size - - # numpy weight - stdv = 1 / math.sqrt(hidden_size) - w_np = np.random.uniform(-stdv, stdv, (weight_size, 1, 1)).astype(np.float32) - - # lstm weight - weights.append(Parameter(initializer(Tensor(w_np), w_np.shape), name="weight" + str(i))) - - # - self.lstms = layers - self.weight = ParameterTuple(tuple(weights)) - - def construct(self, x, hx): - """construct""" - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - # stack lstm - h, c = hx - hn = cn = None - for i in range(self.num_layers): - x, hn, cn, _, _ = self.lstms[i](x, h[i], c[i], self.weight[i]) - if self.batch_first: - x = self.transpose(x, (1, 0, 2)) - return x, (hn, cn) - - -class LstmNet(nn.Cell): - def __init__(self, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(LstmNet, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = StackLSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - input_np = np.array([[[0.6755, -1.6607, 0.1367], [0.4276, -0.7850, -0.3758]], - [[-0.6424, -0.6095, 0.6639], [0.7918, 0.4147, -0.5089]], - [[-1.5612, 0.0120, -0.7289], [-0.6656, -0.6626, -0.5883]], - [[-0.9667, -0.6296, -0.7310], [0.1026, -0.6821, -0.4387]], - [[-0.4710, 0.6558, -0.3144], [-0.8449, -0.2184, -0.1806]] - ]).astype(np.float32) - self.x = Tensor(input_np) - - self.h = Tensor(np.array([0., 0., 0., 0.]).reshape((num_directions, batch_size, hidden_size)).astype( - np.float32)) - - self.c = Tensor(np.array([0., 0., 0., 0.]).reshape((num_directions, batch_size, hidden_size)).astype( - np.float32)) - self.h = tuple((self.h,)) - self.c = tuple((self.c,)) - wih = np.array([[3.4021e-01, -4.6622e-01, 4.5117e-01], - [-6.4257e-02, -2.4807e-01, 1.3550e-02], # i - [-3.2140e-01, 5.5578e-01, 6.3589e-01], - [1.6547e-01, -7.9030e-02, -2.0045e-01], - [-6.9863e-01, 5.9773e-01, -3.9062e-01], - [-3.0253e-01, -1.9464e-01, 7.0591e-01], - [-4.0835e-01, 3.6751e-01, 4.7989e-01], - [-5.6894e-01, -5.0359e-01, 4.7491e-01]]).astype(np.float32).reshape([1, -1]) - whh = np.array([[-0.4820, -0.2350], - [-0.1195, 0.0519], - [0.2162, -0.1178], - [0.6237, 0.0711], - [0.4511, -0.3961], - [-0.5962, 0.0906], - [0.1867, -0.1225], - [0.1831, 0.0850]]).astype(np.float32).reshape([1, -1]) - bih = np.zeros((1, 8)).astype(np.float32) - w_np = np.concatenate((wih, whh, bih), axis=1).reshape([-1, 1, 1]) - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - self.lstm.weight = ParameterTuple((self.w,)) - - @ms_function - def construct(self): - return self.lstm(self.x, (self.h, self.c)) + w_ih_list = [] + w_hh_list = [] + b_ih_list = [] + b_hh_list = [] + stdv = 1 / math.sqrt(self.hidden_size) + for layer in range(self.num_layers): + for direction in range(self.num_directions): + layer_input_size = self.input_size if layer == 0 else self.hidden_size * self.num_directions + suffix = '_reverse' if direction == 1 else '' + w_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, layer_input_size)).astype(np.float32)), + name='weight_ih_l{}{}'.format(layer, suffix))) + w_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, self.hidden_size)).astype(np.float32)), + name='weight_hh_l{}{}'.format(layer, suffix))) + if self.has_bias: + b_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_ih_l{}{}'.format(layer, suffix))) + b_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_hh_l{}{}'.format(layer, suffix))) + w_ih_list = ParameterTuple(w_ih_list) + w_hh_list = ParameterTuple(w_hh_list) + b_ih_list = ParameterTuple(b_ih_list) + b_hh_list = ParameterTuple(b_hh_list) + return w_ih_list, w_hh_list, b_ih_list, b_hh_list @pytest.mark.level0 @pytest.mark.platform_x86_cpu @pytest.mark.env_onecard -def test_lstm(): - seq_len = 5 - batch_size = 2 - input_size = 3 - hidden_size = 2 - num_layers = 1 +def test_sit_lstm_forward_input_3_32_32_is_32_hs_16(): + """ + Feature: LSTM forward + Description: LSTM with input (3, 32, 32) + Expectation: Graph mode equal to pynative mode + """ + input_s = 32 + hidden_s = 16 has_bias = True bidirectional = False - dropout = 0.0 + num_layers = 1 num_directions = 1 - if bidirectional: - num_directions = 2 - net = LstmNet(batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - y, (h, c) = net() - print(y) - print(c) - print(h) - expect_y = [[[-0.17992045, 0.07819052], - [-0.10745212, -0.06291768]], - [[-0.28830513, 0.30579978], - [-0.07570618, -0.08868407]], + fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() - [[-0.00814095, 0.16889746], - [0.02814853, -0.11208838]], + h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + input_ms = Tensor(np.random.randn(3, 32, 32).astype(np.float32)) - [[0.08157863, 0.06088024], - [-0.04227093, -0.11514835]], + # graph mode + context.set_context(mode=context.GRAPH_MODE) + net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list + out, (hy, cy) = net(input_ms, h0, c0) - [[0.18908429, -0.02963362], - [0.09106826, -0.00602506]]] - expect_h = [[[0.18908429, -0.02963362], - [0.09106826, -0.00602506]]] - expect_c = [[[0.3434288, -0.06561527], - [0.16838229, -0.00972614]]] + # pynative mode + context.set_context(mode=context.PYNATIVE_MODE) + net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list + out_pynative, (hy_pynative, cy_pynative) = net_pynative(input_ms, h0, c0) + context.set_context(mode=context.GRAPH_MODE) - diff_y = y.asnumpy() - expect_y - error_y = np.ones([seq_len, batch_size, hidden_size]) * 1.0e-4 - assert np.all(diff_y < error_y) - assert np.all(-diff_y < error_y) - diff_h = h.asnumpy() - expect_h - error_h = np.ones([num_layers * num_directions, batch_size, hidden_size]) * 1.0e-4 - assert np.all(diff_h < error_h) - assert np.all(-diff_h < error_h) - diff_c = c.asnumpy() - expect_c - error_c = np.ones([num_layers * num_directions, batch_size, hidden_size]) * 1.0e-4 - assert np.all(diff_c < error_c) - assert np.all(-diff_c < error_c) + assert np.allclose(out.asnumpy(), out_pynative.asnumpy(), 0.0001, 0.0001) + assert np.allclose(hy.asnumpy(), hy_pynative.asnumpy(), 0.0001, 0.0001) + assert np.allclose(cy.asnumpy(), cy_pynative.asnumpy(), 0.0001, 0.0001) - -class MultiLayerBiLstmNet(nn.Cell): - def __init__(self, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(MultiLayerBiLstmNet, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = StackLSTM(input_size=input_size, hidden_size=hidden_size, num_layers=num_layers, has_bias=has_bias, - bidirectional=bidirectional, dropout=dropout) - - input_np = np.array([[[-0.1887, -0.4144, -0.0235, 0.7489, 0.7522, 0.5969, 0.3342, 1.2198, 0.6786, -0.9404], - [-0.8643, -1.6835, -2.4965, 2.8093, 0.1741, 0.2707, 0.7387, -0.0939, -1.7990, 0.4765]], - - [[-0.5963, -1.2598, -0.7226, 1.1365, -1.7320, -0.7302, 0.1221, -0.2111, -1.6173, -0.0706], - [0.8964, 0.1737, -1.0077, -0.1389, 0.4889, 0.4391, 0.7911, 0.3614, -1.9533, -0.9936]], - - [[0.3260, -1.3312, 0.0601, 1.0726, -1.6010, -1.8733, -1.5775, 1.1579, -0.8801, -0.5742], - [-2.2998, -0.6344, -0.5409, -0.9221, -0.6500, 0.1206, 1.5215, 0.7517, 1.3691, 2.0021]], - - [[-0.1245, -0.3690, 2.1193, 1.3852, -0.1841, -0.8899, -0.3646, -0.8575, -0.3131, 0.2026], - [1.0218, -1.4331, 0.1744, 0.5442, -0.7808, 0.2527, 0.1566, 1.1484, -0.7766, -0.6747]], - - [[-0.6752, 0.9906, -0.4973, 0.3471, -0.1202, -0.4213, 2.0213, 0.0441, 0.9016, 1.0365], - [1.2223, -1.3248, 0.1207, -0.8256, 0.1816, 0.7057, -0.3105, 0.5713, 0.2804, - -1.0685]]]).astype(np.float32) - - self.x = Tensor(input_np) - - self.h0 = Tensor(np.ones((num_directions, batch_size, hidden_size)).astype(np.float32)) - self.c0 = Tensor(np.ones((num_directions, batch_size, hidden_size)).astype(np.float32)) - self.h1 = Tensor(np.ones((num_directions, batch_size, hidden_size)).astype(np.float32)) - self.c1 = Tensor(np.ones((num_directions, batch_size, hidden_size)).astype(np.float32)) - - self.h = tuple((self.h0, self.h1)) - self.c = tuple((self.c0, self.c1)) - input_size_list = [input_size, hidden_size * num_directions] - weights = [] - bias_size = 0 if not has_bias else num_directions * hidden_size * 4 - for i in range(num_layers): - weight_size = (input_size_list[i] + hidden_size) * num_directions * hidden_size * 4 - w_np = np.ones([weight_size, 1, 1]).astype(np.float32) * 0.02 - if has_bias: - bias_np = np.zeros([bias_size, 1, 1]).astype(np.float32) - w_np = np.concatenate([w_np, bias_np], axis=0) - weights.append(Parameter(initializer(Tensor(w_np), w_np.shape), name='weight' + str(i))) - self.lstm.weight = weights - - @ms_function - def construct(self): - return self.lstm(self.x, (self.h, self.c)) - - -@pytest.mark.level1 +@pytest.mark.level0 @pytest.mark.platform_x86_cpu @pytest.mark.env_onecard -def test_multi_layer_bilstm(): - batch_size = 2 - input_size = 10 - hidden_size = 2 - num_layers = 2 - has_bias = True - bidirectional = True - dropout = 0.0 - - net = MultiLayerBiLstmNet(batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, - dropout) - y, (h, c) = net() - print(y) - print(h) - print(c) - - -class Grad(nn.Cell): - def __init__(self, network): - super(Grad, self).__init__() - self.network = network - self.weights = ParameterTuple(network.trainable_params()) - self.grad = C.GradOperation(get_by_list=True, - sens_param=True) - - @ms_function - def construct(self, output_grad): - weights = self.weights - grads = self.grad(self.network, weights)(output_grad) - return grads - - -class Net(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(Net, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - input_np = np.array([[[0.6755, -1.6607, 0.1367], [0.4276, -0.7850, -0.3758]], - [[-0.6424, -0.6095, 0.6639], [0.7918, 0.4147, -0.5089]], - [[-1.5612, 0.0120, -0.7289], [-0.6656, -0.6626, -0.5883]], - [[-0.9667, -0.6296, -0.7310], [0.1026, -0.6821, -0.4387]], - [[-0.4710, 0.6558, -0.3144], [-0.8449, -0.2184, -0.1806]] - ]).astype(np.float32) - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - self.hlist = [] - self.clist = [] - self.hlist.append(Parameter(initializer( - Tensor( - np.array([0.1, 0.1, 0.1, 0.1]).reshape((num_directions, batch_size, hidden_size)).astype( - np.float32)), - [num_directions, batch_size, hidden_size]), name='h')) - self.clist.append(Parameter(initializer( - Tensor( - np.array([0.2, 0.2, 0.2, 0.2]).reshape((num_directions, batch_size, hidden_size)).astype( - np.float32)), - [num_directions, batch_size, hidden_size]), name='c')) - self.h = ParameterTuple(tuple(self.hlist)) - self.c = ParameterTuple(tuple(self.clist)) - wih = np.array([[3.4021e-01, -4.6622e-01, 4.5117e-01], - [-6.4257e-02, -2.4807e-01, 1.3550e-02], # i - [-3.2140e-01, 5.5578e-01, 6.3589e-01], - [1.6547e-01, -7.9030e-02, -2.0045e-01], - [-6.9863e-01, 5.9773e-01, -3.9062e-01], - [-3.0253e-01, -1.9464e-01, 7.0591e-01], - [-4.0835e-01, 3.6751e-01, 4.7989e-01], - [-5.6894e-01, -5.0359e-01, 4.7491e-01]]).astype(np.float32).reshape([1, -1]) - whh = np.array([[-0.4820, -0.2350], - [-0.1195, 0.0519], - [0.2162, -0.1178], - [0.6237, 0.0711], - [0.4511, -0.3961], - [-0.5962, 0.0906], - [0.1867, -0.1225], - [0.1831, 0.0850]]).astype(np.float32).reshape([1, -1]) - bih = np.zeros((1, 8)).astype(np.float32) - w_np = np.concatenate((wih, whh, bih), axis=1).reshape([-1, 1, 1]) - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='weight0') - self.lstm = StackLSTM(input_size=input_size, hidden_size=hidden_size, num_layers=num_layers, - has_bias=has_bias, bidirectional=bidirectional, dropout=dropout) - self.lstm.weight = ParameterTuple(tuple([self.w])) - - @ms_function - def construct(self): - return self.lstm(self.x, (self.h, self.c))[0] - - -@pytest.mark.level1 -@pytest.mark.platform_x86_cpu -@pytest.mark.env_onecard -def test_grad(): - seq_len = 5 - batch_size = 2 - input_size = 3 - hidden_size = 2 - num_layers = 1 +def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): + """ + Feature: LSTM backward + Description: LSTM with input (3, 32, 32) + Expectation: Graph mode equal to pynative mode + """ + input_s = 32 + hidden_s = 16 has_bias = True bidirectional = False - dropout = 0.0 - net = Grad(Net(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout)) - dy = np.array([[[-3.5471e-01, 7.0540e-01], - [2.7161e-01, 1.0865e+00]], + num_layers = 1 + num_directions = 1 - [[-4.2431e-01, 1.4955e+00], - [-4.0418e-01, -2.3282e-01]], + fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() - [[-1.3654e+00, 1.9251e+00], - [-4.6481e-01, 1.3138e+00]], + h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + input_ms = Tensor(np.random.randn(3, 32, 32).astype(np.float32)) - [[1.2914e+00, -2.3753e-01], - [5.3589e-01, -1.0981e-01]], + # graph mode + context.set_context(mode=context.GRAPH_MODE) + net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list - [[-1.6032e+00, -1.8818e-01], - [1.0065e-01, 9.2045e-01]]]).astype(np.float32) - dx, dhx, dcx, dw = net(Tensor(dy)) - print(dx) - print(dhx) - print(dcx) - print(dw) + grad_net_inp = GradOfAllInputsAndParams(net, sens_param=False) + grad_net_inp.set_train() + out_grad, _ = grad_net_inp(input_ms, h0, c0) + x_grad = out_grad[0].asnumpy() + h_grad = out_grad[1].asnumpy() + c_grad = out_grad[2].asnumpy() -test_multi_layer_bilstm() -test_lstm() -test_grad() + # pynative mode + context.set_context(mode=context.PYNATIVE_MODE) + net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list + + grad_net_inp_pynative = GradOfAllInputsAndParams(net_pynative, sens_param=False) + grad_net_inp_pynative.set_train() + out_grad_pynative, _ = grad_net_inp_pynative(input_ms, h0, c0) + x_grad_pynative = out_grad_pynative[0].asnumpy() + h_grad_pynative = out_grad_pynative[1].asnumpy() + c_grad_pynative = out_grad_pynative[2].asnumpy() + context.set_context(mode=context.GRAPH_MODE) + + assert np.allclose(x_grad, x_grad_pynative, 0.001, 0.001) + assert np.allclose(h_grad, h_grad_pynative, 0.001, 0.001) + assert np.allclose(c_grad, c_grad_pynative, 0.001, 0.001) diff --git a/tests/st/ops/gpu/test_lstm_op.py b/tests/st/ops/gpu/test_lstm_op.py index c5800db6224..194a07e79af 100644 --- a/tests/st/ops/gpu/test_lstm_op.py +++ b/tests/st/ops/gpu/test_lstm_op.py @@ -1,4 +1,4 @@ -# Copyright 2019 Huawei Technologies Co., Ltd +# Copyright 2021 Huawei Technologies Co., Ltd # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -11,972 +11,186 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -# ============================================================================ +# ============================================================================== -import numpy as np +import math import pytest - -import mindspore.context as context -import mindspore.nn as nn -from mindspore.common.api import ms_function -from mindspore.common.initializer import initializer -from mindspore.common.parameter import ParameterTuple, Parameter -from mindspore.common.tensor import Tensor -from mindspore.ops import composite as C -from mindspore.ops import operations as P - -context.set_context(mode=context.PYNATIVE_MODE, device_target='GPU') - - -class LstmNet(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(LstmNet, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = P.LSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - - input_np = np.array([[[0.6755, -1.6607, 0.1367, -0.9209, -1.7088, 0.3953, 2.7120, 0.1103, 0.1504, -0.3611], - [0.4276, -0.7850, -0.3758, 0.8604, -0.1361, -1.3618, -0.6251, -0.8391, 0.8142, 0.4068]], - - [[-0.6424, -0.6095, 0.6639, -0.7253, 2.1190, -0.2840, 0.3858, 0.1691, 0.6764, 1.2903], - [0.7918, 0.4147, -0.5089, -0.3582, -1.4279, -0.7975, -0.0390, -0.4718, 0.4322, -0.7995]], - - [[-1.5612, 0.0120, -0.7289, -1.2479, -0.6197, -0.6099, 0.9543, 0.4362, -1.3141, 0.4273], - [-0.6656, -0.6626, -0.5883, -0.6922, 0.5512, 1.7031, -1.2812, -0.2004, -0.9224, 0.4106]], - - [[-0.9667, -0.6296, -0.7310, 1.2503, -0.1650, 1.2050, -0.1704, -0.5215, 0.1595, 0.3904], - [0.1026, -0.6821, -0.4387, -1.1637, -0.5000, 0.0590, 0.5219, -0.6835, 2.4406, 0.7135]], - - [[-0.4710, 0.6558, -0.3144, -1.2213, 0.1556, -0.3836, -0.1081, -0.1440, -1.1231, 0.6279], - [-0.8449, -0.2184, -0.1806, -0.0615, -0.5660, -0.3556, 1.6891, -1.0286, 1.3361, - -0.4313]]]).astype(np.float32) - - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - - self.h = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='h') - - self.c = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='c') - - wih = np.array([[3.4021e-01, -4.6622e-01, 4.5117e-01, 2.3627e-01, 3.7844e-01, - 2.8770e-01, 4.1631e-01, -6.2628e-01, -4.8008e-01, -4.9148e-01], - [-6.4257e-02, -2.4807e-01, 1.3550e-02, 6.8946e-01, -1.2608e-02, - -7.1719e-02, -1.3566e-01, -4.9215e-01, 2.8509e-01, -6.3540e-01], - [-6.9863e-01, 5.9773e-01, -3.9062e-01, -7.6151e-02, 5.6803e-04, - -7.0420e-01, -6.1822e-01, 4.1854e-01, 4.0596e-01, 6.4867e-01], - [-3.0253e-01, -1.9464e-01, 7.0591e-01, 4.9368e-01, -5.9758e-01, - 1.3251e-02, 3.5685e-01, -3.7640e-01, -4.4612e-01, 5.1794e-01], - [-3.2140e-01, 5.5578e-01, 6.3589e-01, -6.4249e-01, 5.7258e-01, - 2.4256e-01, -2.7954e-01, 2.5202e-01, 2.9235e-01, -3.9979e-01], - [1.6547e-01, -7.9030e-02, -2.0045e-01, 6.2484e-01, -1.0727e-01, - -5.0010e-01, -2.9165e-01, -1.7620e-01, 1.5939e-01, -2.2744e-01], - [-4.0835e-01, 3.6751e-01, 4.7989e-01, 5.8886e-01, 5.3598e-01, - -2.9055e-01, -2.8129e-01, 6.0219e-01, 4.9193e-01, 3.3115e-01], - [-5.6894e-01, -5.0359e-01, 4.7491e-01, 5.8110e-01, -5.4921e-01, - -6.1343e-01, -5.8236e-02, -3.7682e-01, 4.8338e-01, -2.1551e-01]]).astype(np.float32).reshape( - [1, -1]) - - whh = np.array([[-0.4820, -0.2350], - [-0.1195, 0.0519], - [0.4511, -0.3961], - [-0.5962, 0.0906], - [0.2162, -0.1178], - [0.6237, 0.0711], - [0.1867, -0.1225], - [0.1831, 0.0850]]).astype(np.float32).reshape([1, -1]) - - bih = np.array([-0.2862, 0.0034, 0.2059, -0.6544, 0.3244, -0.2472, 0.0852, -0.3050]).astype(np.float32).reshape( - [1, -1]) - bhh = np.array([-0.6575, 0.1562, -0.6434, 0.0212, -0.2493, -0.5626, 0.1530, -0.5235]).astype( - np.float32).reshape([1, -1]) - - w_np = np.concatenate((wih, whh, bih, bhh), axis=1).reshape([-1, 1, 1]) - - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - - @ms_function - def construct(self): - return self.lstm(self.x, self.h, self.c, self.w) - - -@pytest.mark.level0 -@pytest.mark.platform_x86_gpu_training -@pytest.mark.env_onecard -def test_lstm(): - seq_len = 5 - batch_size = 2 - - input_size = 10 - hidden_size = 2 - num_layers = 1 - has_bias = True - bidirectional = False - dropout = 0.0 - - num_directions = 1 - if bidirectional: - num_directions = 2 - - net = LstmNet(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - y, h, c, _, _ = net() - expect_y = np.array([[[-2.1429e-02, 1.1760e-01], - [3.1144e-01, 6.3090e-01]], - - [[-5.0190e-04, -4.5812e-02], - [2.0324e-02, 2.0392e-01]], - - [[-1.0370e-02, -6.0141e-02], - [6.0931e-02, -1.8913e-02]], - - [[-1.6031e-01, -2.3428e-01], - [4.1886e-02, -2.2162e-01]], - - [[-3.9243e-02, -3.2950e-02], - [-4.1257e-02, -4.5276e-01]]]) - - error = np.ones([num_layers, batch_size, hidden_size]) * 1.0e-4 - diff = y.asnumpy() - expect_y - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_h = np.array([[[-0.0392, -0.0329], - [-0.0413, -0.4528]]]) - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-4 - diff = h.asnumpy() - expect_h - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_c = np.array([[[-0.0984, -0.3665], - [-0.1010, -0.6792]]]) - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-4 - diff = c.asnumpy() - expect_c - assert np.all(diff < error) - assert np.all(-diff < error) - - -class BiLstmNet(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(BiLstmNet, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = P.LSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - - input_np = np.array([[[-1.7322, 1.6642, -1.1861, 0.2955, -0.7907, 0.2982, -1.3413, 1.0665, -0.0436, -0.1883], - [0.2195, 0.5917, -0.6739, 0.2388, -0.5364, -1.3309, -0.6018, -0.3081, -0.9648, -1.1627]], - - [[-0.5094, -2.6025, -0.9302, -1.1937, 0.6501, -0.1903, -0.0661, 0.1080, 0.9829, -0.2280], - [1.3961, 0.2239, -0.1947, -0.3206, 0.5791, 0.3396, 0.1728, -1.2007, -1.0994, -1.3278]], - - [[0.1870, -1.1090, -0.9705, 0.2207, 0.3743, 0.1158, -0.5443, -0.5559, 0.1538, -0.3975], - [-0.2347, -0.1245, -0.2335, 0.3164, 1.0997, -0.3928, -1.8517, 1.1136, -1.5051, -0.0071]], - - [[1.2739, 2.5438, -0.4289, -0.7981, -1.3682, -2.2509, 0.2028, 1.3410, 2.9502, -1.1650], - [0.1254, 0.2726, 0.0251, 0.9323, 0.7315, 0.8231, -0.2123, -0.6885, 0.9893, -0.2047]], - - [[0.1870, -0.9066, 0.7155, 0.5438, -0.9757, -0.5828, -0.3417, 1.5681, 1.0326, -0.0179], - [-0.7746, -1.0695, -0.5278, 2.5307, -0.1002, -1.5773, 0.7717, 1.0266, -0.0798, - 1.2333]]]).astype(np.float32) - - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - - self.h = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='h') - - self.c = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='c') - - wih = np.array([[-0.2959, -0.1142, 0.3662, 0.5406, 0.1738, 0.2697, -0.6960, -0.0464, 0.3486, 0.1888], - [0.3043, 0.1505, -0.1207, -0.2456, 0.2735, 0.6673, -0.3352, -0.6153, -0.5731, -0.2726], - [-0.2657, -0.5570, 0.6785, -0.1861, -0.0652, 0.5757, 0.6442, -0.4068, -0.3260, 0.7054], - [0.6607, 0.6927, -0.1354, 0.2484, 0.2053, 0.5743, -0.0212, 0.3340, -0.5685, -0.5668], - [0.6701, -0.3013, -0.1202, -0.4200, -0.4280, -0.6329, -0.6074, -0.4997, -0.6215, -0.6259], - [0.0299, -0.6071, -0.4683, -0.3363, -0.0044, -0.0007, 0.2700, 0.0202, -0.2880, -0.6869], - [0.3025, -0.2461, -0.5128, 0.6327, -0.1438, -0.5100, 0.1924, 0.2023, 0.3129, 0.2271], - [0.3777, 0.0546, 0.4790, -0.1895, 0.3588, 0.4490, 0.6850, 0.6240, -0.2739, -0.4474]]).astype( - np.float32).reshape([1, -1]) - - whh = np.array([[0.6346, -0.6366], - [-0.0248, -0.6156], - [-0.3821, 0.6327], - [-0.6132, -0.5071], - [0.4029, 0.0906], - [-0.5671, 0.2556], - [0.0268, -0.4347], - [0.1152, -0.3124]]).astype(np.float32).reshape([1, -1]) - - bih = np.array([-0.3839, -0.5365, -0.6691, 0.1697, -0.1564, -0.0451, -0.5921, -0.5367]).astype( - np.float32).reshape([1, -1]) - bhh = np.array([0.5952, -0.4905, 0.0423, -0.0293, -0.6638, 0.4348, -0.4291, -0.5541]).astype( - np.float32).reshape([1, -1]) - - wih_reverse = np.array([[-0.2938, 0.0048, 0.2704, -0.3387, -0.4529, -0.2586, 0.1352, -0.1208, -0.1423, -0.0220], - [-0.3701, 0.0201, -0.0255, 0.1340, -0.1938, -0.7056, -0.2303, 0.4814, 0.3636, -0.5018], - [-0.0284, -0.0108, -0.5788, 0.2389, 0.2604, 0.6774, -0.5525, 0.6265, -0.6126, 0.3197], - [-0.6906, 0.6991, -0.6138, 0.0044, 0.5714, 0.4176, 0.5451, -0.5114, -0.2286, 0.1105], - [0.3547, 0.6233, -0.4543, -0.6799, 0.1109, 0.5601, 0.0212, 0.6926, 0.0597, -0.4383], - [-0.1370, -0.5852, 0.0596, 0.5494, 0.5789, -0.0534, 0.1092, 0.3544, -0.1571, 0.4444], - [-0.5886, -0.4765, -0.3837, -0.6634, 0.0963, -0.1385, -0.0837, -0.1354, 0.0547, - -0.2870], - [0.2049, -0.7057, -0.1736, 0.4724, 0.1957, -0.3037, 0.4626, -0.6465, 0.4575, - 0.4230]]).astype(np.float32).reshape([1, -1]) - - whh_reverse = np.array([[0.2339, -0.0307], - [-0.5850, 0.6328], - [0.5856, -0.5601], - [0.4875, -0.6929], - [0.0314, 0.2531], - [-0.2523, 0.3244], - [0.5199, 0.5146], - [0.3968, 0.4511]]).astype(np.float32).reshape([1, -1]) - - bih_reverse = np.array([-0.1760, 0.2828, 0.2450, -0.4016, -0.4664, 0.4031, -0.1945, -0.1509]).astype( - np.float32).reshape([1, -1]) - bhh_reverse = np.array([0.6427, 0.4806, 0.6278, 0.1596, 0.0038, -0.3418, 0.0549, -0.3900]).astype( - np.float32).reshape([1, -1]) - - w_np = np.concatenate((wih, whh, wih_reverse, whh_reverse, bih, bhh, bih_reverse, bhh_reverse), axis=1).reshape( - [-1, 1, 1]) - - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - - @ms_function - def construct(self): - return self.lstm(self.x, self.h, self.c, self.w) - - -@pytest.mark.level0 -@pytest.mark.platform_x86_gpu_training -@pytest.mark.env_onecard -def test_bilstm(): - seq_len = 5 - batch_size = 2 - - input_size = 10 - hidden_size = 2 - num_layers = 1 - has_bias = True - bidirectional = True - dropout = 0.0 - - num_directions = 1 - if bidirectional: - num_directions = 2 - - net = BiLstmNet(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - y, h, c, _, _ = net() - expect_y = np.array([[[-0.0826, 0.0209, 0.1715, -0.0072], - [0.1035, 0.0594, -0.0867, -0.1077]], - - [[-0.1647, 0.0293, -0.2189, 0.3809], - [0.0466, 0.4461, 0.0784, 0.0905]], - - [[-0.0182, 0.0512, 0.1758, -0.1147], - [0.0460, 0.1588, -0.0314, 0.0886]], - - [[-0.0330, 0.0551, 0.2084, -0.1154], - [-0.1641, 0.1118, -0.0122, 0.4916]], - - [[-0.2997, 0.0223, 0.1328, 0.3377], - [-0.6669, 0.0089, 0.1138, 0.7786]]]) - - error = np.ones([num_layers, batch_size, hidden_size * num_directions]) * 1.0e-4 - diff = y.asnumpy() - expect_y - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_h = np.array([[[-0.2997, 0.0223], - [-0.6669, 0.0089]], - - [[0.1715, -0.0072], - [-0.0867, -0.1077]]]) - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-4 - diff = h.asnumpy() - expect_h - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_c = np.array([[[-0.6049, 0.0825], - [-0.9433, 0.1006]], - - [[0.3037, -0.2036], - [-0.1633, -0.5663]]]) - - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-3 - diff = c.asnumpy() - expect_c - assert np.all(diff < error) - assert np.all(-diff < error) - - -class MultiLayerBiLstmNet(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(MultiLayerBiLstmNet, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = P.LSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - - input_np = np.array([[[-0.1887, -0.4144, -0.0235, 0.7489, 0.7522, 0.5969, 0.3342, 1.2198, 0.6786, -0.9404], - [-0.8643, -1.6835, -2.4965, 2.8093, 0.1741, 0.2707, 0.7387, -0.0939, -1.7990, 0.4765]], - - [[-0.5963, -1.2598, -0.7226, 1.1365, -1.7320, -0.7302, 0.1221, -0.2111, -1.6173, -0.0706], - [0.8964, 0.1737, -1.0077, -0.1389, 0.4889, 0.4391, 0.7911, 0.3614, -1.9533, -0.9936]], - - [[0.3260, -1.3312, 0.0601, 1.0726, -1.6010, -1.8733, -1.5775, 1.1579, -0.8801, -0.5742], - [-2.2998, -0.6344, -0.5409, -0.9221, -0.6500, 0.1206, 1.5215, 0.7517, 1.3691, 2.0021]], - - [[-0.1245, -0.3690, 2.1193, 1.3852, -0.1841, -0.8899, -0.3646, -0.8575, -0.3131, 0.2026], - [1.0218, -1.4331, 0.1744, 0.5442, -0.7808, 0.2527, 0.1566, 1.1484, -0.7766, -0.6747]], - - [[-0.6752, 0.9906, -0.4973, 0.3471, -0.1202, -0.4213, 2.0213, 0.0441, 0.9016, 1.0365], - [1.2223, -1.3248, 0.1207, -0.8256, 0.1816, 0.7057, -0.3105, 0.5713, 0.2804, - -1.0685]]]).astype(np.float32) - - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - - self.h = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='h') - - self.c = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='c') - - wih_l0 = np.array([[0.3715, -0.0723, 0.6017, 0.5115, -0.5357, 0.3794, -0.3752, -0.6205, -0.0370, -0.2904], - [0.7055, -0.4156, -0.3650, -0.0964, 0.4141, -0.2584, -0.4765, -0.0045, 0.2943, -0.2648], - [0.1355, 0.1697, 0.1883, 0.3754, 0.3744, -0.6128, 0.2328, -0.1275, 0.6604, 0.6498], - [-0.0266, 0.5805, -0.5358, -0.0929, 0.0797, 0.3744, 0.3299, -0.3825, 0.5804, -0.0855], - [0.1141, 0.2587, -0.4370, 0.6430, -0.0017, 0.4865, 0.2814, 0.6213, -0.6415, 0.4574], - [-0.3958, -0.5827, -0.1056, 0.6987, -0.6591, -0.1326, 0.5237, 0.4667, -0.7001, -0.2326], - [0.3074, -0.3118, -0.4591, 0.2481, -0.2978, -0.1850, 0.4770, -0.0126, 0.3655, -0.4306], - [0.3033, -0.6264, -0.6551, 0.0069, -0.5238, -0.3950, 0.5681, -0.4931, -0.6258, - 0.4079]]).astype(np.float32).reshape([1, -1]) - - whh_l0 = np.array([[-0.3870, 0.0238], - [-0.3758, 0.2490], - [0.5437, -0.4117], - [0.1181, -0.2043], - [-0.5335, 0.1188], - [-0.0822, 0.2154], - [0.5844, -0.3239], - [-0.6537, 0.0278]]).astype(np.float32).reshape([1, -1]) - - bih_l0 = np.array([0.5440, 0.5995, 0.0155, -0.6254, 0.5114, 0.3364, -0.1824, -0.6262]).astype( - np.float32).reshape([1, -1]) - bhh_l0 = np.array([0.4139, -0.2513, -0.4023, 0.4222, 0.6387, -0.6147, 0.0677, 0.5355]).astype( - np.float32).reshape([1, -1]) - - wih_reverse_l0 = np.array([[6.5219e-01, 5.6162e-01, -1.8653e-01, 6.8789e-01, 1.3240e-01, 1.7699e-01, 1.2940e-01, - -1.8520e-01, -5.5439e-01, -3.4946e-01], - [3.7645e-01, 6.5475e-01, 3.5964e-01, 2.2433e-01, -1.7869e-01, -2.9047e-01, - 1.7615e-01, -5.3353e-01, -7.4204e-02, -2.5270e-01], - [5.8095e-01, -4.6426e-04, 1.9262e-01, -5.1306e-01, -3.6811e-01, 4.4858e-01, - 6.2580e-01, 9.5494e-02, -6.9505e-01, 4.9500e-01], - [-3.7810e-01, 1.5485e-01, -1.4735e-01, -1.5327e-01, -4.5702e-01, 3.0816e-01, - -3.4280e-01, 2.1604e-01, 1.4087e-01, -5.7707e-01], - [-3.8700e-01, -6.4653e-01, 6.0653e-01, -4.7297e-01, 6.8413e-02, -1.2681e-01, - 6.8464e-02, 6.7011e-01, 3.9950e-01, -2.0577e-01], - [-1.8648e-01, -6.7198e-01, 3.8017e-01, -3.3147e-01, 5.3193e-01, -5.4952e-01, - 2.1774e-01, -4.6271e-01, 3.2611e-01, 6.3554e-02], - [-4.5403e-01, -1.5910e-01, -7.5886e-02, 2.6313e-01, 6.8093e-01, -3.9960e-01, - 5.5428e-01, 1.0429e-01, 5.1322e-01, 1.9406e-01], - [3.9698e-01, -5.2101e-01, 5.1372e-01, -3.9866e-01, 1.0115e-01, -4.1290e-02, - -3.0980e-01, 2.1607e-01, 4.8420e-01, -1.9267e-01]]).astype(np.float32).reshape( - [1, -1]) - - whh_reverse_l0 = np.array([[-0.3231, -0.3960], - [-0.1625, -0.3032], - [0.3892, -0.0666], - [0.0159, -0.4870], - [-0.4953, 0.2278], - [-0.5380, -0.5250], - [0.0371, -0.4534], - [-0.5452, 0.5012]]).astype(np.float32).reshape([1, -1]) - - bih_reverse_l0 = np.array([0.0469, -0.0107, 0.3783, -0.2657, -0.0089, 0.5032, -0.0757, -0.2022]).astype( - np.float32).reshape([1, -1]) - bhh_reverse_l0 = np.array([-0.6584, 0.3977, 0.5597, -0.4784, 0.5360, -0.2532, 0.5362, -0.1063]).astype( - np.float32).reshape([1, -1]) - - wih_l1 = np.array([[0.0602, 0.6977, -0.3882, 0.3734], - [-0.6896, -0.6014, -0.2311, 0.6433], - [-0.6778, -0.5100, -0.1496, 0.5774], - [-0.5824, 0.4656, -0.2835, -0.5688], - [0.5623, 0.3599, 0.1731, 0.3124], - [0.1492, -0.6663, -0.1099, -0.5282], - [0.4696, -0.1795, -0.6712, -0.3903], - [0.4995, 0.0709, -0.1738, 0.2822]]).astype(np.float32).reshape([1, -1]) - - whh_l1 = np.array([[0.3770, 0.4139], - [0.5351, 0.6394], - [0.3901, -0.1072], - [0.1106, 0.1331], - [0.3970, 0.4693], - [0.2958, -0.3813], - [-0.3064, 0.5519], - [-0.2827, 0.5844]]).astype(np.float32).reshape([1, -1]) - - bih_l1 = np.array([0.5242, 0.5896, 0.3709, 0.6202, 0.5008, 0.2674, 0.4356, -0.3261]).astype(np.float32).reshape( - [1, -1]) - bhh_l1 = np.array([-0.6648, 0.6680, 0.2510, -0.1245, -0.0524, 0.5439, -0.1650, 0.5303]).astype( - np.float32).reshape([1, -1]) - - wih_reverse_l1 = np.array([[0.6477, 0.4416, 0.3803, -0.4708], - [0.4497, 0.2833, -0.4739, -0.6361], - [-0.5573, -0.3867, -0.0349, -0.4128], - [-0.1545, 0.3720, 0.2354, -0.6090], - [0.5965, 0.6301, -0.4591, -0.0120], - [-0.1253, -0.1881, -0.4388, 0.4335], - [0.1944, -0.1230, -0.6170, 0.1043], - [-0.6700, 0.4343, 0.6474, 0.0113]]).astype(np.float32).reshape([1, -1]) - - whh_reverse_l1 = np.array([[0.6576, 0.5573], - [0.2318, 0.0187], - [-0.6365, 0.5744], - [-0.6494, -0.1820], - [0.6461, -0.3344], - [0.0906, -0.5405], - [-0.5999, 0.5571], - [-0.0488, 0.5345]]).astype(np.float32).reshape([1, -1]) - - bih_reverse_l1 = np.array([-0.6058, -0.2812, -0.4449, -0.0802, 0.4931, 0.4066, 0.5960, 0.1968]).astype( - np.float32).reshape([1, -1]) - bhh_reverse_l1 = np.array([-0.2490, -0.3402, -0.5089, -0.3875, 0.4852, -0.0402, -0.0072, -0.1017]).astype( - np.float32).reshape([1, -1]) - - ''' - weight - layer0 - forward - wih - whh - reverse - wih - whh - layer1 - forward - wih - whh - reverse - wih - whh - ... ... - bias: - layer0 - forward - bih - bhh - reverse - bih - bhh - layer1 - forward - bih - bhh - reverse - bih - bhh - ... ... - ''' - w_np = np.concatenate( - (wih_l0, whh_l0, wih_reverse_l0, whh_reverse_l0, wih_l1, whh_l1, wih_reverse_l1, whh_reverse_l1, - bih_l0, bhh_l0, bih_reverse_l0, bhh_reverse_l0, bih_l1, bhh_l1, bih_reverse_l1, bhh_reverse_l1), - axis=1).reshape([-1, 1, 1]) - - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - - @ms_function - def construct(self): - return self.lstm(self.x, self.h, self.c, self.w) - - -@pytest.mark.level0 -@pytest.mark.platform_x86_gpu_training -@pytest.mark.env_onecard -def test_multi_layer_bilstm(): - seq_len = 5 - batch_size = 2 - - input_size = 10 - hidden_size = 2 - num_layers = 2 - has_bias = True - bidirectional = True - dropout = 0.0 - - num_directions = 1 - if bidirectional: - num_directions = 2 - - net = MultiLayerBiLstmNet(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, - dropout) - y, h, c, _, _ = net() - expect_y = np.array([[[0.5186, 0.5419, 0.2710, 0.0384], - [0.6196, 0.5539, 0.3266, 0.0866]], - - [[0.5244, 0.5276, 0.3042, 0.0510], - [0.5143, 0.4937, 0.2828, 0.0387]], - - [[0.5124, 0.5079, 0.2951, 0.0548], - [0.4051, 0.4493, 0.2369, 0.0077]], - - [[0.4532, 0.4749, 0.2557, 0.0611], - [0.4879, 0.4812, 0.3160, 0.0368]], - - [[0.4535, 0.4806, 0.3880, 0.0462], - [0.4674, 0.4849, 0.3890, 0.1008]]]) - - error = np.ones([seq_len, batch_size, hidden_size * num_directions]) * 1.0e-4 - diff = y.asnumpy() - expect_y - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_h = np.array([[[0.4730, 0.1638], - [0.1406, -0.0697]], - - [[0.3887, -0.0518], - [-0.3988, -0.0071]], - - [[0.4535, 0.4806], - [0.4674, 0.4849]], - - [[0.2710, 0.0384], - [0.3266, 0.0866]]]) - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-4 - diff = h.asnumpy() - expect_h - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_c = np.array([[[0.8713, 0.2694], - [0.2075, -0.2201]], - - [[0.5084, -0.0964], - [-0.5155, -0.2452]], - - [[1.1724, 1.0334], - [1.2003, 1.1058]], - - [[0.5179, 0.0750], - [0.5309, 0.2012]]]) - - error = np.ones((num_layers * num_directions, batch_size, hidden_size)) * 1.0e-3 - diff = c.asnumpy() - expect_c - assert np.all(diff < error) - assert np.all(-diff < error) - - -class Grad(nn.Cell): - def __init__(self, network): - super(Grad, self).__init__() +import numpy as np +from mindspore import context +from mindspore import nn +from mindspore import Tensor +from mindspore.common.parameter import ParameterTuple +from mindspore.common.parameter import Parameter +from mindspore.ops import composite as c + + +class GradOfAllInputsAndParams(nn.Cell): + def __init__(self, network, sens_param): + super().__init__() + self.grad = c.GradOperation(get_all=True, get_by_list=True, sens_param=sens_param) self.network = network - self.weights = ParameterTuple(network.trainable_params()) - self.grad = C.GradOperation(get_by_list=True, - sens_param=True) + self.params = ParameterTuple(self.network.trainable_params()) - @ms_function - def construct(self, output_grad): - weights = self.weights - grads = self.grad(self.network, weights)(output_grad) - return grads + def construct(self, *inputs): + gout = self.grad(self.network, self.params)(*inputs) + return gout -class Net(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(Net, self).__init__() +class LSTM(nn.Cell): + def __init__(self, input_s, hidden_s, num_layers, has_bias, batch_first, bidirectional, dropout): + super().__init__() + self.lstm = nn.LSTM(input_size=input_s, hidden_size=hidden_s, num_layers=num_layers, has_bias=has_bias, + batch_first=batch_first, bidirectional=bidirectional, dropout=dropout) - num_directions = 1 - if bidirectional: - num_directions = 2 + def construct(self, inp, h0, c0): + return self.lstm(inp, (h0, c0)) - self.lstm = P.LSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - input_np = np.array([[[-0.5907, 1.0557, 1.7283, 0.6706, -1.2550, -0.5298, -0.2290, -0.6735, 0.8555, 1.4836], - [-1.7070, -0.5347, -0.9105, -0.2598, 0.0588, 1.5496, 1.0757, 0.3760, -1.2020, -0.2868]], +class LSTMWeightBias(): + def __init__(self, num_layers, has_bias, input_size, num_directions, hidden_size, bidirectional): + self.num_layers = num_layers + self.has_bias = has_bias + self.input_size = input_size + self.num_directions = num_directions + self.hidden_size = hidden_size + self.bidirectional = bidirectional - [[0.0151, 0.2126, 0.8090, -0.5292, -2.5590, 0.4279, -0.3081, -1.4706, -0.0498, 1.2301], - [0.4165, -0.5391, -0.0996, 0.1928, -0.4909, -0.1255, 0.4444, -1.3687, 1.3096, 0.6553]], + def get_weight_bias(self): + gate_size = 4 * self.hidden_size - [[-0.7802, -0.2083, -0.6388, 1.3757, 0.4293, 0.5363, 0.3202, -0.6687, -1.3864, -0.2953], - [1.0799, -0.7204, 0.1130, -0.5857, -0.4855, -1.1068, 1.0126, 0.8716, 1.5460, -0.7392]], - - [[2.2645, -0.6586, -0.2227, 1.4290, -0.5006, -1.6576, -0.1793, 0.5319, 0.1360, 0.2707], - [-0.4071, 0.1575, 1.4199, -0.9156, 0.1855, 0.4947, 1.0460, -0.6365, 0.1191, -0.6374]], - - [[0.2468, 1.0815, -0.4893, 0.0664, 0.6405, -2.2967, 0.7612, 0.8759, 0.5685, -1.0999], - [-0.7272, -1.7750, -0.1164, -0.7159, 0.0061, -0.7839, -1.8329, 0.3434, -0.5634, - 0.5384]]]).astype(np.float32) - - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - - self.h = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='h') - - self.c = Parameter(initializer( - Tensor(np.ones((num_layers * num_directions, batch_size, hidden_size)).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='c') - - wih_l0 = np.array([[0.2300, 0.6668, 0.4703, 0.0425, 0.0464, 0.6825, 0.2249, -0.4315, -0.2449, 0.2964], - [-0.2811, -0.3444, 0.2557, -0.5137, -0.5518, 0.1652, -0.6720, 0.1066, 0.3586, 0.6299], - [0.5728, -0.1784, 0.5661, 0.4012, 0.3856, -0.1899, 0.3102, 0.3717, -0.5651, 0.1952], - [0.1026, -0.0527, 0.1198, -0.3080, 0.2292, 0.5757, -0.3567, -0.2731, -0.0586, -0.2849], - [0.2194, -0.1622, 0.3219, -0.3008, -0.3713, -0.3034, -0.2385, 0.0412, -0.5205, 0.0280], - [-0.5499, -0.0733, -0.5236, -0.6753, -0.7045, -0.1839, -0.1037, -0.5026, -0.4055, -0.3416], - [0.1573, -0.1301, -0.2882, -0.3464, 0.6643, 0.1980, -0.6804, 0.5359, 0.5996, 0.0124], - [-0.6436, 0.0587, -0.6520, -0.0471, 0.1667, 0.6042, 0.5752, -0.6296, -0.2976, - -0.3757]]).astype(np.float32).reshape([1, -1]) - - whh_l0 = np.array([[0.3358, 0.2790], - [-0.5355, 0.0989], - [-0.1402, 0.5120], - [0.1335, 0.1653], - [0.3533, -0.3531], - [0.4166, -0.4420], - [-0.5454, -0.1720], - [0.0041, -0.0799]]).astype(np.float32).reshape([1, -1]) - - bih_l0 = np.array([0.5518, 0.1083, 0.4829, 0.0607, -0.1770, -0.6944, 0.3059, 0.5354]).astype( - np.float32).reshape([1, -1]) - bhh_l0 = np.array([0.5025, -0.1261, -0.5405, 0.3220, -0.3441, 0.6488, -0.0284, -0.2334]).astype( - np.float32).reshape([1, -1]) - - wih_reverse_l0 = np.array( - [[-0.7048, -0.1768, 0.2288, -0.0760, -0.1319, 0.0820, -0.4132, 0.3644, 0.3919, 0.2449], - [0.0551, -0.0530, -0.5883, 0.0799, -0.5025, 0.1500, -0.4067, -0.3764, -0.3018, 0.2467], - [-0.2279, 0.3144, 0.5705, 0.4617, 0.1729, 0.6539, -0.2086, 0.5355, 0.4439, 0.0122], - [0.6967, -0.5245, 0.3527, 0.3386, 0.0429, -0.3803, -0.4328, -0.4767, 0.4481, -0.2405], - [0.6744, -0.2776, 0.0798, 0.1543, 0.6421, 0.6102, 0.3591, -0.4431, -0.6327, -0.0075], - [-0.4520, 0.4201, -0.2374, -0.1556, -0.4175, -0.6834, 0.3096, -0.1581, 0.0127, 0.6872], - [0.1788, -0.5442, -0.3675, -0.2887, -0.3004, 0.5813, 0.1618, 0.6875, -0.4678, 0.0071], - [-0.6453, -0.2528, 0.5675, -0.5154, -0.4129, -0.0214, 0.5539, 0.0343, 0.1712, 0.5644]]).astype( - np.float32).reshape([1, -1]) - - whh_reverse_l0 = np.array([[-0.6657, 0.6330], - [-0.2290, 0.6556], - [0.4808, -0.2712], - [0.0407, -0.2587], - [0.3837, 0.0382], - [0.2268, 0.1217], - [-0.6404, -0.3336], - [0.5461, -0.0764]]).astype(np.float32).reshape([1, -1]) - - bih_reverse_l0 = np.array([0.0314, 0.1009, 0.3664, -0.6732, -0.6944, 0.5098, -0.1251, 0.2644]).astype( - np.float32).reshape([1, -1]) - bhh_reverse_l0 = np.array([-0.1961, -0.3836, 0.1191, -0.7022, -0.0961, 0.5493, -0.6979, 0.0017]).astype( - np.float32).reshape([1, -1]) - - wih_l1 = np.array([[1.2746e-01, -3.3346e-01, 1.5589e-01, -4.7986e-01], - [6.5835e-01, 3.8135e-01, -3.8409e-01, -3.6499e-01], - [-6.0374e-04, -1.2227e-01, -1.5955e-01, 4.2772e-01], - [-1.8281e-01, -5.0484e-01, 7.0204e-01, 6.5872e-01], - [3.7765e-01, -4.3494e-01, 3.1503e-01, -4.2504e-02], - [6.3506e-01, -4.3049e-02, -5.7413e-01, -2.5134e-01], - [8.7181e-02, -5.5216e-01, 5.5436e-01, -3.9599e-01], - [4.4611e-01, -4.2690e-01, 6.6142e-01, 6.3882e-01]]).astype(np.float32).reshape([1, -1]) - - whh_l1 = np.array([[-0.0049, -0.3267], - [0.0863, -0.6277], - [0.4815, -0.2236], - [0.5996, -0.3441], - [0.3959, -0.0249], - [0.3986, -0.0922], - [-0.5321, 0.0877], - [0.2811, -0.0483]]).astype(np.float32).reshape([1, -1]) - - bih_l1 = np.array([0.0032, -0.0893, 0.5706, 0.3712, 0.0590, 0.0044, 0.2417, 0.1291]).astype(np.float32).reshape( - [1, -1]) - bhh_l1 = np.array([-0.0704, 0.3908, -0.1121, 0.6970, -0.6216, 0.6340, -0.2945, 0.5224]).astype( - np.float32).reshape([1, -1]) - - wih_reverse_l1 = np.array([[-0.2693, 0.3487, 0.0692, 0.0047], - [0.6187, 0.5649, 0.0680, 0.5110], - [-0.5262, -0.3307, -0.3892, 0.5382], - [-0.2925, 0.5185, -0.1385, 0.3431], - [-0.3252, 0.3809, -0.4680, 0.3379], - [0.4763, -0.5465, 0.0033, -0.5144], - [0.3826, -0.3879, -0.2439, 0.2571], - [-0.0422, -0.0359, -0.4197, -0.2209]]).astype(np.float32).reshape([1, -1]) - - whh_reverse_l1 = np.array([[-0.4691, 0.5944], - [-0.6885, 0.1708], - [0.6391, -0.3690], - [-0.5919, 0.1805], - [-0.6853, -0.6215], - [-0.4635, -0.6714], - [-0.2050, 0.0513], - [0.3411, -0.2833]]).astype(np.float32).reshape([1, -1]) - - bih_reverse_l1 = np.array([0.5764, -0.7010, -0.0831, -0.3779, -0.2743, 0.0480, -0.2707, -0.5583]).astype( - np.float32).reshape([1, -1]) - bhh_reverse_l1 = np.array([0.3379, -0.2671, -0.2789, -0.6611, -0.5542, -0.0188, 0.1831, 0.3612]).astype( - np.float32).reshape([1, -1]) - - ''' - weight - layer0 - forward - wih - whh - reverse - wih - whh - layer1 - forward - wih - whh - reverse - wih - whh - ... ... - bias: - layer0 - forward - bih - bhh - reverse - bih - bhh - layer1 - forward - bih - bhh - reverse - bih - bhh - ... ... - ''' - w_np = np.concatenate( - (wih_l0, whh_l0, wih_reverse_l0, whh_reverse_l0, wih_l1, whh_l1, wih_reverse_l1, whh_reverse_l1, - bih_l0, bhh_l0, bih_reverse_l0, bhh_reverse_l0, bih_l1, bhh_l1, bih_reverse_l1, bhh_reverse_l1), - axis=1).reshape([-1, 1, 1]) - - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - - @ms_function - def construct(self): - return self.lstm(self.x, self.h, self.c, self.w)[0] + w_ih_list = [] + w_hh_list = [] + b_ih_list = [] + b_hh_list = [] + stdv = 1 / math.sqrt(self.hidden_size) + for layer in range(self.num_layers): + for direction in range(self.num_directions): + layer_input_size = self.input_size if layer == 0 else self.hidden_size * self.num_directions + suffix = '_reverse' if direction == 1 else '' + w_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, layer_input_size)).astype(np.float32)), + name='weight_ih_l{}{}'.format(layer, suffix))) + w_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size, self.hidden_size)).astype(np.float32)), + name='weight_hh_l{}{}'.format(layer, suffix))) + if self.has_bias: + b_ih_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_ih_l{}{}'.format(layer, suffix))) + b_hh_list.append(Parameter( + Tensor(np.random.uniform(-stdv, stdv, (gate_size)).astype(np.float32)), + name='bias_hh_l{}{}'.format(layer, suffix))) + w_ih_list = ParameterTuple(w_ih_list) + w_hh_list = ParameterTuple(w_hh_list) + b_ih_list = ParameterTuple(b_ih_list) + b_hh_list = ParameterTuple(b_hh_list) + return w_ih_list, w_hh_list, b_ih_list, b_hh_list @pytest.mark.level0 @pytest.mark.platform_x86_gpu_training @pytest.mark.env_onecard -def test_grad(): - seq_len = 5 - batch_size = 2 - - input_size = 10 - hidden_size = 2 - num_layers = 2 - has_bias = True - bidirectional = True - dropout = 0.0 - - net = Grad(Net(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout)) - - dy = np.array([[[-3.5471e-01, 7.0540e-01, -7.5945e-01, -1.2322e+00], - [2.7161e-01, 1.0865e+00, -2.1827e-03, 8.8031e-01]], - - [[-4.2431e-01, 1.4955e+00, 4.6576e-01, -2.7230e+00], - [-4.0418e-01, -2.3282e-01, 9.1253e-01, -2.7379e-01]], - - [[-1.3654e+00, 1.9251e+00, -1.6808e+00, -3.2642e-02], - [-4.6481e-01, 1.3138e+00, 1.2956e-02, 1.0198e+00]], - - [[1.2914e+00, -2.3753e-01, 9.4763e-01, 1.7930e-02], - [5.3589e-01, -1.0981e-01, 1.5377e+00, 6.2709e-01]], - - [[-1.6032e+00, -1.8818e-01, 7.0441e-01, -2.8765e+00], - [1.0065e-01, 9.2045e-01, 2.7426e-01, 2.6196e-01]]]).astype(np.float32) - - dx, dh, dc, _ = net(Tensor(dy)) - expect_dx = np.array([[[0.01697153, -0.0096909, 0.01306139, 0.00863109, -0.00122794, -0.00746152, -0.00879683, - 0.00643571, 0.0015958, 0.01480642], - [0.05794962, -0.02326604, 0.01862703, 0.02053947, 0.02607713, -0.01278067, 0.04250786, - -0.02686035, -0.07441005, 0.00806021]], - - [[-0.026675, -0.01024149, -0.02492021, -0.00457492, -0.0085863, 0.02341479, 0.02188834, - -0.04139283, -0.01367766, -0.00305065], - [-0.00762213, -0.01914341, -0.03233681, -0.03580827, -0.02201782, -0.00153102, -0.00097455, - -0.02708411, -0.03711082, -0.02804472]], - - [[-0.0040581, -0.00116989, 0.01652471, 0.02182668, -0.02547193, -0.04171437, 0.04185125, - 0.01589275, -0.00517019, 0.06554792], - [-0.02294365, -0.00589715, -0.01425684, -0.01499153, -0.05327821, -0.03133425, 0.00755623, - -0.04192506, -0.02122675, -0.01214214]], - - [[-0.00041491, 0.00240709, -0.00942589, 0.00719656, 0.01438523, 0.00931082, 0.00534746, - -0.0004002, 0.01299422, 0.00181135], - [-0.01704482, -0.00887032, -0.01746774, -0.03289891, -0.04259495, -0.01928082, -0.01570587, - -0.01242383, -0.01799918, -0.00610236]], - - [[0.00207505, -0.0008109, 0.00114241, 0.00251349, -0.00065676, 0.00151333, -0.00077485, - -0.00034354, -0.00028289, -0.0006986], - [-0.00240827, -0.0001309, 0.01401818, -0.01272261, -0.02665948, -0.01095799, -0.007761, - -0.0087831, 0.01038029, 0.02021475]]]).astype(np.float32) - - error = np.ones(dx.asnumpy().shape) * 1.0e-4 - diff = dx.asnumpy() - expect_dx - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_dh = np.array([[[-0.00696833, 0.00212885], - [0.01416209, 0.0002706]], - - [[0.00297393, -0.0021012], - [0.00458834, 0.00400078]], - - [[0.08658642, -0.10590762], - [0.1516603, -0.10525411]], - - [[0.11888178, -0.04759264], - [0.05898442, -0.08082277]]]).astype(np.float32) - - error = np.ones(dh.asnumpy().shape) * 1.0e-4 - diff = dh.asnumpy() - expect_dh - assert np.all(diff < error) - assert np.all(-diff < error) - - expect_dc = np.array([[[0.00887521, -0.01391486], - [0.03858164, -0.04941981]], - - [[0.00665188, 0.00184223], - [-0.00541833, 0.01410913]], - - [[-0.2068854, 0.5585638], - [0.01735374, 0.3537254]], - - [[0.20350647, -0.2792883], - [0.18456826, 0.02278761]]]).astype(np.float32) - - error = np.ones(dc.asnumpy().shape) * 1.0e-4 - diff = dc.asnumpy() - expect_dc - assert np.all(diff < error) - assert np.all(-diff < error) - - -class LstmNetWithDropout(nn.Cell): - def __init__(self, seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout): - super(LstmNetWithDropout, self).__init__() - - num_directions = 1 - if bidirectional: - num_directions = 2 - - self.lstm = P.LSTM(input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - - input_np = np.array([[[-2.48789445e-01, -2.18991071e-01, -8.41492534e-01, -5.73351622e-01, 8.20644796e-02, - 4.14313585e-01, -1.30143976e+00, -4.43366140e-01, -1.21003680e-01, -2.11284861e-01], - [9.94045794e-01, 3.18840504e-01, 4.81898338e-01, -4.83986028e-02, -9.26419497e-02, - -2.57977694e-01, 1.82191110e+00, 5.95121741e-01, 6.30752742e-01, -6.01903737e-01]], - - [[7.67166913e-01, 5.41202351e-02, -1.24094069e+00, 1.38814664e+00, 2.05845284e+00, - 7.29744852e-01, -1.12405574e+00, 3.78702253e-01, 2.28524983e-01, 2.02445173e+00], - [-1.85264975e-01, -4.55119252e-01, 1.23624969e+00, 1.24347043e+00, -1.68316591e+00, - -3.55918944e-01, 3.07149738e-01, -3.44966322e-01, -1.08978853e-01, 1.80912763e-01]], - - [[-6.47622466e-01, 1.31204927e+00, 6.47477210e-01, -7.93370783e-01, 3.08402872e-04, - -5.12097359e-01, -1.69133916e-01, 8.57838035e-01, -3.63963723e-01, 6.35978997e-01], - [-3.92911851e-01, 8.27334300e-02, -1.11347124e-01, 8.79961967e-01, 6.02812059e-02, - -3.76448452e-01, -1.48800862e+00, -9.48699772e-01, -1.24202335e+00, 1.65264118e+00]], - - [[4.05404866e-01, 5.67396320e-02, -2.05705926e-01, -8.70196745e-02, -7.34854519e-01, - -1.07580565e-01, 1.33716142e+00, -1.18140256e+00, 2.66074872e+00, -3.26788813e-01], - [6.97183967e-01, -2.32625628e+00, 1.20393467e+00, -2.32532692e+00, 2.03347206e+00, - -7.58083522e-01, 1.35564697e+00, -2.32149422e-01, 9.85125721e-01, 1.00944638e+00]], - - [[9.89606023e-01, -5.30669808e-01, -2.66087383e-01, 8.14819038e-01, 1.07067376e-01, - -1.76214290e+00, -5.04977465e-01, 1.94490123e+00, 5.10450959e-01, -2.29238123e-01], - [-1.32928836e+00, -1.18175328e-01, -5.17818272e-01, -1.45089477e-01, 7.13987231e-01, - -7.41293788e-01, -3.67817104e-01, 1.18039274e+00, -6.03745162e-01, - -5.83392143e-01]]]).astype(np.float32) - - self.x = Parameter(initializer(Tensor(input_np), [seq_len, batch_size, input_size]), name='x') - - self.h = Parameter(initializer( - Tensor(np.array([[[-0.47240502, 1.6824378], - [-0.00978304, 0.8179632]]]).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='h') - - self.c = Parameter(initializer( - Tensor(np.array([[[-0.85975164, -0.3198615], - [-0.9821871, 0.26311848]]]).astype(np.float32)), - [num_layers * num_directions, batch_size, hidden_size]), name='c') - - wih = np.array([[0.4473, -0.5509, -0.1585, -0.6215, 0.6228, 0.3462, 0.3015, -0.3714, 0.3119, -0.1151], - [-0.6923, 0.1373, 0.2214, 0.2280, 0.6960, -0.6368, 0.5725, -0.1359, 0.0742, -0.6777], - [-0.4432, 0.6162, -0.1066, -0.6138, -0.2529, -0.5638, -0.0603, 0.3039, 0.1068, -0.5300], - [0.4337, -0.1215, -0.5088, -0.0045, 0.2828, 0.1411, 0.0741, 0.6936, -0.4603, 0.6986], - [-0.2079, -0.5518, 0.5375, -0.2168, 0.3662, 0.0948, -0.0564, -0.1808, -0.6672, -0.2410], - [0.5142, 0.0790, -0.1123, -0.2351, 0.3982, -0.6351, 0.5906, 0.3917, -0.0850, -0.5397], - [-0.4795, -0.6576, 0.5693, 0.0047, -0.6626, 0.1013, -0.4015, -0.4040, -0.2817, 0.4430], - [0.0251, -0.3035, -0.6026, 0.2693, -0.2749, 0.1501, -0.5778, 0.5570, -0.7065, -0.6196]]).astype( - np.float32).reshape([1, -1]) - - whh = np.array([[-0.4344, -0.2529], - [0.0377, 0.7046], - [-0.0579, -0.5240], - [-0.4801, -0.1149], - [-0.4010, -0.5614], - [0.4721, 0.4366], - [-0.4282, 0.0816], - [0.1574, -0.3359]]).astype(np.float32).reshape([1, -1]) - - bih = np.array([0.2431, 0.5967, -0.2417, -0.4169, -0.5326, 0.5685, -0.2971, -0.4326]).astype( - np.float32).reshape([1, -1]) - bhh = np.array([-0.1751, -0.2270, -0.3980, -0.4983, -0.3527, -0.2774, 0.6371, -0.3330]).astype( - np.float32).reshape([1, -1]) - - w_np = np.concatenate((wih, whh, bih, bhh), axis=1).reshape([-1, 1, 1]) - - self.w = Parameter(initializer(Tensor(w_np), w_np.shape), name='w') - - def construct(self): - return self.lstm(self.x, self.h, self.c, self.w) - - -@pytest.mark.level0 -@pytest.mark.platform_x86_gpu_training -@pytest.mark.env_onecard -def test_lstm_dropout(): - seq_len = 5 - batch_size = 2 - - input_size = 10 - hidden_size = 2 - num_layers = 1 +def test_sit_lstm_forward_input_3_32_32_is_32_hs_16(): + """ + Feature: LSTM forward + Description: LSTM with input (3, 32, 32) + Expectation: Graph mode equal to pynative mode + """ + input_s = 32 + hidden_s = 16 has_bias = True bidirectional = False - dropout = 1.0 + num_layers = 1 + num_directions = 1 - net = LstmNetWithDropout(seq_len, batch_size, input_size, hidden_size, num_layers, has_bias, bidirectional, dropout) - y, _, _, _, _ = net() - expect_y = np.array([[[-0.45210335, -0.0844336], - [-0.14677924, 0.07140275]], + fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() - [[-0.18895914, -0.11084185], - [-0.26356253, -0.06367199]], + h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + input_ms = Tensor(np.random.randn(3, 32, 32).astype(np.float32)) - [[-0.33480304, 0.00812318], - [-0.0887147, -0.1564593]], + # graph mode + context.set_context(mode=context.GRAPH_MODE) + net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list + out, (hy, cy) = net(input_ms, h0, c0) - [[-0.33231455, 0.00743252], - [0.428218, 0.00723737]], + # pynative mode + context.set_context(mode=context.PYNATIVE_MODE) + net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list + out_pynative, (hy_pynative, cy_pynative) = net_pynative(input_ms, h0, c0) + context.set_context(mode=context.GRAPH_MODE) - [[-0.20026046, 0.43491203], - [0.17739448, 0.5313992]]]) + assert np.allclose(out.asnumpy(), out_pynative.asnumpy(), 0.0001, 0.0001) + assert np.allclose(hy.asnumpy(), hy_pynative.asnumpy(), 0.0001, 0.0001) + assert np.allclose(cy.asnumpy(), cy_pynative.asnumpy(), 0.0001, 0.0001) - error = np.ones([num_layers, batch_size, hidden_size]) * 1.0e-4 - diff = y.asnumpy() - expect_y - assert np.all(diff < error) - assert np.all(-diff < error) +@pytest.mark.level0 +@pytest.mark.platform_x86_gpu_training +@pytest.mark.env_onecard +def test_sit_lstm_grad_input_3_32_32_is_32_hs_16(): + """ + Feature: LSTM backward + Description: LSTM with input (3, 32, 32) + Expectation: Graph mode equal to pynative mode + """ + input_s = 32 + hidden_s = 16 + has_bias = True + bidirectional = False + num_layers = 1 + num_directions = 1 + + fact = LSTMWeightBias(num_layers, has_bias, input_s, num_directions, hidden_s, bidirectional) + w_ih_list, w_hh_list, b_ih_list, b_hh_list = fact.get_weight_bias() + + h0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + c0 = Tensor(np.random.randn(num_layers * 1, 32, 16).astype(np.float32)) + input_ms = Tensor(np.random.randn(3, 32, 32).astype(np.float32)) + + # graph mode + context.set_context(mode=context.GRAPH_MODE) + net = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net.lstm.w_ih_list = w_ih_list + net.lstm.w_hh_list = w_hh_list + net.lstm.b_ih_list = b_ih_list + net.lstm.b_hh_list = b_hh_list + + grad_net_inp = GradOfAllInputsAndParams(net, sens_param=False) + grad_net_inp.set_train() + out_grad, _ = grad_net_inp(input_ms, h0, c0) + x_grad = out_grad[0].asnumpy() + h_grad = out_grad[1].asnumpy() + c_grad = out_grad[2].asnumpy() + + # pynative mode + context.set_context(mode=context.PYNATIVE_MODE) + net_pynative = LSTM(input_s=input_s, hidden_s=16, num_layers=num_layers, has_bias=has_bias, batch_first=False, + bidirectional=bidirectional, dropout=0.0) + net_pynative.lstm.w_ih_list = w_ih_list + net_pynative.lstm.w_hh_list = w_hh_list + net_pynative.lstm.b_ih_list = b_ih_list + net_pynative.lstm.b_hh_list = b_hh_list + + grad_net_inp_pynative = GradOfAllInputsAndParams(net_pynative, sens_param=False) + grad_net_inp_pynative.set_train() + out_grad_pynative, _ = grad_net_inp_pynative(input_ms, h0, c0) + x_grad_pynative = out_grad_pynative[0].asnumpy() + h_grad_pynative = out_grad_pynative[1].asnumpy() + c_grad_pynative = out_grad_pynative[2].asnumpy() + context.set_context(mode=context.GRAPH_MODE) + + assert np.allclose(x_grad, x_grad_pynative, 0.001, 0.001) + assert np.allclose(h_grad, h_grad_pynative, 0.001, 0.001) + assert np.allclose(c_grad, c_grad_pynative, 0.001, 0.001)