### Implementing RNN from scratch

In [2]:
%matplotlib inline
import math
import torch
from torch import nn
from torch.nn import functional as F
from d2l import torch as d2l

In [3]:
class RNNScratch(d2l.Module):
    """RNN implemented from scratch"""
    def __init__(self, num_inputs, num_hiddens, sigma=0.01):
        super().__init__()
        self.save_hyperparameters()                             # Stores parameters for later use
        self.W_xh = nn.Parameters( 
            torch.randn(num_inputs, num_hiddens) * sigma)       # Initialize weight matrix of shape (num_inputs, num_hiddens). Initialized with a random normal distribution scaled by sigma
        self.W_hh = nn.Parameters(
            torch.randn(num_hiddens, num_hiddens) * sigma)      # Initialize weight matrix of shape (num_hiddens, num_hiddens). Initialized with a random normal distribution scaled by sigma
        self.b_h = nn.Parameters(torch.zeros(num_hiddens))      # Initialize bias vector of shape (num_hiddens) with zeros

In [4]:
device = 'cuda' if torch.cuda.is_available() else 'cpu'
print(f"Using device: {device}")

Using device: cuda


#### Introduce a forward method

In [None]:
@d2l.add_to_class(RNNScratch)
def forward(self, inputs, state=None):
    if state is None:
        # Initialize state with shape: (batch_size, num_hiddens)
        state = torch.zeros((inputs.shape[1], self.num_hiddens),
                            device=inputs.device)
        
    else: 
        state, = state
    outputs = []
    for X in inputs:                                                    # Shape of inputs: (num_steps, batch_size, num_inputs)
        state = torch.tanh(torch.matmul(X, self.W_xh) +
                           torch.matmul(state, self.W_hh) + self.b_h)
        outputs.append(state)
    return outputs, state
