Implementing Neural Network using pure Numpy (Softmax + CrossEntropy)

Viewed 4087

I am trying a simple implementation of a multi-layer perceptron (MLP) using pure NumPy. My previous implementation using RMSE and sigmoid activation at the output (single output) works perfectly with appropriate data. However, when I consider multi-output system (Due to one-hot encoding) with Cross-entropy loss function and softmax activation always fails.

I believe I am doing something wrong with my implementation for gradient calculation but unable to figure it out. So I am here for help.

For the current implementation, I use IRIS dataset for testing the model.

The data for IRIS is obtained as follows:

import numpy as np
from sklearn.datasets import load_iris
from sklearn.preprocessing import minmax_scale

def one_hot_encoder(y):
    y_oh = np.zeros((len(y), np.max(y)+1))

    for t in np.unique(y):
        y_oh[y==t,t] = 1
    return y_oh

data = load_iris().data
target = load_iris().target

data_scaled = minmax_scale(data)
target_oh = one_hot_encoder(target)

A Neural network class is defined with a simple 1-hidden layer network as follows:

class NeuralNetwork:
    def __init__(self, x, y):
        self.x = x

        # hidden layer with 16 nodes
        self.weights1= np.random.rand(self.x.shape[1],16)
        self.bias1 = np.random.rand(16)

        # output layer with 3 nodes (for 3 output - One-hot encoded)
        self.weights2 = np.random.rand(16,3)
        self.bias2 = np.random.rand(3)

        self.y = y
        self.pred = np.zeros(y.shape)

        self.lr = 0.001

    def feedforward(self):
        self.layer1 = sigmoid(np.dot(self.x, self.weights1) + self.bias1)
        self.layer2 = softmax(np.dot(self.layer1, self.weights2) + self.bias2)
        # print(self.layer2.shape)
        return self.layer2.clip(min=1e-8, max=None)

    def backprop(self):
        dloss = cross_entropy_derivative(self.pred, self.y)  # 2*(self.y - self.pred)
        d_weights2 = np.dot(self.layer1.T, dloss*softmax_derivative(self.pred))
        d_bias2 = np.dot(np.ones([self.x.shape[0]]), dloss*softmax_derivative(self.pred))

        d_weights1 = np.dot(self.x.T, np.dot(dloss*softmax_derivative(self.pred), self.weights2.T)*sigmoid_derivative(self.layer1))
        d_bias1 = np.dot(np.ones([self.x.shape[0]]), np.dot(dloss*softmax_derivative(self.pred), self.weights2.T)*sigmoid_derivative(self.layer1))

        self.weights1 += self.lr*d_weights1
        self.weights2 += self.lr*d_weights2

        self.bias1 += self.lr*d_bias1
        self.bias2 += self.lr*d_bias2

    def train(self, X, y):
        self.x = X
        self.y = y
        self.pred = self.feedforward()
        self.backprop()

    def predict(self, X):
        self.x = X
        self.pred = self.feedforward()
        return self.pred

    def evaluate(self, y, pred):
        self.y = y
        self.pred = pred

        # self.loss = np.sqrt(np.mean((self.pred-self.y)**2))
        self.loss = cross_entropy(self.pred, self.y)
        return self.loss

The activation functions and their derivatives are computed as follows (I feel there is something wrong here)

# Activation functions
def sigmoid(t):
    return 1/(1+np.exp(-t))

# Derivative of sigmoid
def sigmoid_derivative(p):
    return p * (1 - p)

# sofmax activation
def softmax(X):
    exps = np.exp(X - np.max(X,axis=1).reshape(-1,1))
    return exps / np.sum(exps,axis=1)[:,None]

# derivative of softmax
def softmax_derivative(pred):
    return pred * (1 -(1 * pred).sum(axis=1)[:,None])

The cross-entropy loss function and its derivatives are as shown below:

def cross_entropy(X,y):
    X = X.clip(min=1e-8,max=None)
    # print('\n\nCE: ', (np.where(y==1,-np.log(X), 0)).sum(axis=1))
    return (np.where(y==1,-np.log(X), 0)).sum(axis=1)

def cross_entropy_derivative(X,y):
    X = X.clip(min=1e-8,max=None)
    # print('\n\nCED: ', np.where(y==1,-1/X, 0))
    return np.where(y==1,-1/X, 0)

The main function call:

NN = NeuralNetwork(data_scaled, target_oh)

for i in range(10000): # trains the NN 10,000 times
    NN.train(data_scaled, target_oh)
    loss.append(NN.evaluate(NN.y, NN.pred))

y_pred = NN.predict(data_scaled)

The output is approximately constant always predicting a single class. What am I doing wrong? Appreciate your help on the code or directions to look at. Thanks.

0 Answers
Related