Extending backpropagation for more than 1 hidden layer

Viewed 44

I have been working on learning about neural networks by following this playlist. In it we build a simple NN with input, 1 hidden layer, and an output layer.

Followed along and very basically understood the math behind the example. I got it working (at least observably) and could get around 96% correct with the MNIST dataset.

However, I tried to expand the class I built to be able to handle more hidden layers. I can get the feed forward and the backpropagation method to compile and run, however I'm only seeing around 10% (not even close to what I saw in the simple network implementation (even when using only 1 hidden layer, just through the new implementation)). In fact it usually seems to get worse and worse the more I train.

NeuralNetwork.cs

{
    int input_nodes;
    int hidden_nodes;
    int output_nodes;
    double learningRate;
    //List<int> hiddenLayers;
    Matrix<double> wih;
    Matrix<double> who;
    Matrix<double> hb;
    Matrix<double> ob;
    // for deep nn
    List<Matrix<double>> weights;
    List<Matrix<double>> biases;

    List<Matrix<double>> layerOutputs;

    //public NeuralNetwork(int numInputs, int numHidden, int numOutput)
    //{
    //    input_nodes = numInputs;
    //    hidden_nodes = numHidden;
    //    output_nodes = numOutput;
    //    learningRate = 0.1d;

    //    wih = CreateMatrix.Dense<double>(numHidden, numInputs, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
    //    Debug.Log($"WIH: {wih.ToString()}");
    //    who = CreateMatrix.Dense<double>(numOutput, numHidden, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
    //    Debug.Log($"WHO: {who.ToString()}");
    //    hb = CreateMatrix.Dense<double>(numHidden, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
    //    Debug.Log($"HB: {hb.ToString()}");
    //    ob = CreateMatrix.Dense<double>(numOutput, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
    //    Debug.Log($"OB: {ob.ToString()}");


    //}


    public NeuralNetwork(int numInputs, int[] numHidden, int numOutput)
    {
        input_nodes = numInputs;
        weights = new List<Matrix<double>>();
        weights.Add(CreateMatrix.Dense<double>(numHidden[0], numInputs, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
        biases = new List<Matrix<double>>();
        biases.Add(CreateMatrix.Dense<double>(numHidden[0], 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
        for (int i = 1; i < numHidden.Length; i++)
        {
            weights.Add(CreateMatrix.Dense<double>(numHidden[i], numHidden[i-1], (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
            biases.Add(CreateMatrix.Dense<double>(numHidden[i], 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
        }
        weights.Add(CreateMatrix.Dense<double>(numOutput, numHidden[numHidden.Length - 1], (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
        biases.Add(CreateMatrix.Dense<double>(numOutput, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
        output_nodes = numOutput;

        layerOutputs = new List<Matrix<double>>();
        softMax = true;

        //Debug.Log($"Weights Count: {weights.Count}; Biases Count: {biases.Count}");
    }


    public double[] FeedForward(int[] inputs)
    {
        return FeedForward(ConvertArray(inputs));
    }


    public double[] FeedForward(double[] inputs)
    {
        layerOutputs.Clear();

        var inputVector = CreateVector.DenseOfArray(inputs);

        Matrix<double> input;
        for(int i = 0; i < weights.Count; i++)
        {
            if(i == 0)
            {
                input = inputVector.ToColumnMatrix();
            }
            else
            {
                input = layerOutputs[i - 1];
            }
            // Generate output
            var layerOutput = weights[i] * input;
            // Add Bias
            layerOutput += biases[i];
            // Activation function
            if (i == weights.Count - 1)
            {
                // Activation for last layer
                layerOutput = layerOutput.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
// Read a bit about softmax, so I normalize my outputs, but haven't figured out how to account for it in backpropagation, so I turn it off when I train.
                if (softMax)
                {
                    layerOutput = NormalizeColumnMatrix(layerOutput);
                }
            }
            else
            {
                // Activation for other layers
                layerOutput = layerOutput.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
            }
            // Add to list of outputs
            layerOutputs.Add(layerOutput);
        }

        return layerOutputs[layerOutputs.Count - 1].ToColumnMajorArray();
    }

    public void Train(int[] inputArray, int[] targetsArray)
    {
        Train(ConvertArray(inputArray), ConvertArray(targetsArray));
    }


    public double[] ConvertArray(int[] array)
    {
        double[] convertedArray = new double[array.Length];

        for(int i = 0; i < array.Length; i++)
        {
            convertedArray[i] = array[i];
        }

        return convertedArray;

    }

    public Matrix<double> NormalizeColumnMatrix(Matrix<double> matrix)
    {
        if (matrix == null)
        {
            throw new NullReferenceException("Matrix is not set to an instance of an object!");
        }
        if(matrix.ColumnCount != 1)
        {
            throw new Exception("Matrix has more or less columns than 1!");
        }

        double sum = 0d;

        for(int i = 0; i < matrix.RowCount; i++)
        {
            sum += matrix[i,0];
        }

        return matrix.Map(x => { return x / sum; });
    }

    public void Train(double[] inputArray, double[] targetsArray)
    {
        softMax = false;
        var output = FeedForward(inputArray);
        softMax = true;



        // -----------------Backpropagation------------------------------


        // Convert correct values into a vector
        var targets = CreateVector.DenseOfArray(targetsArray);
        var outputs = CreateVector.DenseOfArray(targetsArray);

        // Calculate Output error
        var outputError = targets.ToColumnMatrix() - outputs.ToColumnMatrix();

        Backpropagate(outputError, CreateVector.DenseOfArray(inputArray).ToColumnMatrix());

// The Following is the old way I used to backpropagate for only 2 layers 

        //var outputCrossEntropy = ElementwiseMult(targets.ToColumnMatrix(), outputs); // I think this would be right for calculating cross entropy for the given target? maybe?
        //Debug.Log($"Target: {targetsArray[0]}; Output: {outputs[0,0]}; Error: {outputError[0,0]}");
        //// OutputGradient is the derivative of the activation function
        //var outputGradient = ElementwiseMult(outputs, (1 - outputs));//outputs.Map(dsigmoid);
        ////var outputGradient =  outputs - targets.ToColumnMatrix();//ElementwiseMult(outputs, (1 - outputs));
        //// Scale stuff
        //var scaledTemp1 = learningRate * ElementwiseMult(outputError, outputGradient);
        //// Get the deltas to adjust the weights by for the Hidden-Output Connections
        //var whodeltas = scaledTemp1 * hiddenOutput.Transpose();
        //// Modify the H-O weights by calculated deltas;
        //who = who + whodeltas;
        //// Adjust the output bias by its delta (which is just the output gradient with errors and scaled)
        //ob = ob + scaledTemp1;

        //// Now do All that again for the I-H weights.
        //var hiddenError = who.Transpose() * outputError;
        ////Debug.Log($"WHOT: {who.Transpose().ToString()}; Output Error: {outputError.ToString()}; Hidden Error: {hiddenError.ToString()}");
        //// Calculate the Gradient of the hidden layer
        //var hiddenGradient = ElementwiseMult(hiddenOutput, (1 - hiddenOutput)); //hiddenOutput.Map(dsigmoid);
        //// scale stuff
        //var scaledTemp2 = learningRate * ElementwiseMult(hiddenError, hiddenGradient);
        //// Get Deltas to adjust the I-H weights
        //var wihdeltas = scaledTemp2 * inputVector.ToColumnMatrix().Transpose();
        //// Modify the I-H weights
        //wih = wih + wihdeltas;
        //// Adjust the hidden bias by its delta (which is just the hidden gradient with errors and scaled)
        //hb = hb + scaledTemp2;

        //// Should now be closer to the correct answer (hopefully)

    }

    private void Backpropagate(Matrix<double> outputError, Matrix<double> inputs)
    {

        List<Matrix<double>> errors = new List<Matrix<double>>();
        errors.Add(outputError);

        Matrix<double> gradient, scaled, deltas, error;

        for(int i = layerOutputs.Count - 1; i >= 0; i--)
        {
            if(i == layerOutputs.Count - 1)
            {
                error = errors[0];
            }
            else 
            {
                error = weights[i + 1].Transpose() * errors[layerOutputs.Count - 2 - i];
                errors.Add(error);
            }

            gradient = ElementwiseMult(layerOutputs[i], 1 - layerOutputs[i]);
            scaled = learningRate * ElementwiseMult(error, gradient);
            if(i == 0)
            {
                // For last iteration
                deltas = scaled * inputs.Transpose();
            }
            else
            {
                deltas = scaled * layerOutputs[i - 1].Transpose();
            }

            weights[i] += deltas;
            biases[i] += scaled;            

        }


    }


    private Matrix<double> SoftMax(Matrix<double> matrix)
    {
        matrix = matrix.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
        return NormalizeColumnMatrix(matrix);
    }

    private Matrix<double> ElementwiseMult(Matrix<double> a, Matrix<double> b)
    {
        Matrix<double> result;
        if(a != null && b != null && a.RowCount == b.RowCount && a.ColumnCount == b.ColumnCount)
        {
            result = CreateMatrix.Dense(a.RowCount, a.ColumnCount, 0d);

            for(int i = 0; i < a.RowCount; i++)
            {
                for(int j = 0; j < a.ColumnCount; j++)
                {
                    result[i, j] = a[i, j] * b[i, j];
                }
            }
        }
        else
        {
            throw new System.Exception("Argument dimensions do not match");
        }

        return result;
    }

}

I keep looking at the old version to extrapolate to the looped version but I must be missing something.

I'm using MathNet.Numerics for my matrix operations.

0 Answers
Related