I have been working on learning about neural networks by following this playlist. In it we build a simple NN with input, 1 hidden layer, and an output layer.
Followed along and very basically understood the math behind the example. I got it working (at least observably) and could get around 96% correct with the MNIST dataset.
However, I tried to expand the class I built to be able to handle more hidden layers. I can get the feed forward and the backpropagation method to compile and run, however I'm only seeing around 10% (not even close to what I saw in the simple network implementation (even when using only 1 hidden layer, just through the new implementation)). In fact it usually seems to get worse and worse the more I train.
NeuralNetwork.cs
{
int input_nodes;
int hidden_nodes;
int output_nodes;
double learningRate;
//List<int> hiddenLayers;
Matrix<double> wih;
Matrix<double> who;
Matrix<double> hb;
Matrix<double> ob;
// for deep nn
List<Matrix<double>> weights;
List<Matrix<double>> biases;
List<Matrix<double>> layerOutputs;
//public NeuralNetwork(int numInputs, int numHidden, int numOutput)
//{
// input_nodes = numInputs;
// hidden_nodes = numHidden;
// output_nodes = numOutput;
// learningRate = 0.1d;
// wih = CreateMatrix.Dense<double>(numHidden, numInputs, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
// Debug.Log($"WIH: {wih.ToString()}");
// who = CreateMatrix.Dense<double>(numOutput, numHidden, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
// Debug.Log($"WHO: {who.ToString()}");
// hb = CreateMatrix.Dense<double>(numHidden, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
// Debug.Log($"HB: {hb.ToString()}");
// ob = CreateMatrix.Dense<double>(numOutput, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); });
// Debug.Log($"OB: {ob.ToString()}");
//}
public NeuralNetwork(int numInputs, int[] numHidden, int numOutput)
{
input_nodes = numInputs;
weights = new List<Matrix<double>>();
weights.Add(CreateMatrix.Dense<double>(numHidden[0], numInputs, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
biases = new List<Matrix<double>>();
biases.Add(CreateMatrix.Dense<double>(numHidden[0], 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
for (int i = 1; i < numHidden.Length; i++)
{
weights.Add(CreateMatrix.Dense<double>(numHidden[i], numHidden[i-1], (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
biases.Add(CreateMatrix.Dense<double>(numHidden[i], 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
}
weights.Add(CreateMatrix.Dense<double>(numOutput, numHidden[numHidden.Length - 1], (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
biases.Add(CreateMatrix.Dense<double>(numOutput, 1, (x, y) => { return UnityEngine.Random.Range(-1f, 1f); }));
output_nodes = numOutput;
layerOutputs = new List<Matrix<double>>();
softMax = true;
//Debug.Log($"Weights Count: {weights.Count}; Biases Count: {biases.Count}");
}
public double[] FeedForward(int[] inputs)
{
return FeedForward(ConvertArray(inputs));
}
public double[] FeedForward(double[] inputs)
{
layerOutputs.Clear();
var inputVector = CreateVector.DenseOfArray(inputs);
Matrix<double> input;
for(int i = 0; i < weights.Count; i++)
{
if(i == 0)
{
input = inputVector.ToColumnMatrix();
}
else
{
input = layerOutputs[i - 1];
}
// Generate output
var layerOutput = weights[i] * input;
// Add Bias
layerOutput += biases[i];
// Activation function
if (i == weights.Count - 1)
{
// Activation for last layer
layerOutput = layerOutput.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
// Read a bit about softmax, so I normalize my outputs, but haven't figured out how to account for it in backpropagation, so I turn it off when I train.
if (softMax)
{
layerOutput = NormalizeColumnMatrix(layerOutput);
}
}
else
{
// Activation for other layers
layerOutput = layerOutput.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
}
// Add to list of outputs
layerOutputs.Add(layerOutput);
}
return layerOutputs[layerOutputs.Count - 1].ToColumnMajorArray();
}
public void Train(int[] inputArray, int[] targetsArray)
{
Train(ConvertArray(inputArray), ConvertArray(targetsArray));
}
public double[] ConvertArray(int[] array)
{
double[] convertedArray = new double[array.Length];
for(int i = 0; i < array.Length; i++)
{
convertedArray[i] = array[i];
}
return convertedArray;
}
public Matrix<double> NormalizeColumnMatrix(Matrix<double> matrix)
{
if (matrix == null)
{
throw new NullReferenceException("Matrix is not set to an instance of an object!");
}
if(matrix.ColumnCount != 1)
{
throw new Exception("Matrix has more or less columns than 1!");
}
double sum = 0d;
for(int i = 0; i < matrix.RowCount; i++)
{
sum += matrix[i,0];
}
return matrix.Map(x => { return x / sum; });
}
public void Train(double[] inputArray, double[] targetsArray)
{
softMax = false;
var output = FeedForward(inputArray);
softMax = true;
// -----------------Backpropagation------------------------------
// Convert correct values into a vector
var targets = CreateVector.DenseOfArray(targetsArray);
var outputs = CreateVector.DenseOfArray(targetsArray);
// Calculate Output error
var outputError = targets.ToColumnMatrix() - outputs.ToColumnMatrix();
Backpropagate(outputError, CreateVector.DenseOfArray(inputArray).ToColumnMatrix());
// The Following is the old way I used to backpropagate for only 2 layers
//var outputCrossEntropy = ElementwiseMult(targets.ToColumnMatrix(), outputs); // I think this would be right for calculating cross entropy for the given target? maybe?
//Debug.Log($"Target: {targetsArray[0]}; Output: {outputs[0,0]}; Error: {outputError[0,0]}");
//// OutputGradient is the derivative of the activation function
//var outputGradient = ElementwiseMult(outputs, (1 - outputs));//outputs.Map(dsigmoid);
////var outputGradient = outputs - targets.ToColumnMatrix();//ElementwiseMult(outputs, (1 - outputs));
//// Scale stuff
//var scaledTemp1 = learningRate * ElementwiseMult(outputError, outputGradient);
//// Get the deltas to adjust the weights by for the Hidden-Output Connections
//var whodeltas = scaledTemp1 * hiddenOutput.Transpose();
//// Modify the H-O weights by calculated deltas;
//who = who + whodeltas;
//// Adjust the output bias by its delta (which is just the output gradient with errors and scaled)
//ob = ob + scaledTemp1;
//// Now do All that again for the I-H weights.
//var hiddenError = who.Transpose() * outputError;
////Debug.Log($"WHOT: {who.Transpose().ToString()}; Output Error: {outputError.ToString()}; Hidden Error: {hiddenError.ToString()}");
//// Calculate the Gradient of the hidden layer
//var hiddenGradient = ElementwiseMult(hiddenOutput, (1 - hiddenOutput)); //hiddenOutput.Map(dsigmoid);
//// scale stuff
//var scaledTemp2 = learningRate * ElementwiseMult(hiddenError, hiddenGradient);
//// Get Deltas to adjust the I-H weights
//var wihdeltas = scaledTemp2 * inputVector.ToColumnMatrix().Transpose();
//// Modify the I-H weights
//wih = wih + wihdeltas;
//// Adjust the hidden bias by its delta (which is just the hidden gradient with errors and scaled)
//hb = hb + scaledTemp2;
//// Should now be closer to the correct answer (hopefully)
}
private void Backpropagate(Matrix<double> outputError, Matrix<double> inputs)
{
List<Matrix<double>> errors = new List<Matrix<double>>();
errors.Add(outputError);
Matrix<double> gradient, scaled, deltas, error;
for(int i = layerOutputs.Count - 1; i >= 0; i--)
{
if(i == layerOutputs.Count - 1)
{
error = errors[0];
}
else
{
error = weights[i + 1].Transpose() * errors[layerOutputs.Count - 2 - i];
errors.Add(error);
}
gradient = ElementwiseMult(layerOutputs[i], 1 - layerOutputs[i]);
scaled = learningRate * ElementwiseMult(error, gradient);
if(i == 0)
{
// For last iteration
deltas = scaled * inputs.Transpose();
}
else
{
deltas = scaled * layerOutputs[i - 1].Transpose();
}
weights[i] += deltas;
biases[i] += scaled;
}
}
private Matrix<double> SoftMax(Matrix<double> matrix)
{
matrix = matrix.Map((x) => { return SpecialFunctions.Logistic(x); }, Zeros.Include);
return NormalizeColumnMatrix(matrix);
}
private Matrix<double> ElementwiseMult(Matrix<double> a, Matrix<double> b)
{
Matrix<double> result;
if(a != null && b != null && a.RowCount == b.RowCount && a.ColumnCount == b.ColumnCount)
{
result = CreateMatrix.Dense(a.RowCount, a.ColumnCount, 0d);
for(int i = 0; i < a.RowCount; i++)
{
for(int j = 0; j < a.ColumnCount; j++)
{
result[i, j] = a[i, j] * b[i, j];
}
}
}
else
{
throw new System.Exception("Argument dimensions do not match");
}
return result;
}
}
I keep looking at the old version to extrapolate to the looped version but I must be missing something.
I'm using MathNet.Numerics for my matrix operations.