Initial commit
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,95 @@
|
||||
from abc import abstractmethod
|
||||
|
||||
import numpy as np
|
||||
|
||||
from neural_net.transform_layer import Layer
|
||||
|
||||
class ActivationLayer(Layer):
|
||||
def __init__(self, index, input_dim, output_dim, weights=None, biases=None):
|
||||
super().__init__('ActivationLayer', index, input_dim, output_dim)
|
||||
self.type = 'ActivationLayer'
|
||||
self.subtype = ''
|
||||
|
||||
self.inputs = np.array([])
|
||||
self.output = np.array([])
|
||||
self.z = np.array([])
|
||||
self.gradient_clip = 1.0
|
||||
|
||||
# Initialize weights and biases
|
||||
if weights is not None:
|
||||
self.weights = weights
|
||||
else:
|
||||
self.initialize_weights()
|
||||
if biases is not None:
|
||||
self.biases = biases
|
||||
else:
|
||||
self.initialize_biases()
|
||||
|
||||
def describe(self):
|
||||
return f"{self.type} ({self.input_dim}x{self.output_dim} neurons, {self.subtype} activation)"
|
||||
|
||||
@abstractmethod
|
||||
def initialize_weights(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def initialize_biases(self):
|
||||
pass
|
||||
|
||||
def forward(self, inputs: np.array):
|
||||
self.inputs = inputs
|
||||
self.z = np.dot(self.inputs, self.weights) + self.biases
|
||||
self.output = self.activation(self.z) # Calls the implemented class's activation function (ie. Sigmoid)
|
||||
return self.output
|
||||
|
||||
def backward(self, dL_dout, learning_rate):
|
||||
"""
|
||||
Backpropagate the error and update weights and biases.
|
||||
:param dL_dout: Gradient of loss with respect to layer outputs
|
||||
:param learning_rate: Learning rate for weight updates
|
||||
:return: Gradient with respect to inputs for previous layer (dL/dinputs)
|
||||
"""
|
||||
# Activation derivative dout/dz
|
||||
# This tells you how much the output of the activation function changes with respect to the pre-activation value z.
|
||||
# Sigmoid derivative formula: σ(z) * (1 - σ(z))
|
||||
dout_dz = self.activation_derivative(self.output)
|
||||
|
||||
# Gradient of the loss with respect to weights (dL/dweights)
|
||||
# This represents how much the loss changes when the weights change.
|
||||
# Formula: dL/dweights = inputs × dL/dout × σ′(z)
|
||||
dL_dweights = np.clip(np.dot(self.inputs.T, dL_dout * dout_dz), -self.gradient_clip, self.gradient_clip)
|
||||
|
||||
dL_dbias = np.sum(dL_dout * dout_dz, axis=0)
|
||||
|
||||
# Gradient of the loss with respect to inputs (dL/dinputs)
|
||||
# This is the gradient of the loss with respect to the input of the neuron or layer, often needed if you want to backpropagate further.
|
||||
# Formula: dL / dinputs = dL/dout × σ′(z) × weights
|
||||
dL_dinputs = np.dot(dL_dout * dout_dz, self.weights.T)
|
||||
|
||||
# Clip gradients to prevent them from being too large
|
||||
# np.clip(dL_dweights, -10.0, 10.0, out=dL_dweights)
|
||||
# np.clip(dL_dbias, -10.0, 10.0, out=dL_dbias)
|
||||
|
||||
# Adjust weights and biases
|
||||
self.weights -= learning_rate * dL_dweights
|
||||
self.biases -= learning_rate * dL_dbias
|
||||
|
||||
return dL_dinputs, dL_dweights, dL_dbias, self.weights, self.biases
|
||||
|
||||
def reset(self):
|
||||
self.initialize_weights()
|
||||
self.initialize_biases()
|
||||
|
||||
@abstractmethod
|
||||
def activation(self, raw_outputs: np.array):
|
||||
"""
|
||||
Apply the activation function (Sigmoid, ReLU, etc.)
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def activation_derivative(self, outputs: np.array):
|
||||
"""
|
||||
Compute the derivative of the activation function
|
||||
"""
|
||||
pass
|
||||
@@ -0,0 +1,23 @@
|
||||
import numpy as np
|
||||
|
||||
from neural_net.activation_layers.activation_layer import ActivationLayer
|
||||
from neural_net.functions.activation import relu_activation, relu_derivative_activation
|
||||
|
||||
|
||||
class ReluLayer(ActivationLayer):
|
||||
def __init__(self, index, input_dim, output_dim, weights=None, biases=None):
|
||||
super().__init__(index, input_dim, output_dim, weights, biases)
|
||||
self.subtype = 'RELU'
|
||||
|
||||
def initialize_weights(self):
|
||||
# He initialization (input_dim x output_dim)
|
||||
self.weights = np.random.randn(self.input_dim, self.output_dim) * np.sqrt(2.0 / self.input_dim)
|
||||
|
||||
def initialize_biases(self):
|
||||
self.biases = np.zeros((1, self.output_dim)) # Biases initialized to zero
|
||||
|
||||
def activation(self, outputs: np.array):
|
||||
return relu_activation(outputs)
|
||||
|
||||
def activation_derivative(self, outputs: np.array):
|
||||
return relu_derivative_activation(outputs)
|
||||
@@ -0,0 +1,24 @@
|
||||
import numpy as np
|
||||
|
||||
from neural_net.activation_layers.activation_layer import ActivationLayer
|
||||
from neural_net.functions.activation import sigmoid_derivative_activation
|
||||
|
||||
|
||||
class SigmoidLayer(ActivationLayer):
|
||||
def __init__(self, input_dim, output_dim, weights=None, biases=None):
|
||||
super().__init__(input_dim, output_dim, weights, biases)
|
||||
self.subtype = 'Sigmoid'
|
||||
|
||||
def initialize_weights(self):
|
||||
# Xavier initialization for sigmoid activation
|
||||
limit = np.sqrt(6 / (self.input_dim + self.output_dim))
|
||||
self.weights = np.random.uniform(-limit, limit, (self.input_dim, self.output_dim))
|
||||
|
||||
def initialize_biases(self):
|
||||
self.biases = np.zeros((1, self.output_dim)) # Biases initialized to zero
|
||||
|
||||
def activation(self, outputs: np.array):
|
||||
return sigmoid_derivative_activation(outputs)
|
||||
|
||||
def activation_derivative(self, outputs: np.array):
|
||||
return sigmoid_derivative_activation(outputs)
|
||||
@@ -0,0 +1,47 @@
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
class Epoch:
|
||||
def __init__(self, epoch, inputs, labels, learning_rate, batch_size):
|
||||
self.epoch = epoch
|
||||
self.loss = -1.0
|
||||
self.duration = 0
|
||||
self.learning_rate = learning_rate
|
||||
self.batch_size = batch_size
|
||||
self.batches = []
|
||||
for i in range(0, len(inputs), self.batch_size):
|
||||
self.batches.append(TrainingBatch(i, inputs[i:i + batch_size], labels[i:i + batch_size]))
|
||||
self.layer_dl_gradients = []
|
||||
self.layer_dl_biases = []
|
||||
self.layer_weights = []
|
||||
self.finished = False
|
||||
|
||||
def start(self):
|
||||
self.start_time = time.time()
|
||||
|
||||
def finish(self, neural_net):
|
||||
self.finished = True
|
||||
self.trained_weights = neural_net.get_all_weights()
|
||||
self.end_time = time.time()
|
||||
self.duration = self.end_time - self.start_time
|
||||
|
||||
def all_predictions(self):
|
||||
return np.concatenate(np.array([batch.predictions for batch in self.batches]))
|
||||
def all_labels(self):
|
||||
return np.concatenate(np.array([batch.labels for batch in self.batches]))
|
||||
def all_inputs(self):
|
||||
return np.concatenate(np.array([batch.inputs for batch in self.batches]))
|
||||
|
||||
def print_epoch(self):
|
||||
print(f"Epoch {self.epoch}:")
|
||||
print(f"Loss: {self.loss}")
|
||||
print(f"dL / Gradients: {self.layer_dl_gradients}")
|
||||
print(f"dL / Bias: {self.layer_dl_gradients}")
|
||||
|
||||
class TrainingBatch:
|
||||
def __init__(self, batch_num, inputs, labels):
|
||||
self.batch_num = batch_num
|
||||
self.inputs = inputs
|
||||
self.labels = labels
|
||||
self.predictions = []
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,13 @@
|
||||
import numpy as np
|
||||
|
||||
def relu_activation(outputs):
|
||||
return np.maximum(0, outputs)
|
||||
|
||||
def relu_derivative_activation(outputs):
|
||||
return np.where(outputs > 0, 1, 0)
|
||||
|
||||
def sigmoid_activation(outputs):
|
||||
return 1 / (1 + np.exp(-outputs))
|
||||
|
||||
def sigmoid_derivative_activation(outputs):
|
||||
return outputs * (1 - outputs)
|
||||
@@ -0,0 +1,27 @@
|
||||
import numpy as np
|
||||
|
||||
def cross_entropy_loss(outputs, targets, clip=True):
|
||||
"""
|
||||
outputs: [
|
||||
[ 0.32, 0.12, 0.04 ],
|
||||
[ 0.62, 0.02, 0.14 ]
|
||||
]
|
||||
targets: [ 2, 1 ]
|
||||
:param outputs: np.array: Vector of all the predicted probabilities vectors
|
||||
:param targets: np.array: Vector of one-hot vectors representing the actual values
|
||||
:param clip: boolean, whether to clip the output probabilities
|
||||
:return:
|
||||
"""
|
||||
if clip:
|
||||
# Clipping the predictions for numerical stability
|
||||
outputs = np.clip(outputs, 1e-12, 1 - 1e-12)
|
||||
# Calculate cross-entropy loss and average over batch size
|
||||
m = targets.shape[0]
|
||||
log_likelihood = -np.log(outputs[range(m), targets])
|
||||
return np.sum(log_likelihood) / m # Average loss
|
||||
|
||||
def cross_entropy_derivative_loss(outputs, targets):
|
||||
# One-hot encode the labels
|
||||
y_true = np.eye(outputs.shape[1])[targets]
|
||||
# Derivative of cross-entropy with respect to softmax inputs
|
||||
return outputs - y_true
|
||||
@@ -0,0 +1,34 @@
|
||||
import numpy as np
|
||||
|
||||
from neural_net.activation_layers.relu_layer import ReluLayer
|
||||
from neural_net.functions.loss import cross_entropy_loss, cross_entropy_derivative_loss
|
||||
from neural_net.neural_net import NeuralNet
|
||||
from neural_net.transform_layer import SoftMaxLayer
|
||||
|
||||
class MNISTNeuralNet(NeuralNet):
|
||||
def __init__(self):
|
||||
super().__init__(layers=[
|
||||
ReluLayer(0, 784, 121),
|
||||
ReluLayer(1, 121, 10),
|
||||
SoftMaxLayer(2, 10)
|
||||
])
|
||||
|
||||
def backward(self, dL_dout, epoch):
|
||||
return super().backward(dL_dout, epoch)
|
||||
|
||||
def loss(self, y_pred: np.array, y_actual: np.array):
|
||||
return cross_entropy_loss(y_pred, y_actual)
|
||||
|
||||
def loss_derivative(self, y_pred: np.array, targets: np.array):
|
||||
return cross_entropy_derivative_loss(y_pred, targets)
|
||||
|
||||
def describe(self):
|
||||
"""Return a human-readable string of the model architecture."""
|
||||
architecture_info = ""
|
||||
for layer in self.layers:
|
||||
architecture_info += f"{layer.describe()}\n"
|
||||
return architecture_info.strip()
|
||||
|
||||
def predict(self, inputs):
|
||||
raw_outputs = super().predict(inputs)
|
||||
return raw_outputs, raw_outputs.argmax(axis=1)
|
||||
@@ -0,0 +1,127 @@
|
||||
from abc import abstractmethod
|
||||
from enum import Enum
|
||||
|
||||
import numpy as np
|
||||
|
||||
from neural_net.epoch import Epoch
|
||||
from neural_net.transform_layer import Layer
|
||||
|
||||
|
||||
class ModelData:
|
||||
def __init__(self, training_inputs, training_targets, test_inputs, test_targets):
|
||||
self.is_loaded = False
|
||||
self.training_inputs = training_inputs
|
||||
self.training_labels = training_targets
|
||||
self.test_inputs = test_inputs
|
||||
self.test_labels = test_targets
|
||||
|
||||
|
||||
# class TrainingSession:
|
||||
# def __init__(self, training_data: ModelData, learning_rate: float, nr_epochs: int, batch_size: int = 1000):
|
||||
# self.training_data = training_data
|
||||
# self.learning_rate = learning_rate
|
||||
# self.nr_epochs = nr_epochs
|
||||
# self.batch_size = batch_size
|
||||
# self.epochs: [Epoch] = []
|
||||
# for i in range(self.nr_epochs):
|
||||
# self.epochs.append(
|
||||
# Epoch(i, self.training_data.training_inputs, self.training_data.training_labels, self.batch_size))
|
||||
#
|
||||
# def get_total_training_duration(self):
|
||||
# duration = 0.0
|
||||
# for epoch in self.epochs:
|
||||
# duration += epoch.duration
|
||||
# return duration
|
||||
|
||||
|
||||
class NeuralNet:
|
||||
def __init__(self, layers: [Layer]):
|
||||
self.layers = layers
|
||||
self.last_loss = None
|
||||
self.last_accuracy = None
|
||||
|
||||
def forward(self, inputs):
|
||||
outputs = inputs
|
||||
for layer in self.layers:
|
||||
outputs = layer.forward(outputs)
|
||||
return outputs
|
||||
|
||||
def reset(self):
|
||||
for layer in self.layers:
|
||||
layer.reset()
|
||||
|
||||
def backward(self, dL_dout, epoch):
|
||||
layer_dl_gradients = []
|
||||
layer_dl_bias = []
|
||||
layer_weights = []
|
||||
layer_biases = []
|
||||
|
||||
for idx, layer in reversed(list(enumerate(self.layers))):
|
||||
dL_dout, dl_gradients, dl_biases, weights, biases = layer.backward(dL_dout, epoch.learning_rate)
|
||||
|
||||
if dl_gradients is not None:
|
||||
layer_dl_gradients.append(dl_gradients)
|
||||
if dl_biases is not None:
|
||||
layer_dl_bias.append(dl_biases)
|
||||
if weights is not None:
|
||||
layer_weights.append(weights)
|
||||
if biases is not None:
|
||||
layer_biases.append(biases)
|
||||
|
||||
return layer_dl_gradients, layer_dl_bias, layer_weights, layer_biases
|
||||
|
||||
# def train(self, training_run: TrainingRun):
|
||||
# self.training_runs.append(training_run)
|
||||
#
|
||||
# for epoch in training_run.epochs:
|
||||
# epoch.start()
|
||||
#
|
||||
# for batch in epoch.batches:
|
||||
# batch.predictions = self.forward(batch.inputs)
|
||||
# dL_dout = self.loss_derivative(batch.predictions, batch.labels)
|
||||
#
|
||||
# layer_dl_gradients, layer_dl_biases, layer_weights, layer_biases = self.backward(dL_dout, training_run.learning_rate, epoch)
|
||||
# epoch.layer_dl_gradients.append(layer_dl_gradients)
|
||||
# epoch.layer_dl_biases.append(layer_dl_biases)
|
||||
#
|
||||
# epoch.finish()
|
||||
# epoch.loss = self.loss(epoch.all_predictions(), epoch.all_labels())
|
||||
#
|
||||
# if training_run.epoch_callback is not None:
|
||||
# training_run.epoch_callback(training_run, epoch)
|
||||
#
|
||||
# self.recalculate_loss(training_run.training_data.test_inputs, training_run.training_data.test_labels)
|
||||
# self.recalculate_loss(training_run.training_data.test_inputs, training_run.training_data.test_labels)
|
||||
|
||||
def get_all_weights(self):
|
||||
all_weights = []
|
||||
for layer in self.layers:
|
||||
if hasattr(layer, 'weights'):
|
||||
all_weights.append(layer.weights)
|
||||
return all_weights
|
||||
|
||||
def recalculate_accuracy(self, inputs, labels):
|
||||
raw_outputs = self.forward(inputs)
|
||||
predictions = raw_outputs.argmax(axis=1)
|
||||
num_correct_predictions = 0
|
||||
for idx, prediction in enumerate(predictions):
|
||||
if prediction == labels[idx]:
|
||||
num_correct_predictions += 1
|
||||
self.last_accuracy = num_correct_predictions / len(predictions)
|
||||
return self.last_accuracy
|
||||
|
||||
def recalculate_loss(self, inputs, labels):
|
||||
raw_outputs = self.forward(inputs)
|
||||
self.last_loss = self.loss(np.array(raw_outputs), np.array(labels))
|
||||
return self.last_loss
|
||||
|
||||
@abstractmethod
|
||||
def loss(self, outputs: np.array, labels: np.array):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def loss_derivative(self, outputs: np.array, labels: np.array):
|
||||
pass
|
||||
|
||||
def predict(self, inputs):
|
||||
return self.forward(inputs)
|
||||
@@ -0,0 +1,65 @@
|
||||
from neural_net.epoch import Epoch
|
||||
from neural_net.neural_net import NeuralNet, ModelData
|
||||
|
||||
|
||||
class NeuralNetTrainer:
|
||||
def __init__(self, neural_net: NeuralNet, model_data: ModelData, learning_rate: float, batch_size: int):
|
||||
self.neural_net = neural_net
|
||||
self.model_data = model_data
|
||||
self.is_running = False
|
||||
self.epoch_history = []
|
||||
self.learning_rate = learning_rate
|
||||
self.batch_size = batch_size
|
||||
|
||||
def set_learning_rate(self, learning_rate: float):
|
||||
self.learning_rate = learning_rate
|
||||
|
||||
def set_batch_size(self, batch_size: int):
|
||||
self.batch_size = batch_size
|
||||
|
||||
def run_epoch(self):
|
||||
epoch = Epoch(len(self.epoch_history),
|
||||
self.model_data.training_inputs,
|
||||
self.model_data.training_labels,
|
||||
self.learning_rate,
|
||||
self.batch_size
|
||||
)
|
||||
self._train_one_epoch(epoch)
|
||||
return epoch
|
||||
|
||||
def start(self, on_epoch_finish=None, on_finish=None):
|
||||
self.is_running = True
|
||||
while True:
|
||||
# Stop function was called causing the trainer to reset
|
||||
if not self.is_running:
|
||||
break
|
||||
|
||||
# Perform one epoch of training
|
||||
# In the future, we will apply a learning-rate algorithm
|
||||
epoch = self.run_epoch()
|
||||
|
||||
if on_epoch_finish is not None:
|
||||
on_epoch_finish(epoch)
|
||||
|
||||
if on_finish is not None:
|
||||
on_finish()
|
||||
self.stop()
|
||||
|
||||
def stop(self):
|
||||
if self.is_running:
|
||||
self.is_running = False
|
||||
|
||||
def _train_one_epoch(self, epoch: Epoch):
|
||||
epoch.start()
|
||||
|
||||
for batch in epoch.batches:
|
||||
batch.predictions = self.neural_net.forward(batch.inputs)
|
||||
dL_dout = self.neural_net.loss_derivative(batch.predictions, batch.labels)
|
||||
|
||||
layer_dl_gradients, layer_dl_biases, layer_weights, layer_biases = self.neural_net.backward(dL_dout, epoch)
|
||||
epoch.layer_dl_gradients.append(layer_dl_gradients)
|
||||
epoch.layer_dl_biases.append(layer_dl_biases)
|
||||
|
||||
epoch.finish(self.neural_net)
|
||||
epoch.loss = self.neural_net.loss(epoch.all_predictions(), epoch.all_labels())
|
||||
self.epoch_history.append(epoch)
|
||||
@@ -0,0 +1,73 @@
|
||||
from abc import abstractmethod
|
||||
|
||||
import numpy as np
|
||||
|
||||
class Layer:
|
||||
def __init__(self, type, index, input_dim, output_dim):
|
||||
self.type = type
|
||||
self.index = index
|
||||
self.input_dim = input_dim
|
||||
self.output_dim = output_dim
|
||||
|
||||
@abstractmethod
|
||||
def forward(self, inputs):
|
||||
raise NotImplementedError("This should be overridden by subclasses")
|
||||
|
||||
@abstractmethod
|
||||
def backward(self, dL_dout, learning_rate):
|
||||
raise NotImplementedError("This should be overridden by subclasses")
|
||||
|
||||
@abstractmethod
|
||||
def reset(self):
|
||||
raise NotImplementedError("This should be overridden by subclasses")
|
||||
|
||||
class TransformLayer(Layer):
|
||||
def __init__(self, index, size):
|
||||
super().__init__('TransformLayer', index, size, size)
|
||||
|
||||
def describe(self):
|
||||
return self.type
|
||||
|
||||
def forward(self, inputs):
|
||||
raise NotImplementedError("This should be overridden by subclasses")
|
||||
|
||||
def backward(self, dL_dout, learning_rate):
|
||||
return dL_dout, None, None, None, None # This is the gradient to propagate to the previous layer
|
||||
|
||||
def reset(self):
|
||||
pass
|
||||
|
||||
class NormalizeLayer(TransformLayer):
|
||||
def __init__(self, index, size):
|
||||
super().__init__(index, size)
|
||||
self.type = 'NormalizeLayer'
|
||||
|
||||
def forward(self, inputs):
|
||||
"""
|
||||
Normalizes the input vector.
|
||||
[1, 5, 5, 3, 6] => [0.05, 0.25, 0.25, 0.15, 0.3]
|
||||
:param inputs: np.array(float)
|
||||
:return: np.array(float)
|
||||
"""
|
||||
return inputs / inputs.sum()
|
||||
|
||||
class SoftMaxLayer(TransformLayer):
|
||||
def __init__(self, index, size):
|
||||
super().__init__(index, size)
|
||||
self.type = 'SoftMaxLayer'
|
||||
|
||||
def forward(self, inputs):
|
||||
"""
|
||||
Normalizes the input vector, but "pushes" higher values to dominate the
|
||||
probability distribution
|
||||
[1, 5, 5, 3, 6] => [0.02, 0.26, 0.26, 0.10, 0.36]
|
||||
:param inputs: np.array(float)
|
||||
:return: np.array(float)
|
||||
"""
|
||||
input_ex = np.exp(inputs - inputs.max()) # Subtract max for numerical stability
|
||||
s = np.sum(input_ex, axis=-1, keepdims=True)
|
||||
|
||||
# To prevent division by zero, ensure that the sum is not zero
|
||||
if np.any(s == 0):
|
||||
return np.ones_like(input_ex) / input_ex.shape[-1] # Return a uniform distribution if sum is 0
|
||||
return input_ex / s
|
||||
Reference in New Issue
Block a user