From 642bba1198a459369a9755a8ad4b0e9780972967 Mon Sep 17 00:00:00 2001 From: Michelle Date: Thu, 6 Aug 2026 09:53:03 +0200 Subject: [PATCH] Softmax + Categorical Crossentropy I fixed the activation+loss function, you can't select it directly, it automaticly uses it if it can. I also fixed one-hot generator. Still haven't tested the activation + loss nor any other backward function. I'll do that when I get to the optimizers which is next. --- include/config/omp.hpp | 42 +-- .../activation/activation_relu.hpp | 10 + .../activation/activation_softmax.hpp | 11 + ..._softmax_loss_categorical_crossentropy.hpp | 42 +-- include/neural_network/layer/layer.hpp | 20 +- include/neural_network/layer/layer_dense.hpp | 34 +- include/neural_network/loss/loss.hpp | 33 +- .../loss/loss_categorical_crossentropy.hpp | 29 +- include/neural_network/model/model.hpp | 124 ++++--- main.cpp | 47 ++- src/math/mean.cpp | 18 +- .../activation/activation_relu.cpp | 7 +- ..._softmax_loss_categorical_crossentropy.cpp | 61 +--- src/neural_network/layer/layer_dense.cpp | 2 +- src/neural_network/loss/loss.cpp | 27 +- .../loss/loss_categorical_crossentropy.cpp | 87 ++--- src/neural_network/model/model.cpp | 324 +++++++++++++++--- src/tensor/generators/one_hot.cpp | 87 +++-- 18 files changed, 627 insertions(+), 378 deletions(-) diff --git a/include/config/omp.hpp b/include/config/omp.hpp index 6c4075d..95ef7f7 100644 --- a/include/config/omp.hpp +++ b/include/config/omp.hpp @@ -134,18 +134,13 @@ // Expands to: // #pragma omp parallel for if(condition) // num_threads(PANIC_OMP_NUM_THREADS) - // schedule(static) // reduction(operation:variable) // // Runs the loop in parallel only when condition is true. // Each thread receives a private copy of variable. // Afterward, OpenMP combines those copies using operation. - // schedule(static) assigns fixed groups of iterations to each thread. #define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \ - _Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \ - num_threads(PANIC_OMP_NUM_THREADS) \ - schedule(static) \ - reduction(operation:variable))) + _Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable) num_threads(PANIC_OMP_NUM_THREADS))) #else @@ -165,19 +160,13 @@ _Pragma(PANIC_STRINGIFY(omp parallel for if(condition))) // Expands to: - // #pragma omp parallel for if(condition) - // num_threads(PANIC_OMP_NUM_THREADS) - // schedule(static) - // reduction(operation:variable) + // #pragma omp parallel for if(condition) reduction(operation:variable) // // Runs the loop in parallel only when condition is true. // Each thread receives a private copy of variable. // Afterward, OpenMP combines those copies using operation. - // schedule(static) assigns fixed groups of iterations to each thread. #define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \ - _Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \ - schedule(static) \ - reduction(operation:variable))) + _Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable))) #endif @@ -199,25 +188,16 @@ // } // // That means the same code still works on microcontrollers and non-OpenMP builds. + + // Without OpenMP, the normal for-loop remains. #define PANIC_OMP_PARALLEL_FOR + + // Check that condition is syntactically valid, but do not evaluate it. #define PANIC_OMP_PARALLEL_FOR_IF(condition) static_cast(sizeof(condition)); + // The operation and variable are only needed by the OpenMP pragma. + // The serial loop itself performs the calculation normally. + #define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) static_cast(sizeof(condition)); + #endif -//--------------------------------------------------------------------------------------------------------------------------- -// TYPE DESCRIPTION -//--------------------------------------------------------------------------------------------------------------------------- - -// None. - -//--------------------------------------------------------------------------------------------------------------------------- -// VARIABLE DESCRIPTION -//--------------------------------------------------------------------------------------------------------------------------- - -// None. - -//--------------------------------------------------------------------------------------------------------------------------- -// FUNCTION PROTOTYPE -//--------------------------------------------------------------------------------------------------------------------------- - -// None. diff --git a/include/neural_network/activation/activation_relu.hpp b/include/neural_network/activation/activation_relu.hpp index 2172932..8bbac62 100644 --- a/include/neural_network/activation/activation_relu.hpp +++ b/include/neural_network/activation/activation_relu.hpp @@ -70,6 +70,16 @@ struct activation_relu : public layer{ */ ~activation_relu() = default; + /** + * @brief get_type function for layer + * + * @returns the layer type + * + */ + layer_type get_type() const override { + return layer_type::activation_relu; + } + /** * @brief Forward function for layer * diff --git a/include/neural_network/activation/activation_softmax.hpp b/include/neural_network/activation/activation_softmax.hpp index 1c9fbec..8214795 100644 --- a/include/neural_network/activation/activation_softmax.hpp +++ b/include/neural_network/activation/activation_softmax.hpp @@ -70,6 +70,17 @@ struct activation_softmax : public layer{ */ ~activation_softmax() = default; + + /** + * @brief get_type function for layer + * + * @returns the layer type + * + */ + layer_type get_type() const override { + return layer_type::activation_softmax; + } + /** * @brief Forward function for layer * diff --git a/include/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp b/include/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp index c52ebd5..7814a00 100644 --- a/include/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp +++ b/include/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp @@ -54,47 +54,9 @@ namespace panic{ * * The struct is used in PANIC nural_network library. */ -struct activation_softmax_loss_categorical_crossentropy: loss{ - - activation_softmax activation; - loss_categorical_crossentropy loss; - - - /** - * @brief Empthy constructor - * - */ - activation_softmax_loss_categorical_crossentropy(); - - /** - * @brief Default de-constructor - * - */ - ~activation_softmax_loss_categorical_crossentropy() = default; - - /** - * @brief forward function to calculate losses - * - * @param y_pred Matrix of model predection. - * @param y_true Vector of true label of data. - * - */ - bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override; - - /** - * @brief forward function to calculate losses - * - * @param y_pred Matrix of model predection. - * @param y_true Vector of true label of data. - * - * @Note Overloaded if one-shot endcoded - * is used. - */ - bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override; - - - +struct activation_softmax_loss_categorical_crossentropy{ + panic::tensor::real_matrix dinputs; /** * @brief backward function to calculate from losses diff --git a/include/neural_network/layer/layer.hpp b/include/neural_network/layer/layer.hpp index a8520df..7cd8899 100644 --- a/include/neural_network/layer/layer.hpp +++ b/include/neural_network/layer/layer.hpp @@ -44,6 +44,14 @@ namespace panic{ +enum struct layer_type { + unknown, + layer_dense, + activation_relu, + activation_softmax +}; + + /** * @brief Base layer for the rest of the neural network library to use * @@ -85,6 +93,16 @@ struct layer{ virtual ~layer() = default; + /** + * @brief Virtual layer_type function for derivative layers + * + * @Note This returns the type of layer it is + * unknown be default + */ + virtual layer_type get_type() const { + return layer_type::unknown; + } + /** * @brief Virtual forward function for derivative layers * @@ -103,7 +121,7 @@ struct layer{ * @Note It's equal to 0 because it make the derivative * object NEEDS to have these function to work. */ - virtual bool backward(const panic::tensor::real_matrix& dinputs) = 0; + virtual bool backward(const panic::tensor::real_matrix& dvalues) = 0; }; diff --git a/include/neural_network/layer/layer_dense.hpp b/include/neural_network/layer/layer_dense.hpp index a4297fe..13797a7 100644 --- a/include/neural_network/layer/layer_dense.hpp +++ b/include/neural_network/layer/layer_dense.hpp @@ -58,30 +58,6 @@ namespace panic{ */ struct layer_dense : trainable_layer{ - /** - * @brief Emphty matrix to store input data - * - */ - panic::tensor::real_matrix ipnuts; - - /** - * @brief Emphty weight matrix to store layer weights - * - * Weight shape: - * input_size x neuron_count - */ - panic::tensor::real_matrix weights; - panic::tensor::real_matrix dweights; - - /** - * @brief Emphty bias vector to store layer bias - * - * Bias shape: - * 1 x neuron_count - */ - panic::tensor::real_vector biases; - panic::tensor::real_vector dbiases; - /** * @brief Empthy constructor * @@ -103,6 +79,16 @@ struct layer_dense : trainable_layer{ */ ~layer_dense() = default; + /** + * @brief get_type function for layer + * + * @returns the layer type + * + */ + layer_type get_type() const override { + return layer_type::layer_dense; + } + /** * @brief Forward function for layer * diff --git a/include/neural_network/loss/loss.hpp b/include/neural_network/loss/loss.hpp index a45b33e..fc4751f 100644 --- a/include/neural_network/loss/loss.hpp +++ b/include/neural_network/loss/loss.hpp @@ -44,6 +44,11 @@ namespace panic{ namespace neural_network{ +enum struct loss_type { + unknown, + categorical_crossentropy +}; + /** * @brief Base loss for the rest of the neural network library to use @@ -58,7 +63,7 @@ namespace panic{ struct loss{ /** - * @brief Emphty vector to store sample losses + * @brief Emphty vector to store sample losses for each sample in the batch. * */ panic::tensor::real_vector sample_losses; @@ -66,19 +71,15 @@ struct loss{ /** * @brief Mean loss over the entire batch. */ - panic::types::real_t data_loss; + panic::types::real_t data_loss = 0; /** - * @brief Matrix for backwards pass + * @brief Gradient with respect to the loss input. + * + * This will be used later during the backward pass. */ panic::tensor::real_matrix dinputs; - /** - * @brief Matrix for output of loss function - */ - panic::tensor::real_matrix outputs; - - /** * @brief Default de-constructor * @@ -86,6 +87,16 @@ struct loss{ virtual ~loss() = default; + /** + * @brief Virtual loss_type function for derivative losses + * + * @Note This returns the type of loss it is + * unknown be default + */ + virtual loss_type get_type() const { + return loss_type::unknown; + } + /** * @brief Virtual forward function for derivative loss functions * @@ -146,7 +157,7 @@ struct loss{ * @param y_true Vector of true label of data. * */ - virtual bool calculate( + bool calculate( const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true); @@ -157,7 +168,7 @@ struct loss{ * @param y_true Matrix of true label of data. * */ - virtual bool calculate( + bool calculate( const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true); diff --git a/include/neural_network/loss/loss_categorical_crossentropy.hpp b/include/neural_network/loss/loss_categorical_crossentropy.hpp index b80daa4..25de2a4 100644 --- a/include/neural_network/loss/loss_categorical_crossentropy.hpp +++ b/include/neural_network/loss/loss_categorical_crossentropy.hpp @@ -53,10 +53,17 @@ namespace panic{ * * The struct is used for PANIC neural_network library. */ -struct loss_categorical_crossentropy: loss{ - - +struct loss_categorical_crossentropy: public loss{ + /** + * @brief get_type function for loss + * + * @returns the loss type + * + */ + loss_type get_type() const override { + return loss_type::categorical_crossentropy; + } /** * @brief forward function to calculate losses @@ -65,9 +72,7 @@ struct loss_categorical_crossentropy: loss{ * @param y_true Vector of true label of data. * */ - bool forward( - const panic::tensor::real_matrix& y_pred, - const panic::tensor::uint_vector& y_true)override; + bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override; /** * @brief forward function to calculate losses @@ -78,9 +83,7 @@ struct loss_categorical_crossentropy: loss{ * @Note Overloaded if one-shot endcoded * is used. */ - bool forward( - const panic::tensor::real_matrix& y_pred, - const panic::tensor::real_matrix& y_true)override; + bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override; /** @@ -90,9 +93,7 @@ struct loss_categorical_crossentropy: loss{ * @param y_true Vector of true label of data. * */ - bool backward( - const panic::tensor::real_matrix& dvalues, - const panic::tensor::uint_vector& y_true) override; + bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::uint_vector& y_true) override; /** * @brief backward function to calculate from losses @@ -103,9 +104,7 @@ struct loss_categorical_crossentropy: loss{ * @Note Overloaded if one-shot endcoded * is used. */ - bool backward( - const panic::tensor::real_matrix& dvalues, - const panic::tensor::real_matrix& y_true) override; + bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::real_matrix& y_true) override; }; diff --git a/include/neural_network/model/model.hpp b/include/neural_network/model/model.hpp index 4af8f3f..f1ef1dd 100644 --- a/include/neural_network/model/model.hpp +++ b/include/neural_network/model/model.hpp @@ -44,6 +44,7 @@ #include // fully connected dense layer #include +#include #include #include @@ -71,44 +72,77 @@ namespace panic{ struct model{ /** - * @brief a pointer to a pointer of layers + * @brief Array of pointers to all model layers. * - * An example: - * layers[0] points to a layer_dense - * layers[1] points to an actication function - * layers[2] points to another layer_dense - * - * @note The model owns these layers and deletes them in clear(). - * + * Example: + * layers[0] points to a layer_dense + * layers[1] points to an activation_ReLU + * layers[2] points to another layer_dense + * + * @note The model owns every object referenced by this array. + * clear() deletes each layer and then deletes the array. */ layer** layers; + /** + * @brief Number of layers currently stored in layers. + */ + panic::types::uint_t layer_count; + + /** + * @brief Array of pointers to the trainable layers. + * + * @note These pointers refer to objects already owned through layers. + * Do not delete the individual objects through this array. + * Only the pointer array itself is owned separately. + */ trainable_layer** trainable_layers; + + /** + * @brief Number of trainable-layer pointers. + */ panic::types::uint_t trainable_layer_count; + + /** + * @brief Configured loss function. + * + * @note The model owns this object and deletes it in clear(). + */ + loss* loss_function; + + /** + * @brief Configured optimizer. + * + * @note The model owns this object and deletes it in clear(). + */ optimizer* optimizer_function; /** - * @brief a pointer the loss function + * @brief Optimized backward helper for the combination of + * Softmax and categorical cross-entropy. * - * - * @note The model owns these layers and deletes them in clear(). - * + * This is a normal member object, not a dynamically allocated object. */ - loss* loss_function; + activation_softmax_loss_categorical_crossentropy softmax_classifier_output; - // Number of layers currently stored in the model. /** - * @brief Stores the number of layers - * + * @brief Whether the optimized Softmax + categorical + * cross-entropy backward path should be used. */ - panic::types::uint_t layer_count; + bool use_softmax_classifier_output; - // model output (may be deleted and also used for debug) + + /** + * @brief Output of the final model layer. + */ panic::tensor::real_matrix outputs; - // model dinputs (may be deleted and also used for debug) + + /** + * @brief Gradient with respect to the model input. + */ panic::tensor::real_matrix dinputs; /** @@ -231,16 +265,32 @@ struct model{ * * Computes: * @code - * model.bacward(dvalues_data_matrix) + * model.backward(model_output, y_true_values) * @endcode * - * @param dvalues diput data. + * @param output Model output. + * @param y_true True values for data. * * @return true looped over every layer. - * * */ - bool backward(const panic::tensor::real_matrix& dvalues); + bool backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true); + + /** + * @brief Loops over all layers backward function + * + * Computes: + * @code + * model.backward(model_output, y_true_values) + * @endcode + * + * @param output Model output. + * @param y_true True values for data. + * + * @return true looped over every layer. + * + */ + bool backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true); /** @@ -259,24 +309,6 @@ struct model{ bool add_loss_categorical_crossentropy(); - - /** - * @brief Adds activation softmax AND loss for categorical crossentropy. - * - * Computes: - * @code - * model.activation_softmax_loss_categorical_crossentropy(); - * @endcode - * - * - * @return true if activation and loss is added - * - * @note This function is convenient, but it allocates a new layer. - */ - bool activation_softmax_loss_categorical_crossentropy(); - - - /** * @brief Adds optimizer_sgd to the model * @@ -295,7 +327,15 @@ struct model{ - + /** + * @brief Finalizes the model configuration. + * + * Detects whether the model can use the optimized + * Softmax + categorical-cross-entropy backward pass. + * + * @return true if the model configuration is valid. + */ + bool finalize(); /** * @brief Trains the model with input data diff --git a/main.cpp b/main.cpp index b4601a3..3b04e5f 100644 --- a/main.cpp +++ b/main.cpp @@ -88,15 +88,6 @@ // #define TEST_FALG 1 - - - - - - - - - //--------------------------------------------------------------------------------------------------------------------------- // Function Name : benchmark_omp_min_work // @@ -725,28 +716,46 @@ int main(void) { panic::neural_network::model mymodel; // Create Dense layer with 2 input features and 3 output values - mymodel.add_layer_dense(2,3); + if (!mymodel.add_layer_dense(2,3)){ + return false; + } // Create an activation ReLU layer - mymodel.add_activation_relu(); - + if (!mymodel.add_activation_relu()){ + return false; + } + // Create a second dense layer with 3 inputs and 3 outputs - mymodel.add_layer_dense(3, 3); + if (!mymodel.add_layer_dense(3, 3)){ + return false; + } + // Create activation softmax layer - mymodel.add_activation_softmax(); + if (!mymodel.add_activation_softmax()){ + return false; + } + - mymodel.add_loss_categorical_crossentropy(); - //mymodel.activation_softmax_loss_categorical_crossentropy(); + if (!mymodel.add_loss_categorical_crossentropy()){ + return false; + } + + if (!mymodel.finalize()){ + return false; + } - mymodel.add_optimizer_sgd(); + //mymodel.add_optimizer_sgd(); panic::types::uint_t epochs = 10; panic::types::uint_t print_every = 1; - mymodel.train(X, y, epochs, print_every); + if (!mymodel.train(X, y, epochs, print_every)){ + std::cout << "Training failed" << std::endl; + return false; + } - + return 0; diff --git a/src/math/mean.cpp b/src/math/mean.cpp index 3aa0ed8..d9b0001 100644 --- a/src/math/mean.cpp +++ b/src/math/mean.cpp @@ -72,6 +72,10 @@ namespace panic { template panic::types::real_t mean(const panic::tensor::vector& a){ + if (a.size() == static_cast(0)){ + return static_cast(0); + } + const panic::types::real_t sum = static_cast(panic::math::sum(a)); return sum / static_cast(a.size()); @@ -104,6 +108,10 @@ panic::types::real_t mean(const panic::tensor::matrix& A){ panic::types::uint_t cols = A.cols(); panic::types::uint_t work = rows*cols; + if (work == static_cast(0)){ + return static_cast(0); + } + panic::types::real_t sum = static_cast (panic::math::sum(A)); return sum / static_cast(work); @@ -138,6 +146,10 @@ bool mean_rowwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto panic::types::uint_t cols = A.cols(); panic::types::uint_t work = rows*cols; + if (work == static_cast(0)){ + return static_cast(0); + } + if ( !b.resize(rows) ){ return false; } @@ -245,6 +257,10 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto panic::types::uint_t cols = A.cols(); panic::types::uint_t work = rows*cols; + if (work == static_cast(0)){ + return static_cast(0); + } + if ( !b.resize(cols) ){ return false; } @@ -256,7 +272,7 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto // Each thread handles separate cols and writes to a separate b[i]. PANIC_OMP_PARALLEL_FOR_IF(work > mean_omp_min_work) for (panic::types::uint_t i = 0; i < cols; ++i){ - b[i] /= cols; + b[i] /= rows; } return true; diff --git a/src/neural_network/activation/activation_relu.cpp b/src/neural_network/activation/activation_relu.cpp index c2a849e..c986bfb 100644 --- a/src/neural_network/activation/activation_relu.cpp +++ b/src/neural_network/activation/activation_relu.cpp @@ -52,6 +52,8 @@ * worker threads can be larger than the work itself. */ static const panic::types::uint_t activation_ReLU_omp_min_size = 500; + +static const panic::types::real_t almost_zero = static_cast (1e-7); //--------------------------------------------------------------------------------------------------------------------------- // INPLEMENTATION //--------------------------------------------------------------------------------------------------------------------------- @@ -80,7 +82,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){ inputs = input_data; - panic::math::clip_lower(inputs, 0.0f, outputs); + panic::math::clip_lower(inputs, almost_zero, outputs); return true; @@ -97,8 +99,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){ bool activation_relu::backward(const panic::tensor::real_matrix& dvalues){ // Zero gradients where input values were negative - panic::types::real_t zero = 0; - dinputs = panic::math::clip_lower(dvalues, zero); + dinputs = panic::math::clip_lower(dvalues, almost_zero); return true; diff --git a/src/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.cpp b/src/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.cpp index f04d6eb..889b2c4 100644 --- a/src/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.cpp +++ b/src/neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.cpp @@ -52,7 +52,7 @@ * Small vectors and matrices are kept serial because the overhead of starting * worker threads can be larger than the work itself. */ -static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 500; +static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 250; //--------------------------------------------------------------------------------------------------------------------------- // INPLEMENTATION //--------------------------------------------------------------------------------------------------------------------------- @@ -60,65 +60,6 @@ static const panic::types::uint_t activation_softmax_loss_categorical_crossentro namespace panic{ namespace neural_network{ - -//-------------------------------------------------------------------------------------------------------------------------- -// Constructor Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy -// -// Description: -// Creates an empty layer. -//-------------------------------------------------------------------------------------------------------------------------- -activation_softmax_loss_categorical_crossentropy::activation_softmax_loss_categorical_crossentropy() { -} - - -//-------------------------------------------------------------------------------------------------------------------------- -// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward -// -// Description: -// Calculated the forward pass -//-------------------------------------------------------------------------------------------------------------------------- -bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){ - - // Output layers activation function - if (!activation.forward(y_pred)){ - return false; - } - // Set the output - outputs = activation.outputs; - - // calculate the loss value. - if (!loss.calculate(outputs, y_true)){ - return false; - } - - return true; -} - -//-------------------------------------------------------------------------------------------------------------------------- -// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward -// -// Description: -// Calculated the forward pass -//-------------------------------------------------------------------------------------------------------------------------- -bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true){ - - // Output layers activation function - if (!activation.forward(y_pred)){ - return false; - } - // Set the output - outputs = activation.outputs; - - // calculate the loss value. - if (!loss.calculate(outputs, y_true)){ - return false; - } - - return true; -} - - - //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.backward // diff --git a/src/neural_network/layer/layer_dense.cpp b/src/neural_network/layer/layer_dense.cpp index 6e88d76..f2f3bb4 100644 --- a/src/neural_network/layer/layer_dense.cpp +++ b/src/neural_network/layer/layer_dense.cpp @@ -86,7 +86,7 @@ layer_dense::layer_dense() { layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons) { weights.resize(input_size, neurons); panic::random::uniform(weights); - panic::math::mul(weights, 0.01f, weights); + panic::math::mul(weights, static_cast(0.01), weights); biases.resize(neurons); biases.fill(0); diff --git a/src/neural_network/loss/loss.cpp b/src/neural_network/loss/loss.cpp index 5cb63db..f01ac3f 100644 --- a/src/neural_network/loss/loss.cpp +++ b/src/neural_network/loss/loss.cpp @@ -69,18 +69,21 @@ bool loss::calculate( const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){ - // Calculate sample losses + data_loss = static_cast(0); + + // Calls the derived loss function's forward(). if (!forward(y_pred, y_true)){ return false; } + // forward() must produce one loss per sample. + if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) { + return false; + } + // Calculate mean loss data_loss = panic::math::mean(sample_losses); - // This is so the model can use every loss functions - // It's a problem to get the output when useing an activation+loss function - outputs = y_pred; - return true; } @@ -94,17 +97,21 @@ bool loss::calculate( const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true){ + data_loss = static_cast(0); + + // Calls the derived loss function's forward(). if (!forward(y_pred, y_true)){ return false; } + // forward() must produce one loss per sample. + if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) { + return false; + } + + // Calculate mean loss data_loss = panic::math::mean(sample_losses); - - // This is so the model can use every loss functions - // It's a problem to get the output when useing an activation+loss function - outputs = y_pred; - return true; } diff --git a/src/neural_network/loss/loss_categorical_crossentropy.cpp b/src/neural_network/loss/loss_categorical_crossentropy.cpp index cccd83f..c359a2f 100644 --- a/src/neural_network/loss/loss_categorical_crossentropy.cpp +++ b/src/neural_network/loss/loss_categorical_crossentropy.cpp @@ -57,7 +57,7 @@ * Small vectors and matrices are kept serial because the overhead of starting * worker threads can be larger than the work itself. */ -static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 500; +static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 250; static const panic::types::real_t clip_min = 1e-7; static const panic::types::real_t clip_max = 1 - 1e-7; @@ -75,20 +75,34 @@ namespace panic{ // Description: // Default implementation for catecorical labels. Derived classes can override it. //-------------------------------------------------------------------------------------------------------------------------- -bool loss_categorical_crossentropy::forward( - const panic::tensor::real_matrix& y_pred, - const panic::tensor::uint_vector& y_true){ +bool loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){ // Number of samples in a batch const panic::types::uint_t samples = y_pred.rows(); - if (samples != y_true.size()){ + // Number of classes + const panic::types::uint_t classes = y_pred.cols(); + + + if (samples == 0 || classes == 0) { return false; } + if (y_true.size() != samples) { + return false; + } - // clip data to prevent log by 0 - // Clip both sides to not drag the mean towards any value + /* + // Validate labels before the OpenMP loop. + // Maybe it has to high overhead. + for (panic::types::uint_t i = 0; i < samples; ++i) { + if (y_true[i] >= classes) { + return false; + } + } + */ + + // Clip probabilities to prevent log(0). panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max); // Vector to hold the correct confidences @@ -96,22 +110,11 @@ bool loss_categorical_crossentropy::forward( PANIC_OMP_PARALLEL_FOR_IF(samples > loss_categorical_crossentropy_omp_min_size) for (panic::types::uint_t i = 0; i < samples; ++i){ - const panic::types::uint_t idx = y_true[i]; - - //if (idx >= y_pred.cols()){ - // return false; - //} - - correct_confidences[i] = y_pred_cliped(i, idx); + correct_confidences[i] = y_pred_cliped(i, y_true[i]); } - // Calculate losses - panic::tensor::real_vector negative_log_likelihoos(samples); - - negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg); - - sample_losses = negative_log_likelihoos; - + // -log(correct confidence) + sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg); return true; } @@ -127,32 +130,33 @@ bool loss_categorical_crossentropy::forward( const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true){ - - if (y_pred.rows() != y_true.rows() || y_pred.cols() != y_true.cols()){ - return false; - } - // Number of samples in a batch const panic::types::uint_t samples = y_pred.rows(); - - // clip data to prevent log by 0 - // Clip both sides to not drag the mean towards any value - panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max); - - // Vector to hold the correct confidences - panic::tensor::real_vector correct_confidences(samples); - - - correct_confidences = panic::math::sum_rowwise(panic::math::mul(y_pred_cliped, y_true)); + // Number of classes + const panic::types::uint_t classes = y_pred.cols(); - // Calculate losses - panic::tensor::real_vector negative_log_likelihoos(samples); + if (samples == 0 || classes == 0) { + return false; + } - negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg); - sample_losses = negative_log_likelihoos; + if (y_true.rows() != samples || y_true.cols() != classes){ + return false; + } + + // Clip probabilities to prevent log(0). + panic::tensor::real_matrix y_pred_clipped = panic::math::clip(y_pred, clip_min, clip_max); + + + // Multiplication masks all predictions except the target. + const panic::tensor::real_matrix target_confidences = panic::math::mul(y_pred_clipped, y_true); + + const panic::tensor::real_vector correct_confidences = panic::math::sum_rowwise(target_confidences); + + // -log(correct confidence) + sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg); return true; } @@ -167,8 +171,7 @@ bool loss_categorical_crossentropy::forward( // Description: // Default implementation for catecorical labels. Derived classes can override it. //-------------------------------------------------------------------------------------------------------------------------- -bool loss_categorical_crossentropy::backward( - const panic::tensor::real_matrix& dvalues, +bool loss_categorical_crossentropy::backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::uint_vector& y_true){ if (dvalues.rows() == 0 || dvalues.cols() == 0 || y_true.size() != dvalues.rows()){ diff --git a/src/neural_network/model/model.cpp b/src/neural_network/model/model.cpp index e50b926..4cede4e 100644 --- a/src/neural_network/model/model.cpp +++ b/src/neural_network/model/model.cpp @@ -58,6 +58,19 @@ // Remember ti disable #include // for std::cout, std::endl +#include + +//--------------------------------------------------------------------------------------------------------------------------- +// PRIVATE CONSTANTS +//--------------------------------------------------------------------------------------------------------------------------- +/** + * @brief Minimum number of element operations before using the OpenMP-enabled loop. + * + * Small vectors and matrices are kept serial because the overhead of starting + * worker threads can be larger than the work itself. + */ +static const panic::types::uint_t model_omp_min_work = 500; +//--------------------------------------------------------------------------------------------------------------------------- //--------------------------------------------------------------------------------------------------------------------------- // IMPLEMENTATION @@ -72,23 +85,17 @@ namespace panic{ // Description: // Creates an empty model. //-------------------------------------------------------------------------------------------------------------------------- -model::model(){ - - // No layers yet. - layers = 0; - - // No loss function yet. - loss_function = 0; - - // Number of layers is zero. - layer_count = 0; - - - trainable_layers = 0; - trainable_layer_count = 0; - - optimizer_function = 0; - +model::model(): + layers(0), + layer_count(0), + trainable_layers(0), + trainable_layer_count(0), + loss_function(0), + optimizer_function(0), + softmax_classifier_output(), + use_softmax_classifier_output(false), + outputs(), + dinputs(){ } @@ -337,6 +344,14 @@ bool model::forward(const panic::tensor::real_matrix& inputs){ return true; } + // Validate the stored layer pointers before using them. + for(panic::types::uint_t i = 0; i < layer_count; ++i){ + + if (layers[i] == 0){ + return false; + } + } + // First layer receives the original model input. if (!layers[0]->forward(inputs)){ return false; @@ -359,16 +374,152 @@ bool model::forward(const panic::tensor::real_matrix& inputs){ //-------------------------------------------------------------------------------------------------------------------------- -// Function Name : panic::neural_network::model::bacward +// Function Name : panic::neural_network::model::backward // // Description: // Runs the dinputs through every layer in reverse order. +// Special case for softmax + categarical crossentropy //-------------------------------------------------------------------------------------------------------------------------- -bool model::backward(const panic::tensor::real_matrix& dvalues){ +bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true){ + + if(layers == 0 || layer_count == static_cast(0) || loss_function == 0){ + return false; + } + + if(output.rows() == static_cast(0) || output.cols() == static_cast(0) || output.rows() != y_true.size()){ + return false; + } + + + bool valid = true; + + // Validate the stored layer pointers before using them. + for(panic::types::uint_t i = 0; i < layer_count; ++i){ + + if (layers[i] == 0){ + return false; + } + } + + //---------------------------------------------------------------------- + // Optimized Softmax + categorical cross-entropy backward pass + //---------------------------------------------------------------------- + + if(use_softmax_classifier_output){ + + /* + * output already contains the probabilities produced by + * the final Softmax layer. + * + * This calculates the combined Softmax + categorical + * cross-entropy derivative. + */ + if (!softmax_classifier_output.backward(output, y_true)){ + return false; + } + + + /* + * Store the gradient in the final Softmax layer as well. + * + * We do not call Softmax::backward(), because its derivative + * has already been included in the combined calculation. + * + * This assignment is useful for consistency and debugging. + */ + layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs; + + } + else{ + + //---------------------------------------------------------------------- + // Normal loss + activation backward pass + //---------------------------------------------------------------------- + + /* + * First calculate the derivative of the loss with respect to + * the model output. + */ + if (!loss_function->backward(output, y_true)){ + return false; + } + + // The final layer receives the loss gradient. + if (!layers[layer_count - 1]->backward(loss_function->dinputs)){ + return false; + } + } + + // Every preceding layer receives dinputs from the next layer. + for (panic::types::uint_t i = layer_count - 1; i > static_cast(0); --i){ + + if (!layers[i - 1]->backward(layers[i]->dinputs)){ + return false; + } + } + + // Gradient with respect to the original model input. + dinputs = layers[0]->dinputs; return true; } + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::neural_network::model::backward +// +// Description: +// Runs the dinputs through every layer in reverse order. +// Special case for softmax + categarical crossentropy for one-hot encoding +//-------------------------------------------------------------------------------------------------------------------------- +bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true){ + + if(layers == 0 || layer_count == static_cast(0) || loss_function == 0 ){ + return false; + } + + if(output.rows() == static_cast(0) || + output.cols() == static_cast(0) || + output.rows() != y_true.rows() || + output.cols() != y_true.cols()){ + + return false; + } + + if(use_softmax_classifier_output){ + + if (!softmax_classifier_output.backward(output, y_true )){ + return false; + } + + // The combined derivative already includes Softmax. + layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs; + + } + else { + + if (!loss_function->backward(output, y_true)){ + return false; + } + + if (!layers[layer_count - 1]->backward(loss_function->dinputs)){ + return false; + } + } + + for (panic::types::uint_t i = layer_count - 1; i > static_cast(0); --i){ + if (!layers[i - 1]->backward(layers[i]->dinputs)){ + return false; + } + } + + dinputs = layers[0]->dinputs; + + return true; +} + + + //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::model::add_loss_categorical_crossentropy // @@ -385,22 +536,6 @@ bool model::add_loss_categorical_crossentropy(){ } -//-------------------------------------------------------------------------------------------------------------------------- -// Function Name : panic::neural_network::model::activation_softmax_loss_categorical_crossentropy -// -// Description: -// Sets the loss function. -//-------------------------------------------------------------------------------------------------------------------------- -bool model::activation_softmax_loss_categorical_crossentropy(){ - - delete loss_function; - - loss_function = new panic::neural_network::activation_softmax_loss_categorical_crossentropy(); - - return true; -} - - //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::model::add_optimizer_sgd // @@ -425,6 +560,53 @@ bool model::add_optimizer_sgd(const panic::types::real_t learning_rate){ } + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::neural_network::model::finalize +// +// Description: +// COnfigures the loss functions if softmax activation and categorical crossentropy loss is used. +// +//-------------------------------------------------------------------------------------------------------------------------- +bool model::finalize(){ + + // Always reset first so that calling finalize() more than + // once cannot leave an old configuration enabled. + use_softmax_classifier_output = false; + + // The model must contain at least one layer. + if (layers == 0 || layer_count == static_cast(0)){ + return false; + } + + // A loss must have been configured. + if (loss_function == 0){ + return false; + } + + // Validate every stored layer pointer. + for (panic::types::uint_t i = 0; i < layer_count; ++i){ + if (layers[i] == 0){ + return false; + } + } + + // Gets last layer + const layer* last_layer = layers[layer_count - 1]; + // Checks if last layer is activation softmax + const bool last_layer_is_softmax = last_layer->get_type() == layer_type::activation_softmax; + // Checks if loss is categorical crossentropy + const bool loss_is_categorical_crossentropy = loss_function->get_type() == loss_type::categorical_crossentropy; + // If ast layer is activation softmax and loss is categorical crossentropy it's true. + use_softmax_classifier_output = last_layer_is_softmax && loss_is_categorical_crossentropy; + // The model is still valid when the optimization cannot be used. + // It will use the normal loss + activation backward path instead. + return true; +} + + + //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::model::train // @@ -436,7 +618,7 @@ bool model::train(const panic::tensor::real_matrix& X_train, const panic::types::uint_t epochs, const panic::types::uint_t print_every){ - panic::tensor::uint_vector prediction; + panic::tensor::uint_vector predictions; panic::types::real_t accuracy; panic::tensor::uint_vector comparisons; @@ -445,31 +627,51 @@ bool model::train(const panic::tensor::real_matrix& X_train, forward(X_train); + // Calculate loss from the model's final output. if (!loss_function->calculate(outputs, y_train)){ return false; } + // There must be one output row for every target label. + if (outputs.rows() != y_train.size() || outputs.cols() == static_cast(0)){ + return false; + } + + // Find the predicted class for every sample. + if (!panic::math::argmax_rowwise(outputs, predictions)){ + return false; + } - prediction = panic::math::argmax_rowwise(loss_function->outputs); + // Compare every prediction with the correct label. + if (!panic::math::equal(predictions, y_train, comparisons)){ + return false; + } - comparisons = panic::math::equal(prediction, y_train); + // Avoid calculating the mean of an empty vector. + if (comparisons.size() == static_cast(0)){ + return false; + } + // comparisons contains only 0 and 1. + // Its mean is therefore the classification accuracy. accuracy = panic::math::mean(comparisons); if (epoch % print_every == static_cast(0)){ std::cout << "Epoch: " << epoch; - std::cout << " loss: " << loss_function->data_loss; + std::cout << " data loss: " << loss_function->data_loss; std::cout << " acc: " << accuracy << std::endl; } - loss_function->backward(loss_function->outputs, y_train); + if (!backward(outputs, y_train)){ + return false; + } - //backward(); //optimize(); } + return true; } @@ -488,32 +690,56 @@ bool model::train(const panic::tensor::real_matrix& X_train, //-------------------------------------------------------------------------------------------------------------------------- void model::clear(){ + // Delete every layer object owned by the model. if (layers != 0){ - // Delete each actual layer object. - for (panic::types::uint_t i = 0; i < layer_count; ++i){ + for ( + panic::types::uint_t i = 0; + i < layer_count; + ++i + ){ delete layers[i]; layers[i] = 0; } - // Delete the array that stored the layer pointers. + // Delete the array containing the layer pointers. delete[] layers; } - - // Reset to empty state. layers = 0; - loss_function = 0; layer_count = 0; + + // trainable_layers only contains references to objects + // that were already deleted through layers. + // + // Therefore, do not delete trainable_layers[i]. delete[] trainable_layers; + trainable_layers = 0; trainable_layer_count = 0; + + // Delete the owned loss object. + delete loss_function; + loss_function = 0; + + + // Delete the owned optimizer object. delete optimizer_function; + optimizer_function = 0; + + // softmax_classifier_output is not a pointer. + // Do not delete it. Reset only its stored result. + softmax_classifier_output.dinputs.resize(0, 0); + use_softmax_classifier_output = false; + + + // Reset model results. outputs.resize(0, 0); -} + dinputs.resize(0, 0); + } - } // namespace neural_network +} // namespace neural_network } // namespace panic \ No newline at end of file diff --git a/src/tensor/generators/one_hot.cpp b/src/tensor/generators/one_hot.cpp index 839dc93..fca7162 100644 --- a/src/tensor/generators/one_hot.cpp +++ b/src/tensor/generators/one_hot.cpp @@ -48,7 +48,7 @@ * Small vectors and matrices are kept serial because the overhead of starting * worker threads can be larger than the work itself. */ -static const panic::types::uint_t one_hot_omp_min_work = 500; +static const panic::types::uint_t one_hot_omp_min_work = 250; //--------------------------------------------------------------------------------------------------------------------------- // INPLEMENTATION //--------------------------------------------------------------------------------------------------------------------------- @@ -61,33 +61,57 @@ namespace panic { // Function Name : panic::tensor::one_hot // // Description: -// Creates a one_hot matrix +// Creates a one-hot encoded matrix. +// +// The resulting matrix has: +// one row per sample +// one column per class //-------------------------------------------------------------------------------------------------------------------------- template -bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector & a, panic::tensor::matrix& B){ +bool one_hot(const panic::types::uint_t classes, const panic::tensor::uint_vector& a, panic::tensor::matrix& B){ - if (!B.resize(size,size)){ + const panic::types::uint_t samples = a.size(); + const panic::types::uint_t work = samples * classes; + + if (samples == static_cast(0) || classes == static_cast(0)){ + return false; + } + + + bool valid = true; + + // Validate labels + // valid remains true only if every divisor is nonzero. + PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(a.size() > one_hot_omp_min_work, &&, valid) + for (panic::types::uint_t i = 0; i < a.size(); ++i){ + const bool nonzero = a[i] <= classes; + valid = valid && nonzero; + } + + if (!valid){ return false; } - if (size != a.size()){ - return false; - } - - PANIC_OMP_PARALLEL_FOR_IF(size*size > one_hot_omp_min_work) - for (panic::types::uint_t i = 0; i < size; ++i){ - for (panic::types::uint_t j = 0; j < size; ++j){ - if (a[i] == j){ - B(i,j) = T{1}; - } - else{ - B(i,j) = T{0}; - } - } - } - return true; + // One row per sample and one column per class. + if (!B.resize(samples, classes)){ + return false; + } + + PANIC_OMP_PARALLEL_FOR_IF(work > one_hot_omp_min_work) + for(panic::types::uint_t i = 0; i < samples; ++i){ + for(panic::types::uint_t j = 0; j < classes; ++j){ + if (a[i] == j){ + B(i, j) = T{1}; + } + else { + B(i, j) = T{0}; + } + } + } + + return true; } //-------------------------------------------------------------------------------------------------------------------------- // EXPLICIT TEMPLATE INSTANTIATION @@ -95,17 +119,22 @@ bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector & // The implementation is in this .cpp file. // Build the overload for the official PANIC numeric types. //-------------------------------------------------------------------------------------------------------------------------- -template bool one_hot(const panic::types::uint_t size, - const panic::tensor::uint_vector& a, - panic::tensor::matrix& B +template bool one_hot( + const panic::types::uint_t classes, + const panic::tensor::uint_vector& a, + panic::tensor::matrix& B ); -template bool one_hot(const panic::types::uint_t size, - const panic::tensor::uint_vector& a, - panic::tensor::matrix& B + +template bool one_hot( + const panic::types::uint_t classes, + const panic::tensor::uint_vector& a, + panic::tensor::matrix& B ); -template bool one_hot(const panic::types::uint_t size, - const panic::tensor::uint_vector& a, - panic::tensor::matrix& B + +template bool one_hot( + const panic::types::uint_t classes, + const panic::tensor::uint_vector& a, + panic::tensor::matrix& B );