Softmax + Categorical Crossentropy

I fixed the activation+loss function, you can't select it directly, it automaticly uses it if it can. I also fixed one-hot generator. Still haven't tested the activation + loss nor any other backward function. I'll do that when I get to the optimizers which is next.
This commit is contained in:
2026-08-06 09:53:03 +02:00
parent fecb4c70b1
commit 642bba1198
18 changed files with 627 additions and 378 deletions
+11 -31
View File
@@ -134,18 +134,13 @@
// Expands to:
// #pragma omp parallel for if(condition)
// num_threads(PANIC_OMP_NUM_THREADS)
// schedule(static)
// reduction(operation:variable)
//
// Runs the loop in parallel only when condition is true.
// Each thread receives a private copy of variable.
// Afterward, OpenMP combines those copies using operation.
// schedule(static) assigns fixed groups of iterations to each thread.
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
num_threads(PANIC_OMP_NUM_THREADS) \
schedule(static) \
reduction(operation:variable)))
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable) num_threads(PANIC_OMP_NUM_THREADS)))
#else
@@ -165,19 +160,13 @@
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition)))
// Expands to:
// #pragma omp parallel for if(condition)
// num_threads(PANIC_OMP_NUM_THREADS)
// schedule(static)
// reduction(operation:variable)
// #pragma omp parallel for if(condition) reduction(operation:variable)
//
// Runs the loop in parallel only when condition is true.
// Each thread receives a private copy of variable.
// Afterward, OpenMP combines those copies using operation.
// schedule(static) assigns fixed groups of iterations to each thread.
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
schedule(static) \
reduction(operation:variable)))
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable)))
#endif
@@ -199,25 +188,16 @@
// }
//
// That means the same code still works on microcontrollers and non-OpenMP builds.
// Without OpenMP, the normal for-loop remains.
#define PANIC_OMP_PARALLEL_FOR
// Check that condition is syntactically valid, but do not evaluate it.
#define PANIC_OMP_PARALLEL_FOR_IF(condition) static_cast<void>(sizeof(condition));
// The operation and variable are only needed by the OpenMP pragma.
// The serial loop itself performs the calculation normally.
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) static_cast<void>(sizeof(condition));
#endif
//---------------------------------------------------------------------------------------------------------------------------
// TYPE DESCRIPTION
//---------------------------------------------------------------------------------------------------------------------------
// None.
//---------------------------------------------------------------------------------------------------------------------------
// VARIABLE DESCRIPTION
//---------------------------------------------------------------------------------------------------------------------------
// None.
//---------------------------------------------------------------------------------------------------------------------------
// FUNCTION PROTOTYPE
//---------------------------------------------------------------------------------------------------------------------------
// None.
@@ -70,6 +70,16 @@ struct activation_relu : public layer{
*/
~activation_relu() = default;
/**
* @brief get_type function for layer
*
* @returns the layer type
*
*/
layer_type get_type() const override {
return layer_type::activation_relu;
}
/**
* @brief Forward function for layer
*
@@ -70,6 +70,17 @@ struct activation_softmax : public layer{
*/
~activation_softmax() = default;
/**
* @brief get_type function for layer
*
* @returns the layer type
*
*/
layer_type get_type() const override {
return layer_type::activation_softmax;
}
/**
* @brief Forward function for layer
*
@@ -54,47 +54,9 @@ namespace panic{
*
* The struct is used in PANIC nural_network library.
*/
struct activation_softmax_loss_categorical_crossentropy: loss{
activation_softmax activation;
loss_categorical_crossentropy loss;
/**
* @brief Empthy constructor
*
*/
activation_softmax_loss_categorical_crossentropy();
/**
* @brief Default de-constructor
*
*/
~activation_softmax_loss_categorical_crossentropy() = default;
/**
* @brief forward function to calculate losses
*
* @param y_pred Matrix of model predection.
* @param y_true Vector of true label of data.
*
*/
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
/**
* @brief forward function to calculate losses
*
* @param y_pred Matrix of model predection.
* @param y_true Vector of true label of data.
*
* @Note Overloaded if one-shot endcoded
* is used.
*/
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
struct activation_softmax_loss_categorical_crossentropy{
panic::tensor::real_matrix dinputs;
/**
* @brief backward function to calculate from losses
+19 -1
View File
@@ -44,6 +44,14 @@ namespace panic{
enum struct layer_type {
unknown,
layer_dense,
activation_relu,
activation_softmax
};
/**
* @brief Base layer for the rest of the neural network library to use
*
@@ -85,6 +93,16 @@ struct layer{
virtual ~layer() = default;
/**
* @brief Virtual layer_type function for derivative layers
*
* @Note This returns the type of layer it is
* unknown be default
*/
virtual layer_type get_type() const {
return layer_type::unknown;
}
/**
* @brief Virtual forward function for derivative layers
*
@@ -103,7 +121,7 @@ struct layer{
* @Note It's equal to 0 because it make the derivative
* object NEEDS to have these function to work.
*/
virtual bool backward(const panic::tensor::real_matrix& dinputs) = 0;
virtual bool backward(const panic::tensor::real_matrix& dvalues) = 0;
};
+10 -24
View File
@@ -58,30 +58,6 @@ namespace panic{
*/
struct layer_dense : trainable_layer{
/**
* @brief Emphty matrix to store input data
*
*/
panic::tensor::real_matrix ipnuts;
/**
* @brief Emphty weight matrix to store layer weights
*
* Weight shape:
* input_size x neuron_count
*/
panic::tensor::real_matrix weights;
panic::tensor::real_matrix dweights;
/**
* @brief Emphty bias vector to store layer bias
*
* Bias shape:
* 1 x neuron_count
*/
panic::tensor::real_vector biases;
panic::tensor::real_vector dbiases;
/**
* @brief Empthy constructor
*
@@ -103,6 +79,16 @@ struct layer_dense : trainable_layer{
*/
~layer_dense() = default;
/**
* @brief get_type function for layer
*
* @returns the layer type
*
*/
layer_type get_type() const override {
return layer_type::layer_dense;
}
/**
* @brief Forward function for layer
*
+22 -11
View File
@@ -44,6 +44,11 @@ namespace panic{
namespace neural_network{
enum struct loss_type {
unknown,
categorical_crossentropy
};
/**
* @brief Base loss for the rest of the neural network library to use
@@ -58,7 +63,7 @@ namespace panic{
struct loss{
/**
* @brief Emphty vector to store sample losses
* @brief Emphty vector to store sample losses for each sample in the batch.
*
*/
panic::tensor::real_vector sample_losses;
@@ -66,19 +71,15 @@ struct loss{
/**
* @brief Mean loss over the entire batch.
*/
panic::types::real_t data_loss;
panic::types::real_t data_loss = 0;
/**
* @brief Matrix for backwards pass
* @brief Gradient with respect to the loss input.
*
* This will be used later during the backward pass.
*/
panic::tensor::real_matrix dinputs;
/**
* @brief Matrix for output of loss function
*/
panic::tensor::real_matrix outputs;
/**
* @brief Default de-constructor
*
@@ -86,6 +87,16 @@ struct loss{
virtual ~loss() = default;
/**
* @brief Virtual loss_type function for derivative losses
*
* @Note This returns the type of loss it is
* unknown be default
*/
virtual loss_type get_type() const {
return loss_type::unknown;
}
/**
* @brief Virtual forward function for derivative loss functions
*
@@ -146,7 +157,7 @@ struct loss{
* @param y_true Vector of true label of data.
*
*/
virtual bool calculate(
bool calculate(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::uint_vector& y_true);
@@ -157,7 +168,7 @@ struct loss{
* @param y_true Matrix of true label of data.
*
*/
virtual bool calculate(
bool calculate(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::real_matrix& y_true);
@@ -53,10 +53,17 @@ namespace panic{
*
* The struct is used for PANIC neural_network library.
*/
struct loss_categorical_crossentropy: loss{
struct loss_categorical_crossentropy: public loss{
/**
* @brief get_type function for loss
*
* @returns the loss type
*
*/
loss_type get_type() const override {
return loss_type::categorical_crossentropy;
}
/**
* @brief forward function to calculate losses
@@ -65,9 +72,7 @@ struct loss_categorical_crossentropy: loss{
* @param y_true Vector of true label of data.
*
*/
bool forward(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::uint_vector& y_true)override;
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
/**
* @brief forward function to calculate losses
@@ -78,9 +83,7 @@ struct loss_categorical_crossentropy: loss{
* @Note Overloaded if one-shot endcoded
* is used.
*/
bool forward(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::real_matrix& y_true)override;
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
/**
@@ -90,9 +93,7 @@ struct loss_categorical_crossentropy: loss{
* @param y_true Vector of true label of data.
*
*/
bool backward(
const panic::tensor::real_matrix& dvalues,
const panic::tensor::uint_vector& y_true) override;
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::uint_vector& y_true) override;
/**
* @brief backward function to calculate from losses
@@ -103,9 +104,7 @@ struct loss_categorical_crossentropy: loss{
* @Note Overloaded if one-shot endcoded
* is used.
*/
bool backward(
const panic::tensor::real_matrix& dvalues,
const panic::tensor::real_matrix& y_true) override;
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::real_matrix& y_true) override;
};
+82 -42
View File
@@ -44,6 +44,7 @@
#include <neural_network/layer/layer_dense.hpp> // fully connected dense layer
#include <neural_network/loss/loss.hpp>
#include <neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp>
#include <neural_network/optimizers/optimizer.hpp>
#include <neural_network/optimizers/optimizer_sgd.hpp>
@@ -71,44 +72,77 @@ namespace panic{
struct model{
/**
* @brief a pointer to a pointer of layers
* @brief Array of pointers to all model layers.
*
* An example:
* layers[0] points to a layer_dense
* layers[1] points to an actication function
* layers[2] points to another layer_dense
*
* @note The model owns these layers and deletes them in clear().
*
* Example:
* layers[0] points to a layer_dense
* layers[1] points to an activation_ReLU
* layers[2] points to another layer_dense
*
* @note The model owns every object referenced by this array.
* clear() deletes each layer and then deletes the array.
*/
layer** layers;
/**
* @brief Number of layers currently stored in layers.
*/
panic::types::uint_t layer_count;
/**
* @brief Array of pointers to the trainable layers.
*
* @note These pointers refer to objects already owned through layers.
* Do not delete the individual objects through this array.
* Only the pointer array itself is owned separately.
*/
trainable_layer** trainable_layers;
/**
* @brief Number of trainable-layer pointers.
*/
panic::types::uint_t trainable_layer_count;
/**
* @brief Configured loss function.
*
* @note The model owns this object and deletes it in clear().
*/
loss* loss_function;
/**
* @brief Configured optimizer.
*
* @note The model owns this object and deletes it in clear().
*/
optimizer* optimizer_function;
/**
* @brief a pointer the loss function
* @brief Optimized backward helper for the combination of
* Softmax and categorical cross-entropy.
*
*
* @note The model owns these layers and deletes them in clear().
*
* This is a normal member object, not a dynamically allocated object.
*/
loss* loss_function;
activation_softmax_loss_categorical_crossentropy softmax_classifier_output;
// Number of layers currently stored in the model.
/**
* @brief Stores the number of layers
*
* @brief Whether the optimized Softmax + categorical
* cross-entropy backward path should be used.
*/
panic::types::uint_t layer_count;
bool use_softmax_classifier_output;
// model output (may be deleted and also used for debug)
/**
* @brief Output of the final model layer.
*/
panic::tensor::real_matrix outputs;
// model dinputs (may be deleted and also used for debug)
/**
* @brief Gradient with respect to the model input.
*/
panic::tensor::real_matrix dinputs;
/**
@@ -231,16 +265,32 @@ struct model{
*
* Computes:
* @code
* model.bacward(dvalues_data_matrix)
* model.backward(model_output, y_true_values)
* @endcode
*
* @param dvalues diput data.
* @param output Model output.
* @param y_true True values for data.
*
* @return true looped over every layer.
*
*
*/
bool backward(const panic::tensor::real_matrix& dvalues);
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true);
/**
* @brief Loops over all layers backward function
*
* Computes:
* @code
* model.backward(model_output, y_true_values)
* @endcode
*
* @param output Model output.
* @param y_true True values for data.
*
* @return true looped over every layer.
*
*/
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true);
/**
@@ -259,24 +309,6 @@ struct model{
bool add_loss_categorical_crossentropy();
/**
* @brief Adds activation softmax AND loss for categorical crossentropy.
*
* Computes:
* @code
* model.activation_softmax_loss_categorical_crossentropy();
* @endcode
*
*
* @return true if activation and loss is added
*
* @note This function is convenient, but it allocates a new layer.
*/
bool activation_softmax_loss_categorical_crossentropy();
/**
* @brief Adds optimizer_sgd to the model
*
@@ -295,7 +327,15 @@ struct model{
/**
* @brief Finalizes the model configuration.
*
* Detects whether the model can use the optimized
* Softmax + categorical-cross-entropy backward pass.
*
* @return true if the model configuration is valid.
*/
bool finalize();
/**
* @brief Trains the model with input data
+28 -19
View File
@@ -88,15 +88,6 @@
// #define TEST_FALG 1
//---------------------------------------------------------------------------------------------------------------------------
// Function Name : benchmark_omp_min_work
//
@@ -725,28 +716,46 @@ int main(void) {
panic::neural_network::model mymodel;
// Create Dense layer with 2 input features and 3 output values
mymodel.add_layer_dense(2,3);
if (!mymodel.add_layer_dense(2,3)){
return false;
}
// Create an activation ReLU layer
mymodel.add_activation_relu();
if (!mymodel.add_activation_relu()){
return false;
}
// Create a second dense layer with 3 inputs and 3 outputs
mymodel.add_layer_dense(3, 3);
if (!mymodel.add_layer_dense(3, 3)){
return false;
}
// Create activation softmax layer
mymodel.add_activation_softmax();
if (!mymodel.add_activation_softmax()){
return false;
}
mymodel.add_loss_categorical_crossentropy();
//mymodel.activation_softmax_loss_categorical_crossentropy();
if (!mymodel.add_loss_categorical_crossentropy()){
return false;
}
if (!mymodel.finalize()){
return false;
}
mymodel.add_optimizer_sgd();
//mymodel.add_optimizer_sgd();
panic::types::uint_t epochs = 10;
panic::types::uint_t print_every = 1;
mymodel.train(X, y, epochs, print_every);
if (!mymodel.train(X, y, epochs, print_every)){
std::cout << "Training failed" << std::endl;
return false;
}
return 0;
+17 -1
View File
@@ -72,6 +72,10 @@ namespace panic {
template <typename T>
panic::types::real_t mean(const panic::tensor::vector<T>& a){
if (a.size() == static_cast<panic::types::uint_t>(0)){
return static_cast<panic::types::real_t>(0);
}
const panic::types::real_t sum = static_cast<panic::types::real_t>(panic::math::sum(a));
return sum / static_cast<panic::types::real_t>(a.size());
@@ -104,6 +108,10 @@ panic::types::real_t mean(const panic::tensor::matrix<T>& A){
panic::types::uint_t cols = A.cols();
panic::types::uint_t work = rows*cols;
if (work == static_cast<panic::types::uint_t>(0)){
return static_cast<panic::types::real_t>(0);
}
panic::types::real_t sum = static_cast<panic::types::real_t> (panic::math::sum(A));
return sum / static_cast<panic::types::real_t>(work);
@@ -138,6 +146,10 @@ bool mean_rowwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
panic::types::uint_t cols = A.cols();
panic::types::uint_t work = rows*cols;
if (work == static_cast<panic::types::uint_t>(0)){
return static_cast<panic::types::real_t>(0);
}
if ( !b.resize(rows) ){
return false;
}
@@ -245,6 +257,10 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
panic::types::uint_t cols = A.cols();
panic::types::uint_t work = rows*cols;
if (work == static_cast<panic::types::uint_t>(0)){
return static_cast<panic::types::real_t>(0);
}
if ( !b.resize(cols) ){
return false;
}
@@ -256,7 +272,7 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
// Each thread handles separate cols and writes to a separate b[i].
PANIC_OMP_PARALLEL_FOR_IF(work > mean_omp_min_work)
for (panic::types::uint_t i = 0; i < cols; ++i){
b[i] /= cols;
b[i] /= rows;
}
return true;
@@ -52,6 +52,8 @@
* worker threads can be larger than the work itself.
*/
static const panic::types::uint_t activation_ReLU_omp_min_size = 500;
static const panic::types::real_t almost_zero = static_cast<panic::types::real_t> (1e-7);
//---------------------------------------------------------------------------------------------------------------------------
// INPLEMENTATION
//---------------------------------------------------------------------------------------------------------------------------
@@ -80,7 +82,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){
inputs = input_data;
panic::math::clip_lower(inputs, 0.0f, outputs);
panic::math::clip_lower(inputs, almost_zero, outputs);
return true;
@@ -97,8 +99,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){
bool activation_relu::backward(const panic::tensor::real_matrix& dvalues){
// Zero gradients where input values were negative
panic::types::real_t zero = 0;
dinputs = panic::math::clip_lower(dvalues, zero);
dinputs = panic::math::clip_lower(dvalues, almost_zero);
return true;
@@ -52,7 +52,7 @@
* Small vectors and matrices are kept serial because the overhead of starting
* worker threads can be larger than the work itself.
*/
static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 500;
static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 250;
//---------------------------------------------------------------------------------------------------------------------------
// INPLEMENTATION
//---------------------------------------------------------------------------------------------------------------------------
@@ -60,65 +60,6 @@ static const panic::types::uint_t activation_softmax_loss_categorical_crossentro
namespace panic{
namespace neural_network{
//--------------------------------------------------------------------------------------------------------------------------
// Constructor Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy
//
// Description:
// Creates an empty layer.
//--------------------------------------------------------------------------------------------------------------------------
activation_softmax_loss_categorical_crossentropy::activation_softmax_loss_categorical_crossentropy() {
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward
//
// Description:
// Calculated the forward pass
//--------------------------------------------------------------------------------------------------------------------------
bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){
// Output layers activation function
if (!activation.forward(y_pred)){
return false;
}
// Set the output
outputs = activation.outputs;
// calculate the loss value.
if (!loss.calculate(outputs, y_true)){
return false;
}
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward
//
// Description:
// Calculated the forward pass
//--------------------------------------------------------------------------------------------------------------------------
bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true){
// Output layers activation function
if (!activation.forward(y_pred)){
return false;
}
// Set the output
outputs = activation.outputs;
// calculate the loss value.
if (!loss.calculate(outputs, y_true)){
return false;
}
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.backward
//
+1 -1
View File
@@ -86,7 +86,7 @@ layer_dense::layer_dense() {
layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons) {
weights.resize(input_size, neurons);
panic::random::uniform(weights);
panic::math::mul(weights, 0.01f, weights);
panic::math::mul(weights, static_cast<panic::types::real_t>(0.01), weights);
biases.resize(neurons);
biases.fill(0);
+17 -10
View File
@@ -69,18 +69,21 @@ bool loss::calculate(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::uint_vector& y_true){
// Calculate sample losses
data_loss = static_cast<panic::types::real_t>(0);
// Calls the derived loss function's forward().
if (!forward(y_pred, y_true)){
return false;
}
// forward() must produce one loss per sample.
if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) {
return false;
}
// Calculate mean loss
data_loss = panic::math::mean(sample_losses);
// This is so the model can use every loss functions
// It's a problem to get the output when useing an activation+loss function
outputs = y_pred;
return true;
}
@@ -94,17 +97,21 @@ bool loss::calculate(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::real_matrix& y_true){
data_loss = static_cast<panic::types::real_t>(0);
// Calls the derived loss function's forward().
if (!forward(y_pred, y_true)){
return false;
}
// forward() must produce one loss per sample.
if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) {
return false;
}
// Calculate mean loss
data_loss = panic::math::mean(sample_losses);
// This is so the model can use every loss functions
// It's a problem to get the output when useing an activation+loss function
outputs = y_pred;
return true;
}
@@ -57,7 +57,7 @@
* Small vectors and matrices are kept serial because the overhead of starting
* worker threads can be larger than the work itself.
*/
static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 500;
static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 250;
static const panic::types::real_t clip_min = 1e-7;
static const panic::types::real_t clip_max = 1 - 1e-7;
@@ -75,20 +75,34 @@ namespace panic{
// Description:
// Default implementation for catecorical labels. Derived classes can override it.
//--------------------------------------------------------------------------------------------------------------------------
bool loss_categorical_crossentropy::forward(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::uint_vector& y_true){
bool loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){
// Number of samples in a batch
const panic::types::uint_t samples = y_pred.rows();
if (samples != y_true.size()){
// Number of classes
const panic::types::uint_t classes = y_pred.cols();
if (samples == 0 || classes == 0) {
return false;
}
if (y_true.size() != samples) {
return false;
}
// clip data to prevent log by 0
// Clip both sides to not drag the mean towards any value
/*
// Validate labels before the OpenMP loop.
// Maybe it has to high overhead.
for (panic::types::uint_t i = 0; i < samples; ++i) {
if (y_true[i] >= classes) {
return false;
}
}
*/
// Clip probabilities to prevent log(0).
panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max);
// Vector to hold the correct confidences
@@ -96,22 +110,11 @@ bool loss_categorical_crossentropy::forward(
PANIC_OMP_PARALLEL_FOR_IF(samples > loss_categorical_crossentropy_omp_min_size)
for (panic::types::uint_t i = 0; i < samples; ++i){
const panic::types::uint_t idx = y_true[i];
//if (idx >= y_pred.cols()){
// return false;
//}
correct_confidences[i] = y_pred_cliped(i, idx);
correct_confidences[i] = y_pred_cliped(i, y_true[i]);
}
// Calculate losses
panic::tensor::real_vector negative_log_likelihoos(samples);
negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg);
sample_losses = negative_log_likelihoos;
// -log(correct confidence)
sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg);
return true;
}
@@ -127,32 +130,33 @@ bool loss_categorical_crossentropy::forward(
const panic::tensor::real_matrix& y_pred,
const panic::tensor::real_matrix& y_true){
if (y_pred.rows() != y_true.rows() || y_pred.cols() != y_true.cols()){
return false;
}
// Number of samples in a batch
const panic::types::uint_t samples = y_pred.rows();
// clip data to prevent log by 0
// Clip both sides to not drag the mean towards any value
panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max);
// Vector to hold the correct confidences
panic::tensor::real_vector correct_confidences(samples);
correct_confidences = panic::math::sum_rowwise(panic::math::mul(y_pred_cliped, y_true));
// Number of classes
const panic::types::uint_t classes = y_pred.cols();
// Calculate losses
panic::tensor::real_vector negative_log_likelihoos(samples);
if (samples == 0 || classes == 0) {
return false;
}
negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg);
sample_losses = negative_log_likelihoos;
if (y_true.rows() != samples || y_true.cols() != classes){
return false;
}
// Clip probabilities to prevent log(0).
panic::tensor::real_matrix y_pred_clipped = panic::math::clip(y_pred, clip_min, clip_max);
// Multiplication masks all predictions except the target.
const panic::tensor::real_matrix target_confidences = panic::math::mul(y_pred_clipped, y_true);
const panic::tensor::real_vector correct_confidences = panic::math::sum_rowwise(target_confidences);
// -log(correct confidence)
sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg);
return true;
}
@@ -167,8 +171,7 @@ bool loss_categorical_crossentropy::forward(
// Description:
// Default implementation for catecorical labels. Derived classes can override it.
//--------------------------------------------------------------------------------------------------------------------------
bool loss_categorical_crossentropy::backward(
const panic::tensor::real_matrix& dvalues,
bool loss_categorical_crossentropy::backward(const panic::tensor::real_matrix& dvalues,
const panic::tensor::uint_vector& y_true){
if (dvalues.rows() == 0 || dvalues.cols() == 0 || y_true.size() != dvalues.rows()){
+275 -49
View File
@@ -58,6 +58,19 @@
// Remember ti disable
#include <iostream> // for std::cout, std::endl
#include <io/print_tensor.hpp>
//---------------------------------------------------------------------------------------------------------------------------
// PRIVATE CONSTANTS
//---------------------------------------------------------------------------------------------------------------------------
/**
* @brief Minimum number of element operations before using the OpenMP-enabled loop.
*
* Small vectors and matrices are kept serial because the overhead of starting
* worker threads can be larger than the work itself.
*/
static const panic::types::uint_t model_omp_min_work = 500;
//---------------------------------------------------------------------------------------------------------------------------
//---------------------------------------------------------------------------------------------------------------------------
// IMPLEMENTATION
@@ -72,23 +85,17 @@ namespace panic{
// Description:
// Creates an empty model.
//--------------------------------------------------------------------------------------------------------------------------
model::model(){
// No layers yet.
layers = 0;
// No loss function yet.
loss_function = 0;
// Number of layers is zero.
layer_count = 0;
trainable_layers = 0;
trainable_layer_count = 0;
optimizer_function = 0;
model::model():
layers(0),
layer_count(0),
trainable_layers(0),
trainable_layer_count(0),
loss_function(0),
optimizer_function(0),
softmax_classifier_output(),
use_softmax_classifier_output(false),
outputs(),
dinputs(){
}
@@ -337,6 +344,14 @@ bool model::forward(const panic::tensor::real_matrix& inputs){
return true;
}
// Validate the stored layer pointers before using them.
for(panic::types::uint_t i = 0; i < layer_count; ++i){
if (layers[i] == 0){
return false;
}
}
// First layer receives the original model input.
if (!layers[0]->forward(inputs)){
return false;
@@ -359,16 +374,152 @@ bool model::forward(const panic::tensor::real_matrix& inputs){
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::bacward
// Function Name : panic::neural_network::model::backward
//
// Description:
// Runs the dinputs through every layer in reverse order.
// Special case for softmax + categarical crossentropy
//--------------------------------------------------------------------------------------------------------------------------
bool model::backward(const panic::tensor::real_matrix& dvalues){
bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true){
if(layers == 0 || layer_count == static_cast<panic::types::uint_t>(0) || loss_function == 0){
return false;
}
if(output.rows() == static_cast<panic::types::uint_t>(0) || output.cols() == static_cast<panic::types::uint_t>(0) || output.rows() != y_true.size()){
return false;
}
bool valid = true;
// Validate the stored layer pointers before using them.
for(panic::types::uint_t i = 0; i < layer_count; ++i){
if (layers[i] == 0){
return false;
}
}
//----------------------------------------------------------------------
// Optimized Softmax + categorical cross-entropy backward pass
//----------------------------------------------------------------------
if(use_softmax_classifier_output){
/*
* output already contains the probabilities produced by
* the final Softmax layer.
*
* This calculates the combined Softmax + categorical
* cross-entropy derivative.
*/
if (!softmax_classifier_output.backward(output, y_true)){
return false;
}
/*
* Store the gradient in the final Softmax layer as well.
*
* We do not call Softmax::backward(), because its derivative
* has already been included in the combined calculation.
*
* This assignment is useful for consistency and debugging.
*/
layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs;
}
else{
//----------------------------------------------------------------------
// Normal loss + activation backward pass
//----------------------------------------------------------------------
/*
* First calculate the derivative of the loss with respect to
* the model output.
*/
if (!loss_function->backward(output, y_true)){
return false;
}
// The final layer receives the loss gradient.
if (!layers[layer_count - 1]->backward(loss_function->dinputs)){
return false;
}
}
// Every preceding layer receives dinputs from the next layer.
for (panic::types::uint_t i = layer_count - 1; i > static_cast<panic::types::uint_t>(0); --i){
if (!layers[i - 1]->backward(layers[i]->dinputs)){
return false;
}
}
// Gradient with respect to the original model input.
dinputs = layers[0]->dinputs;
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::backward
//
// Description:
// Runs the dinputs through every layer in reverse order.
// Special case for softmax + categarical crossentropy for one-hot encoding
//--------------------------------------------------------------------------------------------------------------------------
bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true){
if(layers == 0 || layer_count == static_cast<panic::types::uint_t>(0) || loss_function == 0 ){
return false;
}
if(output.rows() == static_cast<panic::types::uint_t>(0) ||
output.cols() == static_cast<panic::types::uint_t>(0) ||
output.rows() != y_true.rows() ||
output.cols() != y_true.cols()){
return false;
}
if(use_softmax_classifier_output){
if (!softmax_classifier_output.backward(output, y_true )){
return false;
}
// The combined derivative already includes Softmax.
layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs;
}
else {
if (!loss_function->backward(output, y_true)){
return false;
}
if (!layers[layer_count - 1]->backward(loss_function->dinputs)){
return false;
}
}
for (panic::types::uint_t i = layer_count - 1; i > static_cast<panic::types::uint_t>(0); --i){
if (!layers[i - 1]->backward(layers[i]->dinputs)){
return false;
}
}
dinputs = layers[0]->dinputs;
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::add_loss_categorical_crossentropy
//
@@ -385,22 +536,6 @@ bool model::add_loss_categorical_crossentropy(){
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::activation_softmax_loss_categorical_crossentropy
//
// Description:
// Sets the loss function.
//--------------------------------------------------------------------------------------------------------------------------
bool model::activation_softmax_loss_categorical_crossentropy(){
delete loss_function;
loss_function = new panic::neural_network::activation_softmax_loss_categorical_crossentropy();
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::add_optimizer_sgd
//
@@ -425,6 +560,53 @@ bool model::add_optimizer_sgd(const panic::types::real_t learning_rate){
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::finalize
//
// Description:
// COnfigures the loss functions if softmax activation and categorical crossentropy loss is used.
//
//--------------------------------------------------------------------------------------------------------------------------
bool model::finalize(){
// Always reset first so that calling finalize() more than
// once cannot leave an old configuration enabled.
use_softmax_classifier_output = false;
// The model must contain at least one layer.
if (layers == 0 || layer_count == static_cast<panic::types::uint_t>(0)){
return false;
}
// A loss must have been configured.
if (loss_function == 0){
return false;
}
// Validate every stored layer pointer.
for (panic::types::uint_t i = 0; i < layer_count; ++i){
if (layers[i] == 0){
return false;
}
}
// Gets last layer
const layer* last_layer = layers[layer_count - 1];
// Checks if last layer is activation softmax
const bool last_layer_is_softmax = last_layer->get_type() == layer_type::activation_softmax;
// Checks if loss is categorical crossentropy
const bool loss_is_categorical_crossentropy = loss_function->get_type() == loss_type::categorical_crossentropy;
// If ast layer is activation softmax and loss is categorical crossentropy it's true.
use_softmax_classifier_output = last_layer_is_softmax && loss_is_categorical_crossentropy;
// The model is still valid when the optimization cannot be used.
// It will use the normal loss + activation backward path instead.
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// Function Name : panic::neural_network::model::train
//
@@ -436,7 +618,7 @@ bool model::train(const panic::tensor::real_matrix& X_train,
const panic::types::uint_t epochs,
const panic::types::uint_t print_every){
panic::tensor::uint_vector prediction;
panic::tensor::uint_vector predictions;
panic::types::real_t accuracy;
panic::tensor::uint_vector comparisons;
@@ -445,31 +627,51 @@ bool model::train(const panic::tensor::real_matrix& X_train,
forward(X_train);
// Calculate loss from the model's final output.
if (!loss_function->calculate(outputs, y_train)){
return false;
}
// There must be one output row for every target label.
if (outputs.rows() != y_train.size() || outputs.cols() == static_cast<panic::types::uint_t>(0)){
return false;
}
// Find the predicted class for every sample.
if (!panic::math::argmax_rowwise(outputs, predictions)){
return false;
}
prediction = panic::math::argmax_rowwise(loss_function->outputs);
// Compare every prediction with the correct label.
if (!panic::math::equal(predictions, y_train, comparisons)){
return false;
}
comparisons = panic::math::equal(prediction, y_train);
// Avoid calculating the mean of an empty vector.
if (comparisons.size() == static_cast<panic::types::uint_t>(0)){
return false;
}
// comparisons contains only 0 and 1.
// Its mean is therefore the classification accuracy.
accuracy = panic::math::mean(comparisons);
if (epoch % print_every == static_cast<panic::types::uint_t>(0)){
std::cout << "Epoch: " << epoch;
std::cout << " loss: " << loss_function->data_loss;
std::cout << " data loss: " << loss_function->data_loss;
std::cout << " acc: " << accuracy << std::endl;
}
loss_function->backward(loss_function->outputs, y_train);
if (!backward(outputs, y_train)){
return false;
}
//backward();
//optimize();
}
return true;
}
@@ -488,32 +690,56 @@ bool model::train(const panic::tensor::real_matrix& X_train,
//--------------------------------------------------------------------------------------------------------------------------
void model::clear(){
// Delete every layer object owned by the model.
if (layers != 0){
// Delete each actual layer object.
for (panic::types::uint_t i = 0; i < layer_count; ++i){
for (
panic::types::uint_t i = 0;
i < layer_count;
++i
){
delete layers[i];
layers[i] = 0;
}
// Delete the array that stored the layer pointers.
// Delete the array containing the layer pointers.
delete[] layers;
}
// Reset to empty state.
layers = 0;
loss_function = 0;
layer_count = 0;
// trainable_layers only contains references to objects
// that were already deleted through layers.
//
// Therefore, do not delete trainable_layers[i].
delete[] trainable_layers;
trainable_layers = 0;
trainable_layer_count = 0;
// Delete the owned loss object.
delete loss_function;
loss_function = 0;
// Delete the owned optimizer object.
delete optimizer_function;
optimizer_function = 0;
// softmax_classifier_output is not a pointer.
// Do not delete it. Reset only its stored result.
softmax_classifier_output.dinputs.resize(0, 0);
use_softmax_classifier_output = false;
// Reset model results.
outputs.resize(0, 0);
}
dinputs.resize(0, 0);
}
} // namespace neural_network
} // namespace neural_network
} // namespace panic
+58 -29
View File
@@ -48,7 +48,7 @@
* Small vectors and matrices are kept serial because the overhead of starting
* worker threads can be larger than the work itself.
*/
static const panic::types::uint_t one_hot_omp_min_work = 500;
static const panic::types::uint_t one_hot_omp_min_work = 250;
//---------------------------------------------------------------------------------------------------------------------------
// INPLEMENTATION
//---------------------------------------------------------------------------------------------------------------------------
@@ -61,33 +61,57 @@ namespace panic {
// Function Name : panic::tensor::one_hot
//
// Description:
// Creates a one_hot matrix
// Creates a one-hot encoded matrix.
//
// The resulting matrix has:
// one row per sample
// one column per class
//--------------------------------------------------------------------------------------------------------------------------
template <typename T>
bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector & a, panic::tensor::matrix<T>& B){
bool one_hot(const panic::types::uint_t classes, const panic::tensor::uint_vector& a, panic::tensor::matrix<T>& B){
if (!B.resize(size,size)){
const panic::types::uint_t samples = a.size();
const panic::types::uint_t work = samples * classes;
if (samples == static_cast<panic::types::uint_t>(0) || classes == static_cast<panic::types::uint_t>(0)){
return false;
}
bool valid = true;
// Validate labels
// valid remains true only if every divisor is nonzero.
PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(a.size() > one_hot_omp_min_work, &&, valid)
for (panic::types::uint_t i = 0; i < a.size(); ++i){
const bool nonzero = a[i] <= classes;
valid = valid && nonzero;
}
if (!valid){
return false;
}
if (size != a.size()){
return false;
}
PANIC_OMP_PARALLEL_FOR_IF(size*size > one_hot_omp_min_work)
for (panic::types::uint_t i = 0; i < size; ++i){
for (panic::types::uint_t j = 0; j < size; ++j){
if (a[i] == j){
B(i,j) = T{1};
}
else{
B(i,j) = T{0};
}
}
}
return true;
// One row per sample and one column per class.
if (!B.resize(samples, classes)){
return false;
}
PANIC_OMP_PARALLEL_FOR_IF(work > one_hot_omp_min_work)
for(panic::types::uint_t i = 0; i < samples; ++i){
for(panic::types::uint_t j = 0; j < classes; ++j){
if (a[i] == j){
B(i, j) = T{1};
}
else {
B(i, j) = T{0};
}
}
}
return true;
}
//--------------------------------------------------------------------------------------------------------------------------
// EXPLICIT TEMPLATE INSTANTIATION
@@ -95,17 +119,22 @@ bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector &
// The implementation is in this .cpp file.
// Build the overload for the official PANIC numeric types.
//--------------------------------------------------------------------------------------------------------------------------
template bool one_hot(const panic::types::uint_t size,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::uint_t>& B
template bool one_hot(
const panic::types::uint_t classes,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::uint_t>& B
);
template bool one_hot(const panic::types::uint_t size,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::int_t>& B
template bool one_hot(
const panic::types::uint_t classes,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::int_t>& B
);
template bool one_hot(const panic::types::uint_t size,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::real_t>& B
template bool one_hot(
const panic::types::uint_t classes,
const panic::tensor::uint_vector& a,
panic::tensor::matrix<panic::types::real_t>& B
);