Softmax + Categorical Crossentropy
I fixed the activation+loss function, you can't select it directly, it automaticly uses it if it can. I also fixed one-hot generator. Still haven't tested the activation + loss nor any other backward function. I'll do that when I get to the optimizers which is next.
This commit is contained in:
+11
-31
@@ -134,18 +134,13 @@
|
||||
// Expands to:
|
||||
// #pragma omp parallel for if(condition)
|
||||
// num_threads(PANIC_OMP_NUM_THREADS)
|
||||
// schedule(static)
|
||||
// reduction(operation:variable)
|
||||
//
|
||||
// Runs the loop in parallel only when condition is true.
|
||||
// Each thread receives a private copy of variable.
|
||||
// Afterward, OpenMP combines those copies using operation.
|
||||
// schedule(static) assigns fixed groups of iterations to each thread.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
|
||||
num_threads(PANIC_OMP_NUM_THREADS) \
|
||||
schedule(static) \
|
||||
reduction(operation:variable)))
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable) num_threads(PANIC_OMP_NUM_THREADS)))
|
||||
|
||||
#else
|
||||
|
||||
@@ -165,19 +160,13 @@
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition)))
|
||||
|
||||
// Expands to:
|
||||
// #pragma omp parallel for if(condition)
|
||||
// num_threads(PANIC_OMP_NUM_THREADS)
|
||||
// schedule(static)
|
||||
// reduction(operation:variable)
|
||||
// #pragma omp parallel for if(condition) reduction(operation:variable)
|
||||
//
|
||||
// Runs the loop in parallel only when condition is true.
|
||||
// Each thread receives a private copy of variable.
|
||||
// Afterward, OpenMP combines those copies using operation.
|
||||
// schedule(static) assigns fixed groups of iterations to each thread.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
|
||||
schedule(static) \
|
||||
reduction(operation:variable)))
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable)))
|
||||
|
||||
#endif
|
||||
|
||||
@@ -199,25 +188,16 @@
|
||||
// }
|
||||
//
|
||||
// That means the same code still works on microcontrollers and non-OpenMP builds.
|
||||
|
||||
// Without OpenMP, the normal for-loop remains.
|
||||
#define PANIC_OMP_PARALLEL_FOR
|
||||
|
||||
// Check that condition is syntactically valid, but do not evaluate it.
|
||||
#define PANIC_OMP_PARALLEL_FOR_IF(condition) static_cast<void>(sizeof(condition));
|
||||
|
||||
// The operation and variable are only needed by the OpenMP pragma.
|
||||
// The serial loop itself performs the calculation normally.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) static_cast<void>(sizeof(condition));
|
||||
|
||||
#endif
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// TYPE DESCRIPTION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// VARIABLE DESCRIPTION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// FUNCTION PROTOTYPE
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
@@ -70,6 +70,16 @@ struct activation_relu : public layer{
|
||||
*/
|
||||
~activation_relu() = default;
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::activation_relu;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
@@ -70,6 +70,17 @@ struct activation_softmax : public layer{
|
||||
*/
|
||||
~activation_softmax() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::activation_softmax;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
+2
-40
@@ -54,47 +54,9 @@ namespace panic{
|
||||
*
|
||||
* The struct is used in PANIC nural_network library.
|
||||
*/
|
||||
struct activation_softmax_loss_categorical_crossentropy: loss{
|
||||
|
||||
activation_softmax activation;
|
||||
loss_categorical_crossentropy loss;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Empthy constructor
|
||||
*
|
||||
*/
|
||||
activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
/**
|
||||
* @brief Default de-constructor
|
||||
*
|
||||
*/
|
||||
~activation_softmax_loss_categorical_crossentropy() = default;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
*
|
||||
* @param y_pred Matrix of model predection.
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
*
|
||||
* @param y_pred Matrix of model predection.
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
|
||||
struct activation_softmax_loss_categorical_crossentropy{
|
||||
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
* @brief backward function to calculate from losses
|
||||
|
||||
@@ -44,6 +44,14 @@ namespace panic{
|
||||
|
||||
|
||||
|
||||
enum struct layer_type {
|
||||
unknown,
|
||||
layer_dense,
|
||||
activation_relu,
|
||||
activation_softmax
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* @brief Base layer for the rest of the neural network library to use
|
||||
*
|
||||
@@ -85,6 +93,16 @@ struct layer{
|
||||
virtual ~layer() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Virtual layer_type function for derivative layers
|
||||
*
|
||||
* @Note This returns the type of layer it is
|
||||
* unknown be default
|
||||
*/
|
||||
virtual layer_type get_type() const {
|
||||
return layer_type::unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Virtual forward function for derivative layers
|
||||
*
|
||||
@@ -103,7 +121,7 @@ struct layer{
|
||||
* @Note It's equal to 0 because it make the derivative
|
||||
* object NEEDS to have these function to work.
|
||||
*/
|
||||
virtual bool backward(const panic::tensor::real_matrix& dinputs) = 0;
|
||||
virtual bool backward(const panic::tensor::real_matrix& dvalues) = 0;
|
||||
|
||||
};
|
||||
|
||||
|
||||
@@ -58,30 +58,6 @@ namespace panic{
|
||||
*/
|
||||
struct layer_dense : trainable_layer{
|
||||
|
||||
/**
|
||||
* @brief Emphty matrix to store input data
|
||||
*
|
||||
*/
|
||||
panic::tensor::real_matrix ipnuts;
|
||||
|
||||
/**
|
||||
* @brief Emphty weight matrix to store layer weights
|
||||
*
|
||||
* Weight shape:
|
||||
* input_size x neuron_count
|
||||
*/
|
||||
panic::tensor::real_matrix weights;
|
||||
panic::tensor::real_matrix dweights;
|
||||
|
||||
/**
|
||||
* @brief Emphty bias vector to store layer bias
|
||||
*
|
||||
* Bias shape:
|
||||
* 1 x neuron_count
|
||||
*/
|
||||
panic::tensor::real_vector biases;
|
||||
panic::tensor::real_vector dbiases;
|
||||
|
||||
/**
|
||||
* @brief Empthy constructor
|
||||
*
|
||||
@@ -103,6 +79,16 @@ struct layer_dense : trainable_layer{
|
||||
*/
|
||||
~layer_dense() = default;
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::layer_dense;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
@@ -44,6 +44,11 @@ namespace panic{
|
||||
namespace neural_network{
|
||||
|
||||
|
||||
enum struct loss_type {
|
||||
unknown,
|
||||
categorical_crossentropy
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* @brief Base loss for the rest of the neural network library to use
|
||||
@@ -58,7 +63,7 @@ namespace panic{
|
||||
struct loss{
|
||||
|
||||
/**
|
||||
* @brief Emphty vector to store sample losses
|
||||
* @brief Emphty vector to store sample losses for each sample in the batch.
|
||||
*
|
||||
*/
|
||||
panic::tensor::real_vector sample_losses;
|
||||
@@ -66,19 +71,15 @@ struct loss{
|
||||
/**
|
||||
* @brief Mean loss over the entire batch.
|
||||
*/
|
||||
panic::types::real_t data_loss;
|
||||
panic::types::real_t data_loss = 0;
|
||||
|
||||
/**
|
||||
* @brief Matrix for backwards pass
|
||||
* @brief Gradient with respect to the loss input.
|
||||
*
|
||||
* This will be used later during the backward pass.
|
||||
*/
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
* @brief Matrix for output of loss function
|
||||
*/
|
||||
panic::tensor::real_matrix outputs;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Default de-constructor
|
||||
*
|
||||
@@ -86,6 +87,16 @@ struct loss{
|
||||
virtual ~loss() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Virtual loss_type function for derivative losses
|
||||
*
|
||||
* @Note This returns the type of loss it is
|
||||
* unknown be default
|
||||
*/
|
||||
virtual loss_type get_type() const {
|
||||
return loss_type::unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Virtual forward function for derivative loss functions
|
||||
*
|
||||
@@ -146,7 +157,7 @@ struct loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
virtual bool calculate(
|
||||
bool calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true);
|
||||
|
||||
@@ -157,7 +168,7 @@ struct loss{
|
||||
* @param y_true Matrix of true label of data.
|
||||
*
|
||||
*/
|
||||
virtual bool calculate(
|
||||
bool calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true);
|
||||
|
||||
|
||||
@@ -53,10 +53,17 @@ namespace panic{
|
||||
*
|
||||
* The struct is used for PANIC neural_network library.
|
||||
*/
|
||||
struct loss_categorical_crossentropy: loss{
|
||||
|
||||
|
||||
struct loss_categorical_crossentropy: public loss{
|
||||
|
||||
/**
|
||||
* @brief get_type function for loss
|
||||
*
|
||||
* @returns the loss type
|
||||
*
|
||||
*/
|
||||
loss_type get_type() const override {
|
||||
return loss_type::categorical_crossentropy;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
@@ -65,9 +72,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true)override;
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
@@ -78,9 +83,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true)override;
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
/**
|
||||
@@ -90,9 +93,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool backward(
|
||||
const panic::tensor::real_matrix& dvalues,
|
||||
const panic::tensor::uint_vector& y_true) override;
|
||||
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief backward function to calculate from losses
|
||||
@@ -103,9 +104,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool backward(
|
||||
const panic::tensor::real_matrix& dvalues,
|
||||
const panic::tensor::real_matrix& y_true) override;
|
||||
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
};
|
||||
|
||||
@@ -44,6 +44,7 @@
|
||||
#include <neural_network/layer/layer_dense.hpp> // fully connected dense layer
|
||||
|
||||
#include <neural_network/loss/loss.hpp>
|
||||
#include <neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp>
|
||||
|
||||
#include <neural_network/optimizers/optimizer.hpp>
|
||||
#include <neural_network/optimizers/optimizer_sgd.hpp>
|
||||
@@ -71,44 +72,77 @@ namespace panic{
|
||||
struct model{
|
||||
|
||||
/**
|
||||
* @brief a pointer to a pointer of layers
|
||||
* @brief Array of pointers to all model layers.
|
||||
*
|
||||
* An example:
|
||||
* layers[0] points to a layer_dense
|
||||
* layers[1] points to an actication function
|
||||
* layers[2] points to another layer_dense
|
||||
*
|
||||
* @note The model owns these layers and deletes them in clear().
|
||||
*
|
||||
* Example:
|
||||
* layers[0] points to a layer_dense
|
||||
* layers[1] points to an activation_ReLU
|
||||
* layers[2] points to another layer_dense
|
||||
*
|
||||
* @note The model owns every object referenced by this array.
|
||||
* clear() deletes each layer and then deletes the array.
|
||||
*/
|
||||
layer** layers;
|
||||
|
||||
/**
|
||||
* @brief Number of layers currently stored in layers.
|
||||
*/
|
||||
panic::types::uint_t layer_count;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Array of pointers to the trainable layers.
|
||||
*
|
||||
* @note These pointers refer to objects already owned through layers.
|
||||
* Do not delete the individual objects through this array.
|
||||
* Only the pointer array itself is owned separately.
|
||||
*/
|
||||
trainable_layer** trainable_layers;
|
||||
|
||||
/**
|
||||
* @brief Number of trainable-layer pointers.
|
||||
*/
|
||||
panic::types::uint_t trainable_layer_count;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Configured loss function.
|
||||
*
|
||||
* @note The model owns this object and deletes it in clear().
|
||||
*/
|
||||
loss* loss_function;
|
||||
|
||||
/**
|
||||
* @brief Configured optimizer.
|
||||
*
|
||||
* @note The model owns this object and deletes it in clear().
|
||||
*/
|
||||
optimizer* optimizer_function;
|
||||
|
||||
|
||||
/**
|
||||
* @brief a pointer the loss function
|
||||
* @brief Optimized backward helper for the combination of
|
||||
* Softmax and categorical cross-entropy.
|
||||
*
|
||||
*
|
||||
* @note The model owns these layers and deletes them in clear().
|
||||
*
|
||||
* This is a normal member object, not a dynamically allocated object.
|
||||
*/
|
||||
loss* loss_function;
|
||||
activation_softmax_loss_categorical_crossentropy softmax_classifier_output;
|
||||
|
||||
// Number of layers currently stored in the model.
|
||||
/**
|
||||
* @brief Stores the number of layers
|
||||
*
|
||||
* @brief Whether the optimized Softmax + categorical
|
||||
* cross-entropy backward path should be used.
|
||||
*/
|
||||
panic::types::uint_t layer_count;
|
||||
bool use_softmax_classifier_output;
|
||||
|
||||
// model output (may be deleted and also used for debug)
|
||||
|
||||
/**
|
||||
* @brief Output of the final model layer.
|
||||
*/
|
||||
panic::tensor::real_matrix outputs;
|
||||
// model dinputs (may be deleted and also used for debug)
|
||||
|
||||
/**
|
||||
* @brief Gradient with respect to the model input.
|
||||
*/
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
@@ -231,16 +265,32 @@ struct model{
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.bacward(dvalues_data_matrix)
|
||||
* model.backward(model_output, y_true_values)
|
||||
* @endcode
|
||||
*
|
||||
* @param dvalues diput data.
|
||||
* @param output Model output.
|
||||
* @param y_true True values for data.
|
||||
*
|
||||
* @return true looped over every layer.
|
||||
*
|
||||
*
|
||||
*/
|
||||
bool backward(const panic::tensor::real_matrix& dvalues);
|
||||
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true);
|
||||
|
||||
/**
|
||||
* @brief Loops over all layers backward function
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.backward(model_output, y_true_values)
|
||||
* @endcode
|
||||
*
|
||||
* @param output Model output.
|
||||
* @param y_true True values for data.
|
||||
*
|
||||
* @return true looped over every layer.
|
||||
*
|
||||
*/
|
||||
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true);
|
||||
|
||||
|
||||
/**
|
||||
@@ -259,24 +309,6 @@ struct model{
|
||||
bool add_loss_categorical_crossentropy();
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Adds activation softmax AND loss for categorical crossentropy.
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.activation_softmax_loss_categorical_crossentropy();
|
||||
* @endcode
|
||||
*
|
||||
*
|
||||
* @return true if activation and loss is added
|
||||
*
|
||||
* @note This function is convenient, but it allocates a new layer.
|
||||
*/
|
||||
bool activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Adds optimizer_sgd to the model
|
||||
*
|
||||
@@ -295,7 +327,15 @@ struct model{
|
||||
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Finalizes the model configuration.
|
||||
*
|
||||
* Detects whether the model can use the optimized
|
||||
* Softmax + categorical-cross-entropy backward pass.
|
||||
*
|
||||
* @return true if the model configuration is valid.
|
||||
*/
|
||||
bool finalize();
|
||||
|
||||
/**
|
||||
* @brief Trains the model with input data
|
||||
|
||||
Reference in New Issue
Block a user