Softmax + Categorical Crossentropy
I fixed the activation+loss function, you can't select it directly, it automaticly uses it if it can. I also fixed one-hot generator. Still haven't tested the activation + loss nor any other backward function. I'll do that when I get to the optimizers which is next.
This commit is contained in:
+11
-31
@@ -134,18 +134,13 @@
|
||||
// Expands to:
|
||||
// #pragma omp parallel for if(condition)
|
||||
// num_threads(PANIC_OMP_NUM_THREADS)
|
||||
// schedule(static)
|
||||
// reduction(operation:variable)
|
||||
//
|
||||
// Runs the loop in parallel only when condition is true.
|
||||
// Each thread receives a private copy of variable.
|
||||
// Afterward, OpenMP combines those copies using operation.
|
||||
// schedule(static) assigns fixed groups of iterations to each thread.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
|
||||
num_threads(PANIC_OMP_NUM_THREADS) \
|
||||
schedule(static) \
|
||||
reduction(operation:variable)))
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable) num_threads(PANIC_OMP_NUM_THREADS)))
|
||||
|
||||
#else
|
||||
|
||||
@@ -165,19 +160,13 @@
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition)))
|
||||
|
||||
// Expands to:
|
||||
// #pragma omp parallel for if(condition)
|
||||
// num_threads(PANIC_OMP_NUM_THREADS)
|
||||
// schedule(static)
|
||||
// reduction(operation:variable)
|
||||
// #pragma omp parallel for if(condition) reduction(operation:variable)
|
||||
//
|
||||
// Runs the loop in parallel only when condition is true.
|
||||
// Each thread receives a private copy of variable.
|
||||
// Afterward, OpenMP combines those copies using operation.
|
||||
// schedule(static) assigns fixed groups of iterations to each thread.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) \
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) \
|
||||
schedule(static) \
|
||||
reduction(operation:variable)))
|
||||
_Pragma(PANIC_STRINGIFY(omp parallel for if(condition) reduction(operation:variable)))
|
||||
|
||||
#endif
|
||||
|
||||
@@ -199,25 +188,16 @@
|
||||
// }
|
||||
//
|
||||
// That means the same code still works on microcontrollers and non-OpenMP builds.
|
||||
|
||||
// Without OpenMP, the normal for-loop remains.
|
||||
#define PANIC_OMP_PARALLEL_FOR
|
||||
|
||||
// Check that condition is syntactically valid, but do not evaluate it.
|
||||
#define PANIC_OMP_PARALLEL_FOR_IF(condition) static_cast<void>(sizeof(condition));
|
||||
|
||||
// The operation and variable are only needed by the OpenMP pragma.
|
||||
// The serial loop itself performs the calculation normally.
|
||||
#define PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(condition, operation, variable) static_cast<void>(sizeof(condition));
|
||||
|
||||
#endif
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// TYPE DESCRIPTION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// VARIABLE DESCRIPTION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// FUNCTION PROTOTYPE
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
// None.
|
||||
|
||||
@@ -70,6 +70,16 @@ struct activation_relu : public layer{
|
||||
*/
|
||||
~activation_relu() = default;
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::activation_relu;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
@@ -70,6 +70,17 @@ struct activation_softmax : public layer{
|
||||
*/
|
||||
~activation_softmax() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::activation_softmax;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
+2
-40
@@ -54,47 +54,9 @@ namespace panic{
|
||||
*
|
||||
* The struct is used in PANIC nural_network library.
|
||||
*/
|
||||
struct activation_softmax_loss_categorical_crossentropy: loss{
|
||||
|
||||
activation_softmax activation;
|
||||
loss_categorical_crossentropy loss;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Empthy constructor
|
||||
*
|
||||
*/
|
||||
activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
/**
|
||||
* @brief Default de-constructor
|
||||
*
|
||||
*/
|
||||
~activation_softmax_loss_categorical_crossentropy() = default;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
*
|
||||
* @param y_pred Matrix of model predection.
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
*
|
||||
* @param y_pred Matrix of model predection.
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
|
||||
struct activation_softmax_loss_categorical_crossentropy{
|
||||
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
* @brief backward function to calculate from losses
|
||||
|
||||
@@ -44,6 +44,14 @@ namespace panic{
|
||||
|
||||
|
||||
|
||||
enum struct layer_type {
|
||||
unknown,
|
||||
layer_dense,
|
||||
activation_relu,
|
||||
activation_softmax
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* @brief Base layer for the rest of the neural network library to use
|
||||
*
|
||||
@@ -85,6 +93,16 @@ struct layer{
|
||||
virtual ~layer() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Virtual layer_type function for derivative layers
|
||||
*
|
||||
* @Note This returns the type of layer it is
|
||||
* unknown be default
|
||||
*/
|
||||
virtual layer_type get_type() const {
|
||||
return layer_type::unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Virtual forward function for derivative layers
|
||||
*
|
||||
@@ -103,7 +121,7 @@ struct layer{
|
||||
* @Note It's equal to 0 because it make the derivative
|
||||
* object NEEDS to have these function to work.
|
||||
*/
|
||||
virtual bool backward(const panic::tensor::real_matrix& dinputs) = 0;
|
||||
virtual bool backward(const panic::tensor::real_matrix& dvalues) = 0;
|
||||
|
||||
};
|
||||
|
||||
|
||||
@@ -58,30 +58,6 @@ namespace panic{
|
||||
*/
|
||||
struct layer_dense : trainable_layer{
|
||||
|
||||
/**
|
||||
* @brief Emphty matrix to store input data
|
||||
*
|
||||
*/
|
||||
panic::tensor::real_matrix ipnuts;
|
||||
|
||||
/**
|
||||
* @brief Emphty weight matrix to store layer weights
|
||||
*
|
||||
* Weight shape:
|
||||
* input_size x neuron_count
|
||||
*/
|
||||
panic::tensor::real_matrix weights;
|
||||
panic::tensor::real_matrix dweights;
|
||||
|
||||
/**
|
||||
* @brief Emphty bias vector to store layer bias
|
||||
*
|
||||
* Bias shape:
|
||||
* 1 x neuron_count
|
||||
*/
|
||||
panic::tensor::real_vector biases;
|
||||
panic::tensor::real_vector dbiases;
|
||||
|
||||
/**
|
||||
* @brief Empthy constructor
|
||||
*
|
||||
@@ -103,6 +79,16 @@ struct layer_dense : trainable_layer{
|
||||
*/
|
||||
~layer_dense() = default;
|
||||
|
||||
/**
|
||||
* @brief get_type function for layer
|
||||
*
|
||||
* @returns the layer type
|
||||
*
|
||||
*/
|
||||
layer_type get_type() const override {
|
||||
return layer_type::layer_dense;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Forward function for layer
|
||||
*
|
||||
|
||||
@@ -44,6 +44,11 @@ namespace panic{
|
||||
namespace neural_network{
|
||||
|
||||
|
||||
enum struct loss_type {
|
||||
unknown,
|
||||
categorical_crossentropy
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* @brief Base loss for the rest of the neural network library to use
|
||||
@@ -58,7 +63,7 @@ namespace panic{
|
||||
struct loss{
|
||||
|
||||
/**
|
||||
* @brief Emphty vector to store sample losses
|
||||
* @brief Emphty vector to store sample losses for each sample in the batch.
|
||||
*
|
||||
*/
|
||||
panic::tensor::real_vector sample_losses;
|
||||
@@ -66,19 +71,15 @@ struct loss{
|
||||
/**
|
||||
* @brief Mean loss over the entire batch.
|
||||
*/
|
||||
panic::types::real_t data_loss;
|
||||
panic::types::real_t data_loss = 0;
|
||||
|
||||
/**
|
||||
* @brief Matrix for backwards pass
|
||||
* @brief Gradient with respect to the loss input.
|
||||
*
|
||||
* This will be used later during the backward pass.
|
||||
*/
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
* @brief Matrix for output of loss function
|
||||
*/
|
||||
panic::tensor::real_matrix outputs;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Default de-constructor
|
||||
*
|
||||
@@ -86,6 +87,16 @@ struct loss{
|
||||
virtual ~loss() = default;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Virtual loss_type function for derivative losses
|
||||
*
|
||||
* @Note This returns the type of loss it is
|
||||
* unknown be default
|
||||
*/
|
||||
virtual loss_type get_type() const {
|
||||
return loss_type::unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Virtual forward function for derivative loss functions
|
||||
*
|
||||
@@ -146,7 +157,7 @@ struct loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
virtual bool calculate(
|
||||
bool calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true);
|
||||
|
||||
@@ -157,7 +168,7 @@ struct loss{
|
||||
* @param y_true Matrix of true label of data.
|
||||
*
|
||||
*/
|
||||
virtual bool calculate(
|
||||
bool calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true);
|
||||
|
||||
|
||||
@@ -53,10 +53,17 @@ namespace panic{
|
||||
*
|
||||
* The struct is used for PANIC neural_network library.
|
||||
*/
|
||||
struct loss_categorical_crossentropy: loss{
|
||||
|
||||
|
||||
struct loss_categorical_crossentropy: public loss{
|
||||
|
||||
/**
|
||||
* @brief get_type function for loss
|
||||
*
|
||||
* @returns the loss type
|
||||
*
|
||||
*/
|
||||
loss_type get_type() const override {
|
||||
return loss_type::categorical_crossentropy;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
@@ -65,9 +72,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true)override;
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief forward function to calculate losses
|
||||
@@ -78,9 +83,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true)override;
|
||||
bool forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
/**
|
||||
@@ -90,9 +93,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @param y_true Vector of true label of data.
|
||||
*
|
||||
*/
|
||||
bool backward(
|
||||
const panic::tensor::real_matrix& dvalues,
|
||||
const panic::tensor::uint_vector& y_true) override;
|
||||
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::uint_vector& y_true) override;
|
||||
|
||||
/**
|
||||
* @brief backward function to calculate from losses
|
||||
@@ -103,9 +104,7 @@ struct loss_categorical_crossentropy: loss{
|
||||
* @Note Overloaded if one-shot endcoded
|
||||
* is used.
|
||||
*/
|
||||
bool backward(
|
||||
const panic::tensor::real_matrix& dvalues,
|
||||
const panic::tensor::real_matrix& y_true) override;
|
||||
bool backward(const panic::tensor::real_matrix& dvalues, const panic::tensor::real_matrix& y_true) override;
|
||||
|
||||
|
||||
};
|
||||
|
||||
@@ -44,6 +44,7 @@
|
||||
#include <neural_network/layer/layer_dense.hpp> // fully connected dense layer
|
||||
|
||||
#include <neural_network/loss/loss.hpp>
|
||||
#include <neural_network/activation_loss/activation_softmax_loss_categorical_crossentropy.hpp>
|
||||
|
||||
#include <neural_network/optimizers/optimizer.hpp>
|
||||
#include <neural_network/optimizers/optimizer_sgd.hpp>
|
||||
@@ -71,44 +72,77 @@ namespace panic{
|
||||
struct model{
|
||||
|
||||
/**
|
||||
* @brief a pointer to a pointer of layers
|
||||
* @brief Array of pointers to all model layers.
|
||||
*
|
||||
* An example:
|
||||
* Example:
|
||||
* layers[0] points to a layer_dense
|
||||
* layers[1] points to an actication function
|
||||
* layers[1] points to an activation_ReLU
|
||||
* layers[2] points to another layer_dense
|
||||
*
|
||||
* @note The model owns these layers and deletes them in clear().
|
||||
*
|
||||
* @note The model owns every object referenced by this array.
|
||||
* clear() deletes each layer and then deletes the array.
|
||||
*/
|
||||
layer** layers;
|
||||
|
||||
/**
|
||||
* @brief Number of layers currently stored in layers.
|
||||
*/
|
||||
panic::types::uint_t layer_count;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Array of pointers to the trainable layers.
|
||||
*
|
||||
* @note These pointers refer to objects already owned through layers.
|
||||
* Do not delete the individual objects through this array.
|
||||
* Only the pointer array itself is owned separately.
|
||||
*/
|
||||
trainable_layer** trainable_layers;
|
||||
|
||||
/**
|
||||
* @brief Number of trainable-layer pointers.
|
||||
*/
|
||||
panic::types::uint_t trainable_layer_count;
|
||||
|
||||
|
||||
/**
|
||||
* @brief Configured loss function.
|
||||
*
|
||||
* @note The model owns this object and deletes it in clear().
|
||||
*/
|
||||
loss* loss_function;
|
||||
|
||||
/**
|
||||
* @brief Configured optimizer.
|
||||
*
|
||||
* @note The model owns this object and deletes it in clear().
|
||||
*/
|
||||
optimizer* optimizer_function;
|
||||
|
||||
|
||||
/**
|
||||
* @brief a pointer the loss function
|
||||
*
|
||||
*
|
||||
* @note The model owns these layers and deletes them in clear().
|
||||
* @brief Optimized backward helper for the combination of
|
||||
* Softmax and categorical cross-entropy.
|
||||
*
|
||||
* This is a normal member object, not a dynamically allocated object.
|
||||
*/
|
||||
loss* loss_function;
|
||||
activation_softmax_loss_categorical_crossentropy softmax_classifier_output;
|
||||
|
||||
// Number of layers currently stored in the model.
|
||||
/**
|
||||
* @brief Stores the number of layers
|
||||
*
|
||||
* @brief Whether the optimized Softmax + categorical
|
||||
* cross-entropy backward path should be used.
|
||||
*/
|
||||
panic::types::uint_t layer_count;
|
||||
bool use_softmax_classifier_output;
|
||||
|
||||
// model output (may be deleted and also used for debug)
|
||||
|
||||
/**
|
||||
* @brief Output of the final model layer.
|
||||
*/
|
||||
panic::tensor::real_matrix outputs;
|
||||
// model dinputs (may be deleted and also used for debug)
|
||||
|
||||
/**
|
||||
* @brief Gradient with respect to the model input.
|
||||
*/
|
||||
panic::tensor::real_matrix dinputs;
|
||||
|
||||
/**
|
||||
@@ -231,16 +265,32 @@ struct model{
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.bacward(dvalues_data_matrix)
|
||||
* model.backward(model_output, y_true_values)
|
||||
* @endcode
|
||||
*
|
||||
* @param dvalues diput data.
|
||||
* @param output Model output.
|
||||
* @param y_true True values for data.
|
||||
*
|
||||
* @return true looped over every layer.
|
||||
*
|
||||
*/
|
||||
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true);
|
||||
|
||||
/**
|
||||
* @brief Loops over all layers backward function
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.backward(model_output, y_true_values)
|
||||
* @endcode
|
||||
*
|
||||
* @param output Model output.
|
||||
* @param y_true True values for data.
|
||||
*
|
||||
* @return true looped over every layer.
|
||||
*
|
||||
*/
|
||||
bool backward(const panic::tensor::real_matrix& dvalues);
|
||||
bool backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true);
|
||||
|
||||
|
||||
/**
|
||||
@@ -259,24 +309,6 @@ struct model{
|
||||
bool add_loss_categorical_crossentropy();
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Adds activation softmax AND loss for categorical crossentropy.
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.activation_softmax_loss_categorical_crossentropy();
|
||||
* @endcode
|
||||
*
|
||||
*
|
||||
* @return true if activation and loss is added
|
||||
*
|
||||
* @note This function is convenient, but it allocates a new layer.
|
||||
*/
|
||||
bool activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Adds optimizer_sgd to the model
|
||||
*
|
||||
@@ -295,7 +327,15 @@ struct model{
|
||||
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief Finalizes the model configuration.
|
||||
*
|
||||
* Detects whether the model can use the optimized
|
||||
* Softmax + categorical-cross-entropy backward pass.
|
||||
*
|
||||
* @return true if the model configuration is valid.
|
||||
*/
|
||||
bool finalize();
|
||||
|
||||
/**
|
||||
* @brief Trains the model with input data
|
||||
|
||||
@@ -88,15 +88,6 @@
|
||||
// #define TEST_FALG 1
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : benchmark_omp_min_work
|
||||
//
|
||||
@@ -725,26 +716,44 @@ int main(void) {
|
||||
panic::neural_network::model mymodel;
|
||||
|
||||
// Create Dense layer with 2 input features and 3 output values
|
||||
mymodel.add_layer_dense(2,3);
|
||||
if (!mymodel.add_layer_dense(2,3)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Create an activation ReLU layer
|
||||
mymodel.add_activation_relu();
|
||||
if (!mymodel.add_activation_relu()){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Create a second dense layer with 3 inputs and 3 outputs
|
||||
mymodel.add_layer_dense(3, 3);
|
||||
if (!mymodel.add_layer_dense(3, 3)){
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Create activation softmax layer
|
||||
mymodel.add_activation_softmax();
|
||||
if (!mymodel.add_activation_softmax()){
|
||||
return false;
|
||||
}
|
||||
|
||||
mymodel.add_loss_categorical_crossentropy();
|
||||
//mymodel.activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
mymodel.add_optimizer_sgd();
|
||||
if (!mymodel.add_loss_categorical_crossentropy()){
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!mymodel.finalize()){
|
||||
return false;
|
||||
}
|
||||
|
||||
//mymodel.add_optimizer_sgd();
|
||||
|
||||
panic::types::uint_t epochs = 10;
|
||||
panic::types::uint_t print_every = 1;
|
||||
|
||||
mymodel.train(X, y, epochs, print_every);
|
||||
if (!mymodel.train(X, y, epochs, print_every)){
|
||||
std::cout << "Training failed" << std::endl;
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
+17
-1
@@ -72,6 +72,10 @@ namespace panic {
|
||||
template <typename T>
|
||||
panic::types::real_t mean(const panic::tensor::vector<T>& a){
|
||||
|
||||
if (a.size() == static_cast<panic::types::uint_t>(0)){
|
||||
return static_cast<panic::types::real_t>(0);
|
||||
}
|
||||
|
||||
const panic::types::real_t sum = static_cast<panic::types::real_t>(panic::math::sum(a));
|
||||
|
||||
return sum / static_cast<panic::types::real_t>(a.size());
|
||||
@@ -104,6 +108,10 @@ panic::types::real_t mean(const panic::tensor::matrix<T>& A){
|
||||
panic::types::uint_t cols = A.cols();
|
||||
panic::types::uint_t work = rows*cols;
|
||||
|
||||
if (work == static_cast<panic::types::uint_t>(0)){
|
||||
return static_cast<panic::types::real_t>(0);
|
||||
}
|
||||
|
||||
panic::types::real_t sum = static_cast<panic::types::real_t> (panic::math::sum(A));
|
||||
|
||||
return sum / static_cast<panic::types::real_t>(work);
|
||||
@@ -138,6 +146,10 @@ bool mean_rowwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
|
||||
panic::types::uint_t cols = A.cols();
|
||||
panic::types::uint_t work = rows*cols;
|
||||
|
||||
if (work == static_cast<panic::types::uint_t>(0)){
|
||||
return static_cast<panic::types::real_t>(0);
|
||||
}
|
||||
|
||||
if ( !b.resize(rows) ){
|
||||
return false;
|
||||
}
|
||||
@@ -245,6 +257,10 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
|
||||
panic::types::uint_t cols = A.cols();
|
||||
panic::types::uint_t work = rows*cols;
|
||||
|
||||
if (work == static_cast<panic::types::uint_t>(0)){
|
||||
return static_cast<panic::types::real_t>(0);
|
||||
}
|
||||
|
||||
if ( !b.resize(cols) ){
|
||||
return false;
|
||||
}
|
||||
@@ -256,7 +272,7 @@ bool mean_colwise(const panic::tensor::real_matrix& A, panic::tensor::real_vecto
|
||||
// Each thread handles separate cols and writes to a separate b[i].
|
||||
PANIC_OMP_PARALLEL_FOR_IF(work > mean_omp_min_work)
|
||||
for (panic::types::uint_t i = 0; i < cols; ++i){
|
||||
b[i] /= cols;
|
||||
b[i] /= rows;
|
||||
}
|
||||
|
||||
return true;
|
||||
|
||||
@@ -52,6 +52,8 @@
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t activation_ReLU_omp_min_size = 500;
|
||||
|
||||
static const panic::types::real_t almost_zero = static_cast<panic::types::real_t> (1e-7);
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INPLEMENTATION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -80,7 +82,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){
|
||||
|
||||
inputs = input_data;
|
||||
|
||||
panic::math::clip_lower(inputs, 0.0f, outputs);
|
||||
panic::math::clip_lower(inputs, almost_zero, outputs);
|
||||
|
||||
|
||||
return true;
|
||||
@@ -97,8 +99,7 @@ bool activation_relu::forward(const panic::tensor::real_matrix& input_data){
|
||||
bool activation_relu::backward(const panic::tensor::real_matrix& dvalues){
|
||||
|
||||
// Zero gradients where input values were negative
|
||||
panic::types::real_t zero = 0;
|
||||
dinputs = panic::math::clip_lower(dvalues, zero);
|
||||
dinputs = panic::math::clip_lower(dvalues, almost_zero);
|
||||
|
||||
|
||||
return true;
|
||||
|
||||
+1
-60
@@ -52,7 +52,7 @@
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 500;
|
||||
static const panic::types::uint_t activation_softmax_loss_categorical_crossentropy_omp_min_size = 250;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INPLEMENTATION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -60,65 +60,6 @@ static const panic::types::uint_t activation_softmax_loss_categorical_crossentro
|
||||
namespace panic{
|
||||
namespace neural_network{
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Constructor Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy
|
||||
//
|
||||
// Description:
|
||||
// Creates an empty layer.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
activation_softmax_loss_categorical_crossentropy::activation_softmax_loss_categorical_crossentropy() {
|
||||
}
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward
|
||||
//
|
||||
// Description:
|
||||
// Calculated the forward pass
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){
|
||||
|
||||
// Output layers activation function
|
||||
if (!activation.forward(y_pred)){
|
||||
return false;
|
||||
}
|
||||
// Set the output
|
||||
outputs = activation.outputs;
|
||||
|
||||
// calculate the loss value.
|
||||
if (!loss.calculate(outputs, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.forward
|
||||
//
|
||||
// Description:
|
||||
// Calculated the forward pass
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool activation_softmax_loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true){
|
||||
|
||||
// Output layers activation function
|
||||
if (!activation.forward(y_pred)){
|
||||
return false;
|
||||
}
|
||||
// Set the output
|
||||
outputs = activation.outputs;
|
||||
|
||||
// calculate the loss value.
|
||||
if (!loss.calculate(outputs, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::activation_softmax_loss_categorical_crossentropy.backward
|
||||
//
|
||||
|
||||
@@ -86,7 +86,7 @@ layer_dense::layer_dense() {
|
||||
layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons) {
|
||||
weights.resize(input_size, neurons);
|
||||
panic::random::uniform(weights);
|
||||
panic::math::mul(weights, 0.01f, weights);
|
||||
panic::math::mul(weights, static_cast<panic::types::real_t>(0.01), weights);
|
||||
|
||||
biases.resize(neurons);
|
||||
biases.fill(0);
|
||||
|
||||
@@ -69,18 +69,21 @@ bool loss::calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true){
|
||||
|
||||
// Calculate sample losses
|
||||
data_loss = static_cast<panic::types::real_t>(0);
|
||||
|
||||
// Calls the derived loss function's forward().
|
||||
if (!forward(y_pred, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// forward() must produce one loss per sample.
|
||||
if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Calculate mean loss
|
||||
data_loss = panic::math::mean(sample_losses);
|
||||
|
||||
// This is so the model can use every loss functions
|
||||
// It's a problem to get the output when useing an activation+loss function
|
||||
outputs = y_pred;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -94,17 +97,21 @@ bool loss::calculate(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true){
|
||||
|
||||
data_loss = static_cast<panic::types::real_t>(0);
|
||||
|
||||
// Calls the derived loss function's forward().
|
||||
if (!forward(y_pred, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// forward() must produce one loss per sample.
|
||||
if (sample_losses.size() == 0 || sample_losses.size() != y_pred.rows()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Calculate mean loss
|
||||
data_loss = panic::math::mean(sample_losses);
|
||||
|
||||
|
||||
// This is so the model can use every loss functions
|
||||
// It's a problem to get the output when useing an activation+loss function
|
||||
outputs = y_pred;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -57,7 +57,7 @@
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 500;
|
||||
static const panic::types::uint_t loss_categorical_crossentropy_omp_min_size = 250;
|
||||
|
||||
static const panic::types::real_t clip_min = 1e-7;
|
||||
static const panic::types::real_t clip_max = 1 - 1e-7;
|
||||
@@ -75,20 +75,34 @@ namespace panic{
|
||||
// Description:
|
||||
// Default implementation for catecorical labels. Derived classes can override it.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool loss_categorical_crossentropy::forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::uint_vector& y_true){
|
||||
bool loss_categorical_crossentropy::forward(const panic::tensor::real_matrix& y_pred, const panic::tensor::uint_vector& y_true){
|
||||
|
||||
// Number of samples in a batch
|
||||
const panic::types::uint_t samples = y_pred.rows();
|
||||
|
||||
if (samples != y_true.size()){
|
||||
// Number of classes
|
||||
const panic::types::uint_t classes = y_pred.cols();
|
||||
|
||||
|
||||
if (samples == 0 || classes == 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (y_true.size() != samples) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// clip data to prevent log by 0
|
||||
// Clip both sides to not drag the mean towards any value
|
||||
/*
|
||||
// Validate labels before the OpenMP loop.
|
||||
// Maybe it has to high overhead.
|
||||
for (panic::types::uint_t i = 0; i < samples; ++i) {
|
||||
if (y_true[i] >= classes) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
*/
|
||||
|
||||
// Clip probabilities to prevent log(0).
|
||||
panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max);
|
||||
|
||||
// Vector to hold the correct confidences
|
||||
@@ -96,22 +110,11 @@ bool loss_categorical_crossentropy::forward(
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(samples > loss_categorical_crossentropy_omp_min_size)
|
||||
for (panic::types::uint_t i = 0; i < samples; ++i){
|
||||
const panic::types::uint_t idx = y_true[i];
|
||||
|
||||
//if (idx >= y_pred.cols()){
|
||||
// return false;
|
||||
//}
|
||||
|
||||
correct_confidences[i] = y_pred_cliped(i, idx);
|
||||
correct_confidences[i] = y_pred_cliped(i, y_true[i]);
|
||||
}
|
||||
|
||||
// Calculate losses
|
||||
panic::tensor::real_vector negative_log_likelihoos(samples);
|
||||
|
||||
negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg);
|
||||
|
||||
sample_losses = negative_log_likelihoos;
|
||||
|
||||
// -log(correct confidence)
|
||||
sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg);
|
||||
|
||||
return true;
|
||||
}
|
||||
@@ -127,32 +130,33 @@ bool loss_categorical_crossentropy::forward(
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true){
|
||||
|
||||
|
||||
if (y_pred.rows() != y_true.rows() || y_pred.cols() != y_true.cols()){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Number of samples in a batch
|
||||
const panic::types::uint_t samples = y_pred.rows();
|
||||
|
||||
|
||||
// clip data to prevent log by 0
|
||||
// Clip both sides to not drag the mean towards any value
|
||||
panic::tensor::real_matrix y_pred_cliped = panic::math::clip(y_pred, clip_min, clip_max);
|
||||
|
||||
// Vector to hold the correct confidences
|
||||
panic::tensor::real_vector correct_confidences(samples);
|
||||
// Number of classes
|
||||
const panic::types::uint_t classes = y_pred.cols();
|
||||
|
||||
|
||||
correct_confidences = panic::math::sum_rowwise(panic::math::mul(y_pred_cliped, y_true));
|
||||
if (samples == 0 || classes == 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Calculate losses
|
||||
panic::tensor::real_vector negative_log_likelihoos(samples);
|
||||
if (y_true.rows() != samples || y_true.cols() != classes){
|
||||
return false;
|
||||
}
|
||||
|
||||
negative_log_likelihoos = panic::math::mul(panic::math::log(correct_confidences), neg);
|
||||
// Clip probabilities to prevent log(0).
|
||||
panic::tensor::real_matrix y_pred_clipped = panic::math::clip(y_pred, clip_min, clip_max);
|
||||
|
||||
sample_losses = negative_log_likelihoos;
|
||||
|
||||
// Multiplication masks all predictions except the target.
|
||||
const panic::tensor::real_matrix target_confidences = panic::math::mul(y_pred_clipped, y_true);
|
||||
|
||||
const panic::tensor::real_vector correct_confidences = panic::math::sum_rowwise(target_confidences);
|
||||
|
||||
// -log(correct confidence)
|
||||
sample_losses = panic::math::mul(panic::math::log(correct_confidences), neg);
|
||||
|
||||
return true;
|
||||
}
|
||||
@@ -167,8 +171,7 @@ bool loss_categorical_crossentropy::forward(
|
||||
// Description:
|
||||
// Default implementation for catecorical labels. Derived classes can override it.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool loss_categorical_crossentropy::backward(
|
||||
const panic::tensor::real_matrix& dvalues,
|
||||
bool loss_categorical_crossentropy::backward(const panic::tensor::real_matrix& dvalues,
|
||||
const panic::tensor::uint_vector& y_true){
|
||||
|
||||
if (dvalues.rows() == 0 || dvalues.cols() == 0 || y_true.size() != dvalues.rows()){
|
||||
|
||||
@@ -58,6 +58,19 @@
|
||||
|
||||
// Remember ti disable
|
||||
#include <iostream> // for std::cout, std::endl
|
||||
#include <io/print_tensor.hpp>
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// PRIVATE CONSTANTS
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
/**
|
||||
* @brief Minimum number of element operations before using the OpenMP-enabled loop.
|
||||
*
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t model_omp_min_work = 500;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// IMPLEMENTATION
|
||||
@@ -72,23 +85,17 @@ namespace panic{
|
||||
// Description:
|
||||
// Creates an empty model.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
model::model(){
|
||||
|
||||
// No layers yet.
|
||||
layers = 0;
|
||||
|
||||
// No loss function yet.
|
||||
loss_function = 0;
|
||||
|
||||
// Number of layers is zero.
|
||||
layer_count = 0;
|
||||
|
||||
|
||||
trainable_layers = 0;
|
||||
trainable_layer_count = 0;
|
||||
|
||||
optimizer_function = 0;
|
||||
|
||||
model::model():
|
||||
layers(0),
|
||||
layer_count(0),
|
||||
trainable_layers(0),
|
||||
trainable_layer_count(0),
|
||||
loss_function(0),
|
||||
optimizer_function(0),
|
||||
softmax_classifier_output(),
|
||||
use_softmax_classifier_output(false),
|
||||
outputs(),
|
||||
dinputs(){
|
||||
}
|
||||
|
||||
|
||||
@@ -337,6 +344,14 @@ bool model::forward(const panic::tensor::real_matrix& inputs){
|
||||
return true;
|
||||
}
|
||||
|
||||
// Validate the stored layer pointers before using them.
|
||||
for(panic::types::uint_t i = 0; i < layer_count; ++i){
|
||||
|
||||
if (layers[i] == 0){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// First layer receives the original model input.
|
||||
if (!layers[0]->forward(inputs)){
|
||||
return false;
|
||||
@@ -359,16 +374,152 @@ bool model::forward(const panic::tensor::real_matrix& inputs){
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::bacward
|
||||
// Function Name : panic::neural_network::model::backward
|
||||
//
|
||||
// Description:
|
||||
// Runs the dinputs through every layer in reverse order.
|
||||
// Special case for softmax + categarical crossentropy
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::backward(const panic::tensor::real_matrix& dvalues){
|
||||
bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::uint_vector& y_true){
|
||||
|
||||
if(layers == 0 || layer_count == static_cast<panic::types::uint_t>(0) || loss_function == 0){
|
||||
return false;
|
||||
}
|
||||
|
||||
if(output.rows() == static_cast<panic::types::uint_t>(0) || output.cols() == static_cast<panic::types::uint_t>(0) || output.rows() != y_true.size()){
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
bool valid = true;
|
||||
|
||||
// Validate the stored layer pointers before using them.
|
||||
for(panic::types::uint_t i = 0; i < layer_count; ++i){
|
||||
|
||||
if (layers[i] == 0){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------
|
||||
// Optimized Softmax + categorical cross-entropy backward pass
|
||||
//----------------------------------------------------------------------
|
||||
|
||||
if(use_softmax_classifier_output){
|
||||
|
||||
/*
|
||||
* output already contains the probabilities produced by
|
||||
* the final Softmax layer.
|
||||
*
|
||||
* This calculates the combined Softmax + categorical
|
||||
* cross-entropy derivative.
|
||||
*/
|
||||
if (!softmax_classifier_output.backward(output, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
/*
|
||||
* Store the gradient in the final Softmax layer as well.
|
||||
*
|
||||
* We do not call Softmax::backward(), because its derivative
|
||||
* has already been included in the combined calculation.
|
||||
*
|
||||
* This assignment is useful for consistency and debugging.
|
||||
*/
|
||||
layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs;
|
||||
|
||||
}
|
||||
else{
|
||||
|
||||
//----------------------------------------------------------------------
|
||||
// Normal loss + activation backward pass
|
||||
//----------------------------------------------------------------------
|
||||
|
||||
/*
|
||||
* First calculate the derivative of the loss with respect to
|
||||
* the model output.
|
||||
*/
|
||||
if (!loss_function->backward(output, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// The final layer receives the loss gradient.
|
||||
if (!layers[layer_count - 1]->backward(loss_function->dinputs)){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Every preceding layer receives dinputs from the next layer.
|
||||
for (panic::types::uint_t i = layer_count - 1; i > static_cast<panic::types::uint_t>(0); --i){
|
||||
|
||||
if (!layers[i - 1]->backward(layers[i]->dinputs)){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Gradient with respect to the original model input.
|
||||
dinputs = layers[0]->dinputs;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::backward
|
||||
//
|
||||
// Description:
|
||||
// Runs the dinputs through every layer in reverse order.
|
||||
// Special case for softmax + categarical crossentropy for one-hot encoding
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::backward(const panic::tensor::real_matrix& output, const panic::tensor::real_matrix& y_true){
|
||||
|
||||
if(layers == 0 || layer_count == static_cast<panic::types::uint_t>(0) || loss_function == 0 ){
|
||||
return false;
|
||||
}
|
||||
|
||||
if(output.rows() == static_cast<panic::types::uint_t>(0) ||
|
||||
output.cols() == static_cast<panic::types::uint_t>(0) ||
|
||||
output.rows() != y_true.rows() ||
|
||||
output.cols() != y_true.cols()){
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
if(use_softmax_classifier_output){
|
||||
|
||||
if (!softmax_classifier_output.backward(output, y_true )){
|
||||
return false;
|
||||
}
|
||||
|
||||
// The combined derivative already includes Softmax.
|
||||
layers[layer_count - 1]->dinputs = softmax_classifier_output.dinputs;
|
||||
|
||||
}
|
||||
else {
|
||||
|
||||
if (!loss_function->backward(output, y_true)){
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!layers[layer_count - 1]->backward(loss_function->dinputs)){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
for (panic::types::uint_t i = layer_count - 1; i > static_cast<panic::types::uint_t>(0); --i){
|
||||
if (!layers[i - 1]->backward(layers[i]->dinputs)){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
dinputs = layers[0]->dinputs;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::add_loss_categorical_crossentropy
|
||||
//
|
||||
@@ -385,22 +536,6 @@ bool model::add_loss_categorical_crossentropy(){
|
||||
}
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::activation_softmax_loss_categorical_crossentropy
|
||||
//
|
||||
// Description:
|
||||
// Sets the loss function.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::activation_softmax_loss_categorical_crossentropy(){
|
||||
|
||||
delete loss_function;
|
||||
|
||||
loss_function = new panic::neural_network::activation_softmax_loss_categorical_crossentropy();
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::add_optimizer_sgd
|
||||
//
|
||||
@@ -425,6 +560,53 @@ bool model::add_optimizer_sgd(const panic::types::real_t learning_rate){
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::finalize
|
||||
//
|
||||
// Description:
|
||||
// COnfigures the loss functions if softmax activation and categorical crossentropy loss is used.
|
||||
//
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::finalize(){
|
||||
|
||||
// Always reset first so that calling finalize() more than
|
||||
// once cannot leave an old configuration enabled.
|
||||
use_softmax_classifier_output = false;
|
||||
|
||||
// The model must contain at least one layer.
|
||||
if (layers == 0 || layer_count == static_cast<panic::types::uint_t>(0)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// A loss must have been configured.
|
||||
if (loss_function == 0){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Validate every stored layer pointer.
|
||||
for (panic::types::uint_t i = 0; i < layer_count; ++i){
|
||||
if (layers[i] == 0){
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Gets last layer
|
||||
const layer* last_layer = layers[layer_count - 1];
|
||||
// Checks if last layer is activation softmax
|
||||
const bool last_layer_is_softmax = last_layer->get_type() == layer_type::activation_softmax;
|
||||
// Checks if loss is categorical crossentropy
|
||||
const bool loss_is_categorical_crossentropy = loss_function->get_type() == loss_type::categorical_crossentropy;
|
||||
// If ast layer is activation softmax and loss is categorical crossentropy it's true.
|
||||
use_softmax_classifier_output = last_layer_is_softmax && loss_is_categorical_crossentropy;
|
||||
// The model is still valid when the optimization cannot be used.
|
||||
// It will use the normal loss + activation backward path instead.
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::train
|
||||
//
|
||||
@@ -436,7 +618,7 @@ bool model::train(const panic::tensor::real_matrix& X_train,
|
||||
const panic::types::uint_t epochs,
|
||||
const panic::types::uint_t print_every){
|
||||
|
||||
panic::tensor::uint_vector prediction;
|
||||
panic::tensor::uint_vector predictions;
|
||||
panic::types::real_t accuracy;
|
||||
panic::tensor::uint_vector comparisons;
|
||||
|
||||
@@ -445,31 +627,51 @@ bool model::train(const panic::tensor::real_matrix& X_train,
|
||||
|
||||
forward(X_train);
|
||||
|
||||
// Calculate loss from the model's final output.
|
||||
if (!loss_function->calculate(outputs, y_train)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// There must be one output row for every target label.
|
||||
if (outputs.rows() != y_train.size() || outputs.cols() == static_cast<panic::types::uint_t>(0)){
|
||||
return false;
|
||||
}
|
||||
|
||||
prediction = panic::math::argmax_rowwise(loss_function->outputs);
|
||||
// Find the predicted class for every sample.
|
||||
if (!panic::math::argmax_rowwise(outputs, predictions)){
|
||||
return false;
|
||||
}
|
||||
|
||||
comparisons = panic::math::equal(prediction, y_train);
|
||||
// Compare every prediction with the correct label.
|
||||
if (!panic::math::equal(predictions, y_train, comparisons)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Avoid calculating the mean of an empty vector.
|
||||
if (comparisons.size() == static_cast<panic::types::uint_t>(0)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// comparisons contains only 0 and 1.
|
||||
// Its mean is therefore the classification accuracy.
|
||||
accuracy = panic::math::mean(comparisons);
|
||||
|
||||
if (epoch % print_every == static_cast<panic::types::uint_t>(0)){
|
||||
std::cout << "Epoch: " << epoch;
|
||||
std::cout << " loss: " << loss_function->data_loss;
|
||||
std::cout << " data loss: " << loss_function->data_loss;
|
||||
std::cout << " acc: " << accuracy << std::endl;
|
||||
}
|
||||
|
||||
|
||||
loss_function->backward(loss_function->outputs, y_train);
|
||||
if (!backward(outputs, y_train)){
|
||||
return false;
|
||||
}
|
||||
|
||||
//backward();
|
||||
|
||||
//optimize();
|
||||
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -488,31 +690,55 @@ bool model::train(const panic::tensor::real_matrix& X_train,
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
void model::clear(){
|
||||
|
||||
// Delete every layer object owned by the model.
|
||||
if (layers != 0){
|
||||
|
||||
// Delete each actual layer object.
|
||||
for (panic::types::uint_t i = 0; i < layer_count; ++i){
|
||||
for (
|
||||
panic::types::uint_t i = 0;
|
||||
i < layer_count;
|
||||
++i
|
||||
){
|
||||
delete layers[i];
|
||||
layers[i] = 0;
|
||||
}
|
||||
|
||||
// Delete the array that stored the layer pointers.
|
||||
// Delete the array containing the layer pointers.
|
||||
delete[] layers;
|
||||
}
|
||||
|
||||
|
||||
// Reset to empty state.
|
||||
layers = 0;
|
||||
loss_function = 0;
|
||||
layer_count = 0;
|
||||
|
||||
|
||||
// trainable_layers only contains references to objects
|
||||
// that were already deleted through layers.
|
||||
//
|
||||
// Therefore, do not delete trainable_layers[i].
|
||||
delete[] trainable_layers;
|
||||
|
||||
trainable_layers = 0;
|
||||
trainable_layer_count = 0;
|
||||
|
||||
delete optimizer_function;
|
||||
|
||||
// Delete the owned loss object.
|
||||
delete loss_function;
|
||||
loss_function = 0;
|
||||
|
||||
|
||||
// Delete the owned optimizer object.
|
||||
delete optimizer_function;
|
||||
optimizer_function = 0;
|
||||
|
||||
|
||||
// softmax_classifier_output is not a pointer.
|
||||
// Do not delete it. Reset only its stored result.
|
||||
softmax_classifier_output.dinputs.resize(0, 0);
|
||||
use_softmax_classifier_output = false;
|
||||
|
||||
|
||||
// Reset model results.
|
||||
outputs.resize(0, 0);
|
||||
dinputs.resize(0, 0);
|
||||
}
|
||||
|
||||
} // namespace neural_network
|
||||
|
||||
@@ -48,7 +48,7 @@
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t one_hot_omp_min_work = 500;
|
||||
static const panic::types::uint_t one_hot_omp_min_work = 250;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INPLEMENTATION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -61,22 +61,47 @@ namespace panic {
|
||||
// Function Name : panic::tensor::one_hot
|
||||
//
|
||||
// Description:
|
||||
// Creates a one_hot matrix
|
||||
// Creates a one-hot encoded matrix.
|
||||
//
|
||||
// The resulting matrix has:
|
||||
// one row per sample
|
||||
// one column per class
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector & a, panic::tensor::matrix<T>& B){
|
||||
bool one_hot(const panic::types::uint_t classes, const panic::tensor::uint_vector& a, panic::tensor::matrix<T>& B){
|
||||
|
||||
if (!B.resize(size,size)){
|
||||
const panic::types::uint_t samples = a.size();
|
||||
const panic::types::uint_t work = samples * classes;
|
||||
|
||||
if (samples == static_cast<panic::types::uint_t>(0) || classes == static_cast<panic::types::uint_t>(0)){
|
||||
return false;
|
||||
}
|
||||
|
||||
if (size != a.size()){
|
||||
|
||||
bool valid = true;
|
||||
|
||||
// Validate labels
|
||||
// valid remains true only if every divisor is nonzero.
|
||||
PANIC_OMP_PARALLEL_FOR_REDUCTION_IF(a.size() > one_hot_omp_min_work, &&, valid)
|
||||
for (panic::types::uint_t i = 0; i < a.size(); ++i){
|
||||
const bool nonzero = a[i] <= classes;
|
||||
valid = valid && nonzero;
|
||||
}
|
||||
|
||||
if (!valid){
|
||||
return false;
|
||||
}
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(size*size > one_hot_omp_min_work)
|
||||
for (panic::types::uint_t i = 0; i < size; ++i){
|
||||
for (panic::types::uint_t j = 0; j < size; ++j){
|
||||
|
||||
|
||||
// One row per sample and one column per class.
|
||||
if (!B.resize(samples, classes)){
|
||||
return false;
|
||||
}
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(work > one_hot_omp_min_work)
|
||||
for(panic::types::uint_t i = 0; i < samples; ++i){
|
||||
for(panic::types::uint_t j = 0; j < classes; ++j){
|
||||
if (a[i] == j){
|
||||
B(i, j) = T{1};
|
||||
}
|
||||
@@ -86,7 +111,6 @@ bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector &
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return true;
|
||||
}
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -95,15 +119,20 @@ bool one_hot(const panic::types::uint_t size, const panic::tensor::uint_vector &
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template bool one_hot(const panic::types::uint_t size,
|
||||
template bool one_hot(
|
||||
const panic::types::uint_t classes,
|
||||
const panic::tensor::uint_vector& a,
|
||||
panic::tensor::matrix<panic::types::uint_t>& B
|
||||
);
|
||||
template bool one_hot(const panic::types::uint_t size,
|
||||
|
||||
template bool one_hot(
|
||||
const panic::types::uint_t classes,
|
||||
const panic::tensor::uint_vector& a,
|
||||
panic::tensor::matrix<panic::types::int_t>& B
|
||||
);
|
||||
template bool one_hot(const panic::types::uint_t size,
|
||||
|
||||
template bool one_hot(
|
||||
const panic::types::uint_t classes,
|
||||
const panic::tensor::uint_vector& a,
|
||||
panic::tensor::matrix<panic::types::real_t>& B
|
||||
);
|
||||
|
||||
Reference in New Issue
Block a user