From 2c802e9e8c8acd757fb4d531d41a408f595ae10f Mon Sep 17 00:00:00 2001 From: Michelle Date: Fri, 7 Aug 2026 18:50:33 +0200 Subject: [PATCH] Next up is dropout layers --- include/math/abs.hpp | 163 +++++++++++ include/neural_network/layer/layer_dense.hpp | 7 +- .../neural_network/layer/trainable_layer.hpp | 9 + include/neural_network/loss/loss.hpp | 15 ++ include/neural_network/model/model.hpp | 22 +- main.cpp | 30 ++- src/math/abs.cpp | 255 ++++++++++++++++++ src/neural_network/layer/layer_dense.cpp | 79 +++++- src/neural_network/loss/loss.cpp | 43 ++- src/neural_network/model/model.cpp | 67 ++++- 10 files changed, 670 insertions(+), 20 deletions(-) create mode 100644 include/math/abs.hpp create mode 100644 src/math/abs.cpp diff --git a/include/math/abs.hpp b/include/math/abs.hpp new file mode 100644 index 0000000..f0bad95 --- /dev/null +++ b/include/math/abs.hpp @@ -0,0 +1,163 @@ +/**++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ + * + * PANIC + * Portable Algorithms and Numerics In C++ + * + * Scientific computing from scratch, with feeling. + * + * Copyright (c) 2026 Michelle Bausager + * + * This file is part of PANIC. + * + * PANIC is free software licensed under the GNU General Public License v3.0 or later. + * You may redistribute and/or modify it under the terms of the GPL. + * + * PANIC is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; + * without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. + * See the LICENSE file for the full license text. + * + * SPDX-License-Identifier: GPL-3.0-or-later + * + *++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ + * + * Project Name: PANIC + * Module Name: math + * File Name: abs.cpp + * Revision: 0.1.0 + * Date: 07-08-2026 + * Author: Michelle Bausager + * + * Description: + * Functions to calculate the abs of value + * + *++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++*/ +#pragma once + +#include // for panic::vector +#include // for panic::matrix + +namespace panic{ +namespace math{ + + + +/** + * @brief calculates the abs of a value. + * + * Computes: + * @code + * abs(x, y) + * @endcode + * + * @tparam T Numeric element type. + * @param x Value to take the abs of. + * @param y Result. + * + * + * @note This function is omp-friendly. + */ +template +bool abs(const T x, T& y); + + +/** + * @brief calculates the abs of a value. + * + * Computes: + * @code + * result = abs(k) + * @endcode + * + * @tparam T Numeric element type. + * @param x Value to take the abs of. + * + * @return The calculated value + * + * @note This function is omp-friendly. + */ +template +T abs(const T x); + + +/** + * @brief Calculates the abs elementwise in a vector + * + * Computes: + * @code + * c[i] = abs(a[i]) + * @endcode + * + * @tparam T Numeric element type. + * @param a Input vector. + * @param c Output vector. Resized to match @p a. + * + * @return true if @p c was resized and filled successfully. + * @return false if resizing @p c failed. + * + * @note This overload writes the result into an existing vector to avoid + * unnecessary temporary allocations. + */ +template +bool abs(const panic::tensor::vector& a, panic::tensor::vector& c); + +/** + * @brief Calculates the abs elementwise in a vector + * + * Computes: + * @code + * result[i] = abs(a[i]) + * @endcode + * + * @tparam T Numeric element type. + * @param a Input vector. + * + * @return A new vector containing the result. + * @return An empty vector if the operation fails. + * + * @note This overload is convenient, but may allocate a new vector. + */ +template +panic::tensor::vector abs(const panic::tensor::vector& a); + +/** + * @brief Calculates the abs elementwise of a matrix + * + * Computes: + * @code + * C(i,j) = abs(A(i,j)) + * @endcode + * + * @tparam T Numeric element type. + * @param A Input matrix. + * @param C Output Matrix. Resized to match @p A. + * + * @return true if @p C was resized and filled successfully. + * @return false if resizing @p C failed. + * + * @note This overload writes the result into an existing vector to avoid + * unnecessary temporary allocations. + */ +template +bool abs(const panic::tensor::matrix& A, panic::tensor::matrix& C); + +/** + * @brief Returns the calculated abs elementwise of the matrix + * + * Computes: + * @code + * result(i,j) = abs(A(i,j)) + * @endcode + * + * @tparam T Numeric element type. + * @param A Input matrix. + * + * @return A new matrix containing the result. + * @return An empty matrix if the operation fails. + * + * @note This overload is convenient, but may allocate a new vector. + */ +template +panic::tensor::matrix abs(const panic::tensor::matrix& A); + +} // namespace math +} // namespace panic \ No newline at end of file diff --git a/include/neural_network/layer/layer_dense.hpp b/include/neural_network/layer/layer_dense.hpp index 13797a7..e1e563d 100644 --- a/include/neural_network/layer/layer_dense.hpp +++ b/include/neural_network/layer/layer_dense.hpp @@ -71,7 +71,12 @@ struct layer_dense : trainable_layer{ * @param neurons Amount of neurons in the layer * */ - layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons); + layer_dense(panic::types::uint_t input_size, + panic::types::uint_t neurons, + panic::types::real_t weight_regularizer_l1 = 0, + panic::types::real_t weight_regularizer_l2 = 0, + panic::types::real_t bias_regularizer_l1 = 0, + panic::types::real_t bias_regularizer_l2 = 0); /** * @brief Default de-constructor diff --git a/include/neural_network/layer/trainable_layer.hpp b/include/neural_network/layer/trainable_layer.hpp index c37ce7f..f3d1846 100644 --- a/include/neural_network/layer/trainable_layer.hpp +++ b/include/neural_network/layer/trainable_layer.hpp @@ -63,6 +63,12 @@ struct trainable_layer:layer{ panic::tensor::real_matrix dweights; panic::tensor::real_vector dbiases; + panic::types::real_t weight_regularizer_l1; + panic::types::real_t weight_regularizer_l2; + + panic::types::real_t bias_regularizer_l1; + panic::types::real_t bias_regularizer_l2; + /** * @brief Previous parameter updates used by momentum SGD. * @@ -82,6 +88,9 @@ struct trainable_layer:layer{ panic::tensor::real_vector bias_cache; + + + virtual ~trainable_layer() = default; diff --git a/include/neural_network/loss/loss.hpp b/include/neural_network/loss/loss.hpp index fc4751f..21e12df 100644 --- a/include/neural_network/loss/loss.hpp +++ b/include/neural_network/loss/loss.hpp @@ -39,6 +39,7 @@ #include // panic::tensor::real_matrix (uint_matrix, int_matrix) #include +#include namespace panic{ namespace neural_network{ @@ -73,6 +74,11 @@ struct loss{ */ panic::types::real_t data_loss = 0; + /** + * @brief Regularization loss over a layer. + */ + panic::types::real_t regularization_loss_value = 0; + /** * @brief Gradient with respect to the loss input. * @@ -172,6 +178,15 @@ struct loss{ const panic::tensor::real_matrix& y_pred, const panic::tensor::real_matrix& y_true); + + /** + * @brief Caclculates the regularization loss of a trainable layer + * + */ + bool regularization_loss(const trainable_layer& layer); + + + }; diff --git a/include/neural_network/model/model.hpp b/include/neural_network/model/model.hpp index f7df230..b97931f 100644 --- a/include/neural_network/model/model.hpp +++ b/include/neural_network/model/model.hpp @@ -205,10 +205,12 @@ struct model{ * * @note This function is convenient, but it allocates a new layer. */ - bool add_layer_dense( - panic::types::uint_t input_size, - panic::types::uint_t neuron_count - ); + bool add_layer_dense(panic::types::uint_t input_size, + panic::types::uint_t neurons, + panic::types::real_t weight_regularizer_l1 = 0, + panic::types::real_t weight_regularizer_l2 = 0, + panic::types::real_t bias_regularizer_l1 = 0, + panic::types::real_t bias_regularizer_l2 = 0); /** * @brief Adds a activation ReLU layer to the model. @@ -412,6 +414,18 @@ struct model{ */ bool optimize(); + /** + * @brief Calculates regulaization for parameters on trainable layers + * + * Computes: + * @code + * model.calculate_regularization_loss() + * @endcode + * + * @return true if optimization is done correctly. + * + */ + bool calculate_regularization_loss(panic::types::real_t& regularization_loss); /** diff --git a/main.cpp b/main.cpp index d32e5af..2c11c29 100644 --- a/main.cpp +++ b/main.cpp @@ -706,9 +706,16 @@ int main(void) { panic::tensor::real_matrix X; panic::tensor::uint_vector y; - panic::types::uint_t samples = 100; + panic::types::uint_t samples = 1000; panic::types::uint_t classes = 3; + panic::types::uint_t layer_size = 256; + + + panic::types::real_t weight_regularizer_l1 = 0; + panic::types::real_t weight_regularizer_l2 = 5e-4; + panic::types::real_t bias_regularizer_l1 = 0; + panic::types::real_t bias_regularizer_l2 = 5e-4; // create spiral data panic::neural_network::spiral_data(samples, classes, X, y); @@ -718,7 +725,11 @@ int main(void) { panic::neural_network::model mymodel; // Create Dense layer with 2 input features and 3 output values - if (!mymodel.add_layer_dense(2,64)){ + if (!mymodel.add_layer_dense(2,layer_size, + weight_regularizer_l1, + weight_regularizer_l2, + bias_regularizer_l1, + bias_regularizer_l2)){ return false; } @@ -726,9 +737,14 @@ int main(void) { if (!mymodel.add_activation_relu()){ return false; } + // Create a second dense layer with 3 inputs and 3 outputs - if (!mymodel.add_layer_dense(64, 3)){ + if (!mymodel.add_layer_dense(layer_size, 3, + weight_regularizer_l1, + weight_regularizer_l2, + bias_regularizer_l1, + bias_regularizer_l2)){ return false; } @@ -769,7 +785,7 @@ int main(void) { } */ - panic::types::real_t learning_rate = 0.02; + panic::types::real_t learning_rate = 0.001; panic::types::real_t decay = 1e-5; panic::types::real_t epsilon = 1e-7; panic::types::real_t beta_1 = 0.9; @@ -794,5 +810,11 @@ int main(void) { } + + + + + + return 0; } \ No newline at end of file diff --git a/src/math/abs.cpp b/src/math/abs.cpp new file mode 100644 index 0000000..f126933 --- /dev/null +++ b/src/math/abs.cpp @@ -0,0 +1,255 @@ +/**++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ + * + * PANIC + * Portable Algorithms and Numerics In C++ + * + * Scientific computing from scratch, with feeling. + * + * Copyright (c) 2026 Michelle Bausager + * + * This file is part of PANIC. + * + * PANIC is free software licensed under the GNU General Public License v3.0 or later. + * You may redistribute and/or modify it under the terms of the GPL. + * + * PANIC is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; + * without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. + * See the LICENSE file for the full license text. + * + * SPDX-License-Identifier: GPL-3.0-or-later + * + *++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ + * + * Project Name: PANIC + * Module Name: math + * File Name: abs.cpp + * Revision: 0.1.0 + * Date: 07-08-2026 + * Author: Michelle Bausager + * + * Description: + * Functions to calculate the sqrt of numbers + * + *++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++*/ + +//--------------------------------------------------------------------------------------------------------------------------- +// INCLUDE DESCRIPTION +//----------------------------------------------------------------------------------------------------- + +#include +#include +#include + +#include // for panic::vector +#include // for panic::matrix + +//--------------------------------------------------------------------------------------------------------------------------- +// PRIVATE CONSTANTS +//--------------------------------------------------------------------------------------------------------------------------- +/** + * @brief Minimum number of element operations before using the OpenMP-enabled loop. + * + * Small vectors and matrices are kept serial because the overhead of starting + * worker threads can be larger than the work itself. + */ +static const panic::types::uint_t abs_omp_min_work = 250; +//--------------------------------------------------------------------------------------------------------------------------- +// INPLEMENTATION +//--------------------------------------------------------------------------------------------------------------------------- + + +namespace panic { + namespace math { +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +//-------------------------------------------------------------------------------------------------------------------------- +template +bool abs(const T x, T& y){ + + if (x < static_cast(0)){ + y = -x; + } + else{ + y = x; + } + + return true; +} + +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template bool abs(const panic::types::real_t x, + panic::types::real_t& y +); + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +// +//-------------------------------------------------------------------------------------------------------------------------- +template +T abs(const T x){ + T y; + if (! abs(x,y)){ + return T(); + } + + return y; +} + +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template panic::types::real_t abs(const panic::types::real_t x +); + + + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +// Calculates the abs elementwise of a vector +//-------------------------------------------------------------------------------------------------------------------------- +template +bool abs(const panic::tensor::vector& a, panic::tensor::vector& c){ + + if (!c.resize(a.size())){ + return false; + } + + bool valid = true; + + // valid remains true only if every abs is valid. + PANIC_OMP_PARALLEL_FOR_REDUCTION_IF( a.size() > abs_omp_min_work, &&, valid ) + for (panic::types::uint_t i = 0; i < a.size(); ++i){ + const bool nonzero = abs(a[i], c[i]); + valid = valid && nonzero; + } + + return valid; +} + +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template bool abs(const panic::tensor::vector& a, + panic::tensor::vector& c +); + + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +// Calculates the abs elementwise for a vector +//-------------------------------------------------------------------------------------------------------------------------- +template +panic::tensor::vector abs(const panic::tensor::vector& a){ + panic::tensor::vector c(a.size()); + + if (!abs(a, c)){ + return panic::tensor::vector(); + } + + return c; +} +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template panic::tensor::vector + abs(const panic::tensor::vector& a +); + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +// calculates the natrual abs elementwise of a matrix +//-------------------------------------------------------------------------------------------------------------------------- +template +bool abs(const panic::tensor::matrix& A, panic::tensor::matrix& C){ + + panic::types::uint_t rows = A.rows(); + panic::types::uint_t cols = A.cols(); + panic::types::uint_t work = rows*cols; + + if ( !C.resize(rows, cols) ){ + return false; + } + + bool valid = true; + + // valid remains true only if every abs is valid. + PANIC_OMP_PARALLEL_FOR_IF(work > abs_omp_min_work) + for (panic::types::uint_t i = 0; i < rows; ++i){ + for (panic::types::uint_t j = 0; j < cols; ++j){ + const bool nonzero = abs(A(i,j), C(i,j)); + valid = valid && nonzero; + } + } + + return valid; +} +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template bool abs(const panic::tensor::matrix& A, + panic::tensor::matrix& C +); + + + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::math::abs +// +// Description: +// Calculates the natrual abs element-wise of a matrix +//-------------------------------------------------------------------------------------------------------------------------- +template +panic::tensor::matrix abs(const panic::tensor::matrix& A){ + panic::tensor::matrix C; + + if (!abs(A, C)){ + return panic::tensor::matrix(); + } + + return C; +} +//-------------------------------------------------------------------------------------------------------------------------- +// EXPLICIT TEMPLATE INSTANTIATION +// +// The implementation is in this .cpp file. +// Build the overload for the official PANIC numeric types. +//-------------------------------------------------------------------------------------------------------------------------- +template panic::tensor::matrix + abs(const panic::tensor::matrix& A +); + + + } // namespace math +} // namespace panic diff --git a/src/neural_network/layer/layer_dense.cpp b/src/neural_network/layer/layer_dense.cpp index 0c30afa..e0ae075 100644 --- a/src/neural_network/layer/layer_dense.cpp +++ b/src/neural_network/layer/layer_dense.cpp @@ -55,7 +55,7 @@ * Small vectors and matrices are kept serial because the overhead of starting * worker threads can be larger than the work itself. */ -static const panic::types::uint_t layer_dense_omp_min_size = 500; +static const panic::types::uint_t layer_dense_omp_min_size = 250; //--------------------------------------------------------------------------------------------------------------------------- // INPLEMENTATION //--------------------------------------------------------------------------------------------------------------------------- @@ -75,6 +75,10 @@ layer_dense::layer_dense() { biases.resize(0); dweights.resize(0,0); dbiases.resize(0); + weight_regularizer_l1 = 0; + weight_regularizer_l2 = 0; + bias_regularizer_l1 = 0; + bias_regularizer_l2 = 0; } //-------------------------------------------------------------------------------------------------------------------------- @@ -83,13 +87,26 @@ layer_dense::layer_dense() { // Description: // Creates an empty layer with neurons. //-------------------------------------------------------------------------------------------------------------------------- -layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons) { +layer_dense::layer_dense(panic::types::uint_t input_size, + panic::types::uint_t neurons, + panic::types::real_t weight_regularizer_l1, + panic::types::real_t weight_regularizer_l2, + panic::types::real_t bias_regularizer_l1, + panic::types::real_t bias_regularizer_l2) { + + weights.resize(input_size, neurons); panic::random::uniform(weights, static_cast(-1), static_cast(1)); panic::math::mul(weights, static_cast(0.01), weights); biases.resize(neurons); biases.fill(0); + + this->weight_regularizer_l1 = weight_regularizer_l1; + this->weight_regularizer_l2 = weight_regularizer_l2; + this-> bias_regularizer_l1 = bias_regularizer_l1; + this-> bias_regularizer_l2 = bias_regularizer_l2; + } @@ -134,11 +151,69 @@ bool layer_dense::forward(const panic::tensor::real_matrix& input_data){ //-------------------------------------------------------------------------------------------------------------------------- bool layer_dense::backward(const panic::tensor::real_matrix& dvalues){ + // Gradients on parameters dweights = panic::math::matmul(panic::math::transpose(inputs), dvalues); dbiases = panic::math::sum_colwise(dvalues); + panic::types::uint_t weight_work = dweights.rows()*dweights.cols(); + + if (weight_regularizer_l1 > static_cast(0)){ + + PANIC_OMP_PARALLEL_FOR_IF(weight_work > layer_dense_omp_min_size) + for (panic::types::uint_t i = 0; i < dweights.rows(); ++i){ + for (panic::types::uint_t j = 0; j < dweights.cols(); ++j){ + + if (weights(i,j) < static_cast(0)){ + dweights(i,j) += weight_regularizer_l1 * -1; + } + else{ + dweights(i,j) += weight_regularizer_l1 * 1; + } + } + } + } + + if (weight_regularizer_l2 > static_cast(0)){ + + PANIC_OMP_PARALLEL_FOR_IF(weight_work > layer_dense_omp_min_size) + for (panic::types::uint_t i = 0; i < dweights.rows(); ++i){ + for (panic::types::uint_t j = 0; j < dweights.cols(); ++j){ + + dweights(i,j) += static_cast(2) * weight_regularizer_l2 * weights(i,j); + } + } + } + + + + + + + if (bias_regularizer_l1 > static_cast(0)){ + + PANIC_OMP_PARALLEL_FOR_IF(dbiases.size() > layer_dense_omp_min_size) + for (panic::types::uint_t i = 0; i < dbiases.size(); ++i){ + + if (biases[i] < static_cast(0)){ + dbiases[i] += bias_regularizer_l1 * -1; + } + else{ + dbiases[i] += bias_regularizer_l1 * 1; + } + } + } + + if (bias_regularizer_l2 > static_cast(0)){ + + PANIC_OMP_PARALLEL_FOR_IF(dbiases.size() > layer_dense_omp_min_size) + for (panic::types::uint_t i = 0; i < dbiases.size(); ++i){ + + dbiases[i] += static_cast(2) * bias_regularizer_l2 * biases[i]; + } + } + // Gradients on values dinputs = panic::math::matmul(dvalues, panic::math::transpose(weights)); diff --git a/src/neural_network/loss/loss.cpp b/src/neural_network/loss/loss.cpp index f01ac3f..64f9fa7 100644 --- a/src/neural_network/loss/loss.cpp +++ b/src/neural_network/loss/loss.cpp @@ -41,6 +41,10 @@ #include #include +#include +#include +#include +#include //--------------------------------------------------------------------------------------------------------------------------- @@ -52,7 +56,7 @@ * Small vectors and matrices are kept serial because the overhead of starting * worker threads can be larger than the work itself. */ -static const panic::types::uint_t loss_omp_min_size = 500; +static const panic::types::uint_t loss_omp_min_size = 250; //--------------------------------------------------------------------------------------------------------------------------- @@ -115,6 +119,43 @@ bool loss::calculate( return true; } + + + + +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::neural_network::loss::regularization_loss +// +// Description: +// Calculates the regularization loss of a trainable layer. +//-------------------------------------------------------------------------------------------------------------------------- +bool loss::regularization_loss(const trainable_layer& layer){ + + regularization_loss_value = 0; + + if (layer.weight_regularizer_l1 > 0){ + regularization_loss_value += panic::math::sum(panic::math::abs(layer.weights))*layer.weight_regularizer_l1; + } + + if (layer.weight_regularizer_l2 > 0){ + regularization_loss_value += panic::math::sum(panic::math::mul(layer.weights,layer.weights))*layer.weight_regularizer_l2; + } + + if (layer.bias_regularizer_l1 > 0){ + regularization_loss_value += panic::math::sum(panic::math::abs(layer.biases))*layer.bias_regularizer_l1; + } + + if (layer.bias_regularizer_l2 > 0){ + regularization_loss_value += panic::math::sum(panic::math::mul(layer.biases,layer.biases))*layer.bias_regularizer_l2; + } + + return true; +} + + + + + //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::loss::forward // diff --git a/src/neural_network/model/model.cpp b/src/neural_network/model/model.cpp index 31dc973..a859a99 100644 --- a/src/neural_network/model/model.cpp +++ b/src/neural_network/model/model.cpp @@ -251,12 +251,20 @@ bool model::add_trainable_layer(trainable_layer* new_layer){ // Example: // model.add_layer_dense(100, 64); //-------------------------------------------------------------------------------------------------------------------------- -bool model::add_layer_dense( - panic::types::uint_t input_size, - panic::types::uint_t neuron_count){ +bool model::add_layer_dense(panic::types::uint_t input_size, + panic::types::uint_t neurons, + panic::types::real_t weight_regularizer_l1, + panic::types::real_t weight_regularizer_l2, + panic::types::real_t bias_regularizer_l1, + panic::types::real_t bias_regularizer_l2){ // Create the dense layer. - layer_dense* new_layer = new layer_dense(input_size, neuron_count); + layer_dense* new_layer = new layer_dense(input_size, + neurons, + weight_regularizer_l1, + weight_regularizer_l2, + bias_regularizer_l1, + bias_regularizer_l2); if (new_layer == 0){ return false; @@ -709,8 +717,6 @@ bool model::optimize(){ return false; } - - if (! optimizer_function->update_params(*trainable_layers[i]) ){ return false; } @@ -726,6 +732,41 @@ bool model::optimize(){ } +//-------------------------------------------------------------------------------------------------------------------------- +// Function Name : panic::neural_network::model::calculate_regularization_loss +// +// Description: +// Calculates regulaization for parameters on trainable layers +//-------------------------------------------------------------------------------------------------------------------------- +bool model::calculate_regularization_loss(panic::types::real_t& regularization_loss){ + + regularization_loss = 0; + + if (loss_function == 0){ + return false; + } + + for(panic::types::uint_t i = 0; i < trainable_layer_count; ++i){ + + // Validate the stored layer pointers before using them. + if (trainable_layers[i] == 0){ + return false; + } + + if (! loss_function->regularization_loss(*trainable_layers[i])){ + return false; + } + + regularization_loss += loss_function->regularization_loss_value; + + + } + + return true; +} + + + //-------------------------------------------------------------------------------------------------------------------------- // Function Name : panic::neural_network::model::train @@ -745,13 +786,21 @@ bool model::train(const panic::tensor::real_matrix& X_train, for (panic::types::uint_t epoch = 0; epoch < epochs+1; ++epoch){ - forward(X_train); - + if (!forward(X_train)){ + return false; + } // Calculate loss from the model's final output. if (!loss_function->calculate(outputs, y_train)){ return false; } + // Caluclates regula + panic::types::real_t regularization_loss = 0; + + if (! calculate_regularization_loss(regularization_loss)){ + return false; + } + // There must be one output row for every target label. if (outputs.rows() != y_train.size() || outputs.cols() == static_cast(0)){ return false; @@ -780,6 +829,8 @@ bool model::train(const panic::tensor::real_matrix& X_train, std::cout << "Epoch: " << epoch; std::cout << " lr: " << optimizer_function->current_learning_rate; std::cout << " data loss: " << loss_function->data_loss; + std::cout << " reg loss: " << regularization_loss; + std::cout << " total loss: " << loss_function->data_loss + regularization_loss; std::cout << " acc: " << accuracy << std::endl; }