Next up is dropout layers
This commit is contained in:
@@ -0,0 +1,163 @@
|
||||
/**++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
|
||||
*
|
||||
* PANIC
|
||||
* Portable Algorithms and Numerics In C++
|
||||
*
|
||||
* Scientific computing from scratch, with feeling.
|
||||
*
|
||||
* Copyright (c) 2026 Michelle Bausager
|
||||
*
|
||||
* This file is part of PANIC.
|
||||
*
|
||||
* PANIC is free software licensed under the GNU General Public License v3.0 or later.
|
||||
* You may redistribute and/or modify it under the terms of the GPL.
|
||||
*
|
||||
* PANIC is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY;
|
||||
* without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.
|
||||
* See the LICENSE file for the full license text.
|
||||
*
|
||||
* SPDX-License-Identifier: GPL-3.0-or-later
|
||||
*
|
||||
*++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
|
||||
*
|
||||
* Project Name: PANIC
|
||||
* Module Name: math
|
||||
* File Name: abs.cpp
|
||||
* Revision: 0.1.0
|
||||
* Date: 07-08-2026
|
||||
* Author: Michelle Bausager
|
||||
*
|
||||
* Description:
|
||||
* Functions to calculate the abs of value
|
||||
*
|
||||
*++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++*/
|
||||
#pragma once
|
||||
|
||||
#include <tensor/vector.hpp> // for panic::vector
|
||||
#include <tensor/matrix.hpp> // for panic::matrix
|
||||
|
||||
namespace panic{
|
||||
namespace math{
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* @brief calculates the abs of a value.
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* abs(x, y)
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param x Value to take the abs of.
|
||||
* @param y Result.
|
||||
*
|
||||
*
|
||||
* @note This function is omp-friendly.
|
||||
*/
|
||||
template <typename T>
|
||||
bool abs(const T x, T& y);
|
||||
|
||||
|
||||
/**
|
||||
* @brief calculates the abs of a value.
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* result = abs(k)
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param x Value to take the abs of.
|
||||
*
|
||||
* @return The calculated value
|
||||
*
|
||||
* @note This function is omp-friendly.
|
||||
*/
|
||||
template <typename T>
|
||||
T abs(const T x);
|
||||
|
||||
|
||||
/**
|
||||
* @brief Calculates the abs elementwise in a vector
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* c[i] = abs(a[i])
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param a Input vector.
|
||||
* @param c Output vector. Resized to match @p a.
|
||||
*
|
||||
* @return true if @p c was resized and filled successfully.
|
||||
* @return false if resizing @p c failed.
|
||||
*
|
||||
* @note This overload writes the result into an existing vector to avoid
|
||||
* unnecessary temporary allocations.
|
||||
*/
|
||||
template <typename T>
|
||||
bool abs(const panic::tensor::vector<T>& a, panic::tensor::vector<T>& c);
|
||||
|
||||
/**
|
||||
* @brief Calculates the abs elementwise in a vector
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* result[i] = abs(a[i])
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param a Input vector.
|
||||
*
|
||||
* @return A new vector containing the result.
|
||||
* @return An empty vector if the operation fails.
|
||||
*
|
||||
* @note This overload is convenient, but may allocate a new vector.
|
||||
*/
|
||||
template <typename T>
|
||||
panic::tensor::vector<T> abs(const panic::tensor::vector<T>& a);
|
||||
|
||||
/**
|
||||
* @brief Calculates the abs elementwise of a matrix
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* C(i,j) = abs(A(i,j))
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param A Input matrix.
|
||||
* @param C Output Matrix. Resized to match @p A.
|
||||
*
|
||||
* @return true if @p C was resized and filled successfully.
|
||||
* @return false if resizing @p C failed.
|
||||
*
|
||||
* @note This overload writes the result into an existing vector to avoid
|
||||
* unnecessary temporary allocations.
|
||||
*/
|
||||
template <typename T>
|
||||
bool abs(const panic::tensor::matrix<T>& A, panic::tensor::matrix<T>& C);
|
||||
|
||||
/**
|
||||
* @brief Returns the calculated abs elementwise of the matrix
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* result(i,j) = abs(A(i,j))
|
||||
* @endcode
|
||||
*
|
||||
* @tparam T Numeric element type.
|
||||
* @param A Input matrix.
|
||||
*
|
||||
* @return A new matrix containing the result.
|
||||
* @return An empty matrix if the operation fails.
|
||||
*
|
||||
* @note This overload is convenient, but may allocate a new vector.
|
||||
*/
|
||||
template <typename T>
|
||||
panic::tensor::matrix<T> abs(const panic::tensor::matrix<T>& A);
|
||||
|
||||
} // namespace math
|
||||
} // namespace panic
|
||||
@@ -71,7 +71,12 @@ struct layer_dense : trainable_layer{
|
||||
* @param neurons Amount of neurons in the layer
|
||||
*
|
||||
*/
|
||||
layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons);
|
||||
layer_dense(panic::types::uint_t input_size,
|
||||
panic::types::uint_t neurons,
|
||||
panic::types::real_t weight_regularizer_l1 = 0,
|
||||
panic::types::real_t weight_regularizer_l2 = 0,
|
||||
panic::types::real_t bias_regularizer_l1 = 0,
|
||||
panic::types::real_t bias_regularizer_l2 = 0);
|
||||
|
||||
/**
|
||||
* @brief Default de-constructor
|
||||
|
||||
@@ -63,6 +63,12 @@ struct trainable_layer:layer{
|
||||
panic::tensor::real_matrix dweights;
|
||||
panic::tensor::real_vector dbiases;
|
||||
|
||||
panic::types::real_t weight_regularizer_l1;
|
||||
panic::types::real_t weight_regularizer_l2;
|
||||
|
||||
panic::types::real_t bias_regularizer_l1;
|
||||
panic::types::real_t bias_regularizer_l2;
|
||||
|
||||
/**
|
||||
* @brief Previous parameter updates used by momentum SGD.
|
||||
*
|
||||
@@ -82,6 +88,9 @@ struct trainable_layer:layer{
|
||||
panic::tensor::real_vector bias_cache;
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
virtual ~trainable_layer() = default;
|
||||
|
||||
|
||||
|
||||
@@ -39,6 +39,7 @@
|
||||
#include <tensor/matrix.hpp> // panic::tensor::real_matrix (uint_matrix, int_matrix)
|
||||
#include <tensor/vector.hpp>
|
||||
|
||||
#include <neural_network/layer/trainable_layer.hpp>
|
||||
|
||||
namespace panic{
|
||||
namespace neural_network{
|
||||
@@ -73,6 +74,11 @@ struct loss{
|
||||
*/
|
||||
panic::types::real_t data_loss = 0;
|
||||
|
||||
/**
|
||||
* @brief Regularization loss over a layer.
|
||||
*/
|
||||
panic::types::real_t regularization_loss_value = 0;
|
||||
|
||||
/**
|
||||
* @brief Gradient with respect to the loss input.
|
||||
*
|
||||
@@ -172,6 +178,15 @@ struct loss{
|
||||
const panic::tensor::real_matrix& y_pred,
|
||||
const panic::tensor::real_matrix& y_true);
|
||||
|
||||
|
||||
/**
|
||||
* @brief Caclculates the regularization loss of a trainable layer
|
||||
*
|
||||
*/
|
||||
bool regularization_loss(const trainable_layer& layer);
|
||||
|
||||
|
||||
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -205,10 +205,12 @@ struct model{
|
||||
*
|
||||
* @note This function is convenient, but it allocates a new layer.
|
||||
*/
|
||||
bool add_layer_dense(
|
||||
panic::types::uint_t input_size,
|
||||
panic::types::uint_t neuron_count
|
||||
);
|
||||
bool add_layer_dense(panic::types::uint_t input_size,
|
||||
panic::types::uint_t neurons,
|
||||
panic::types::real_t weight_regularizer_l1 = 0,
|
||||
panic::types::real_t weight_regularizer_l2 = 0,
|
||||
panic::types::real_t bias_regularizer_l1 = 0,
|
||||
panic::types::real_t bias_regularizer_l2 = 0);
|
||||
|
||||
/**
|
||||
* @brief Adds a activation ReLU layer to the model.
|
||||
@@ -412,6 +414,18 @@ struct model{
|
||||
*/
|
||||
bool optimize();
|
||||
|
||||
/**
|
||||
* @brief Calculates regulaization for parameters on trainable layers
|
||||
*
|
||||
* Computes:
|
||||
* @code
|
||||
* model.calculate_regularization_loss()
|
||||
* @endcode
|
||||
*
|
||||
* @return true if optimization is done correctly.
|
||||
*
|
||||
*/
|
||||
bool calculate_regularization_loss(panic::types::real_t& regularization_loss);
|
||||
|
||||
|
||||
/**
|
||||
|
||||
@@ -706,9 +706,16 @@ int main(void) {
|
||||
panic::tensor::real_matrix X;
|
||||
panic::tensor::uint_vector y;
|
||||
|
||||
panic::types::uint_t samples = 100;
|
||||
panic::types::uint_t samples = 1000;
|
||||
panic::types::uint_t classes = 3;
|
||||
|
||||
panic::types::uint_t layer_size = 256;
|
||||
|
||||
|
||||
panic::types::real_t weight_regularizer_l1 = 0;
|
||||
panic::types::real_t weight_regularizer_l2 = 5e-4;
|
||||
panic::types::real_t bias_regularizer_l1 = 0;
|
||||
panic::types::real_t bias_regularizer_l2 = 5e-4;
|
||||
|
||||
// create spiral data
|
||||
panic::neural_network::spiral_data(samples, classes, X, y);
|
||||
@@ -718,7 +725,11 @@ int main(void) {
|
||||
panic::neural_network::model mymodel;
|
||||
|
||||
// Create Dense layer with 2 input features and 3 output values
|
||||
if (!mymodel.add_layer_dense(2,64)){
|
||||
if (!mymodel.add_layer_dense(2,layer_size,
|
||||
weight_regularizer_l1,
|
||||
weight_regularizer_l2,
|
||||
bias_regularizer_l1,
|
||||
bias_regularizer_l2)){
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -727,8 +738,13 @@ int main(void) {
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Create a second dense layer with 3 inputs and 3 outputs
|
||||
if (!mymodel.add_layer_dense(64, 3)){
|
||||
if (!mymodel.add_layer_dense(layer_size, 3,
|
||||
weight_regularizer_l1,
|
||||
weight_regularizer_l2,
|
||||
bias_regularizer_l1,
|
||||
bias_regularizer_l2)){
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -769,7 +785,7 @@ int main(void) {
|
||||
}
|
||||
*/
|
||||
|
||||
panic::types::real_t learning_rate = 0.02;
|
||||
panic::types::real_t learning_rate = 0.001;
|
||||
panic::types::real_t decay = 1e-5;
|
||||
panic::types::real_t epsilon = 1e-7;
|
||||
panic::types::real_t beta_1 = 0.9;
|
||||
@@ -794,5 +810,11 @@ int main(void) {
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,255 @@
|
||||
/**++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
|
||||
*
|
||||
* PANIC
|
||||
* Portable Algorithms and Numerics In C++
|
||||
*
|
||||
* Scientific computing from scratch, with feeling.
|
||||
*
|
||||
* Copyright (c) 2026 Michelle Bausager
|
||||
*
|
||||
* This file is part of PANIC.
|
||||
*
|
||||
* PANIC is free software licensed under the GNU General Public License v3.0 or later.
|
||||
* You may redistribute and/or modify it under the terms of the GPL.
|
||||
*
|
||||
* PANIC is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY;
|
||||
* without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.
|
||||
* See the LICENSE file for the full license text.
|
||||
*
|
||||
* SPDX-License-Identifier: GPL-3.0-or-later
|
||||
*
|
||||
*++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
|
||||
*
|
||||
* Project Name: PANIC
|
||||
* Module Name: math
|
||||
* File Name: abs.cpp
|
||||
* Revision: 0.1.0
|
||||
* Date: 07-08-2026
|
||||
* Author: Michelle Bausager
|
||||
*
|
||||
* Description:
|
||||
* Functions to calculate the sqrt of numbers
|
||||
*
|
||||
*++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++*/
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INCLUDE DESCRIPTION
|
||||
//-----------------------------------------------------------------------------------------------------
|
||||
|
||||
#include <math/abs.hpp>
|
||||
#include <config/omp.hpp>
|
||||
#include <config/types.hpp>
|
||||
|
||||
#include <tensor/vector.hpp> // for panic::vector
|
||||
#include <tensor/matrix.hpp> // for panic::matrix
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// PRIVATE CONSTANTS
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
/**
|
||||
* @brief Minimum number of element operations before using the OpenMP-enabled loop.
|
||||
*
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t abs_omp_min_work = 250;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INPLEMENTATION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
|
||||
namespace panic {
|
||||
namespace math {
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
bool abs(const T x, T& y){
|
||||
|
||||
if (x < static_cast<T>(0)){
|
||||
y = -x;
|
||||
}
|
||||
else{
|
||||
y = x;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template bool abs(const panic::types::real_t x,
|
||||
panic::types::real_t& y
|
||||
);
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
//
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
T abs(const T x){
|
||||
T y;
|
||||
if (! abs(x,y)){
|
||||
return T();
|
||||
}
|
||||
|
||||
return y;
|
||||
}
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template panic::types::real_t abs(const panic::types::real_t x
|
||||
);
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
// Calculates the abs elementwise of a vector
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
bool abs(const panic::tensor::vector<T>& a, panic::tensor::vector<T>& c){
|
||||
|
||||
if (!c.resize(a.size())){
|
||||
return false;
|
||||
}
|
||||
|
||||
bool valid = true;
|
||||
|
||||
// valid remains true only if every abs is valid.
|
||||
PANIC_OMP_PARALLEL_FOR_REDUCTION_IF( a.size() > abs_omp_min_work, &&, valid )
|
||||
for (panic::types::uint_t i = 0; i < a.size(); ++i){
|
||||
const bool nonzero = abs(a[i], c[i]);
|
||||
valid = valid && nonzero;
|
||||
}
|
||||
|
||||
return valid;
|
||||
}
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template bool abs<panic::types::real_t>(const panic::tensor::vector<panic::types::real_t>& a,
|
||||
panic::tensor::vector<panic::types::real_t>& c
|
||||
);
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
// Calculates the abs elementwise for a vector
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
panic::tensor::vector<T> abs(const panic::tensor::vector<T>& a){
|
||||
panic::tensor::vector<T> c(a.size());
|
||||
|
||||
if (!abs(a, c)){
|
||||
return panic::tensor::vector<T>();
|
||||
}
|
||||
|
||||
return c;
|
||||
}
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template panic::tensor::vector<panic::types::real_t>
|
||||
abs(const panic::tensor::vector<panic::types::real_t>& a
|
||||
);
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
// calculates the natrual abs elementwise of a matrix
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
bool abs(const panic::tensor::matrix<T>& A, panic::tensor::matrix<T>& C){
|
||||
|
||||
panic::types::uint_t rows = A.rows();
|
||||
panic::types::uint_t cols = A.cols();
|
||||
panic::types::uint_t work = rows*cols;
|
||||
|
||||
if ( !C.resize(rows, cols) ){
|
||||
return false;
|
||||
}
|
||||
|
||||
bool valid = true;
|
||||
|
||||
// valid remains true only if every abs is valid.
|
||||
PANIC_OMP_PARALLEL_FOR_IF(work > abs_omp_min_work)
|
||||
for (panic::types::uint_t i = 0; i < rows; ++i){
|
||||
for (panic::types::uint_t j = 0; j < cols; ++j){
|
||||
const bool nonzero = abs(A(i,j), C(i,j));
|
||||
valid = valid && nonzero;
|
||||
}
|
||||
}
|
||||
|
||||
return valid;
|
||||
}
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template bool abs(const panic::tensor::matrix<panic::types::real_t>& A,
|
||||
panic::tensor::matrix<panic::types::real_t>& C
|
||||
);
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::math::abs
|
||||
//
|
||||
// Description:
|
||||
// Calculates the natrual abs element-wise of a matrix
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template <typename T>
|
||||
panic::tensor::matrix<T> abs(const panic::tensor::matrix<T>& A){
|
||||
panic::tensor::matrix<T> C;
|
||||
|
||||
if (!abs(A, C)){
|
||||
return panic::tensor::matrix<T>();
|
||||
}
|
||||
|
||||
return C;
|
||||
}
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// EXPLICIT TEMPLATE INSTANTIATION
|
||||
//
|
||||
// The implementation is in this .cpp file.
|
||||
// Build the overload for the official PANIC numeric types.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
template panic::tensor::matrix<panic::types::real_t>
|
||||
abs(const panic::tensor::matrix<panic::types::real_t>& A
|
||||
);
|
||||
|
||||
|
||||
} // namespace math
|
||||
} // namespace panic
|
||||
@@ -55,7 +55,7 @@
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t layer_dense_omp_min_size = 500;
|
||||
static const panic::types::uint_t layer_dense_omp_min_size = 250;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
// INPLEMENTATION
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -75,6 +75,10 @@ layer_dense::layer_dense() {
|
||||
biases.resize(0);
|
||||
dweights.resize(0,0);
|
||||
dbiases.resize(0);
|
||||
weight_regularizer_l1 = 0;
|
||||
weight_regularizer_l2 = 0;
|
||||
bias_regularizer_l1 = 0;
|
||||
bias_regularizer_l2 = 0;
|
||||
}
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -83,7 +87,14 @@ layer_dense::layer_dense() {
|
||||
// Description:
|
||||
// Creates an empty layer with neurons.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t neurons) {
|
||||
layer_dense::layer_dense(panic::types::uint_t input_size,
|
||||
panic::types::uint_t neurons,
|
||||
panic::types::real_t weight_regularizer_l1,
|
||||
panic::types::real_t weight_regularizer_l2,
|
||||
panic::types::real_t bias_regularizer_l1,
|
||||
panic::types::real_t bias_regularizer_l2) {
|
||||
|
||||
|
||||
weights.resize(input_size, neurons);
|
||||
panic::random::uniform(weights, static_cast<panic::types::real_t>(-1), static_cast<panic::types::real_t>(1));
|
||||
panic::math::mul(weights, static_cast<panic::types::real_t>(0.01), weights);
|
||||
@@ -91,6 +102,12 @@ layer_dense::layer_dense(panic::types::uint_t input_size, panic::types::uint_t n
|
||||
biases.resize(neurons);
|
||||
biases.fill(0);
|
||||
|
||||
this->weight_regularizer_l1 = weight_regularizer_l1;
|
||||
this->weight_regularizer_l2 = weight_regularizer_l2;
|
||||
this-> bias_regularizer_l1 = bias_regularizer_l1;
|
||||
this-> bias_regularizer_l2 = bias_regularizer_l2;
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -135,10 +152,68 @@ bool layer_dense::forward(const panic::tensor::real_matrix& input_data){
|
||||
bool layer_dense::backward(const panic::tensor::real_matrix& dvalues){
|
||||
|
||||
|
||||
|
||||
// Gradients on parameters
|
||||
dweights = panic::math::matmul(panic::math::transpose(inputs), dvalues);
|
||||
dbiases = panic::math::sum_colwise(dvalues);
|
||||
|
||||
panic::types::uint_t weight_work = dweights.rows()*dweights.cols();
|
||||
|
||||
if (weight_regularizer_l1 > static_cast<panic::types::real_t>(0)){
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(weight_work > layer_dense_omp_min_size)
|
||||
for (panic::types::uint_t i = 0; i < dweights.rows(); ++i){
|
||||
for (panic::types::uint_t j = 0; j < dweights.cols(); ++j){
|
||||
|
||||
if (weights(i,j) < static_cast<panic::types::real_t>(0)){
|
||||
dweights(i,j) += weight_regularizer_l1 * -1;
|
||||
}
|
||||
else{
|
||||
dweights(i,j) += weight_regularizer_l1 * 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (weight_regularizer_l2 > static_cast<panic::types::real_t>(0)){
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(weight_work > layer_dense_omp_min_size)
|
||||
for (panic::types::uint_t i = 0; i < dweights.rows(); ++i){
|
||||
for (panic::types::uint_t j = 0; j < dweights.cols(); ++j){
|
||||
|
||||
dweights(i,j) += static_cast<panic::types::real_t>(2) * weight_regularizer_l2 * weights(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
if (bias_regularizer_l1 > static_cast<panic::types::real_t>(0)){
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(dbiases.size() > layer_dense_omp_min_size)
|
||||
for (panic::types::uint_t i = 0; i < dbiases.size(); ++i){
|
||||
|
||||
if (biases[i] < static_cast<panic::types::real_t>(0)){
|
||||
dbiases[i] += bias_regularizer_l1 * -1;
|
||||
}
|
||||
else{
|
||||
dbiases[i] += bias_regularizer_l1 * 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (bias_regularizer_l2 > static_cast<panic::types::real_t>(0)){
|
||||
|
||||
PANIC_OMP_PARALLEL_FOR_IF(dbiases.size() > layer_dense_omp_min_size)
|
||||
for (panic::types::uint_t i = 0; i < dbiases.size(); ++i){
|
||||
|
||||
dbiases[i] += static_cast<panic::types::real_t>(2) * bias_regularizer_l2 * biases[i];
|
||||
}
|
||||
}
|
||||
|
||||
// Gradients on values
|
||||
dinputs = panic::math::matmul(dvalues, panic::math::transpose(weights));
|
||||
|
||||
|
||||
@@ -41,6 +41,10 @@
|
||||
#include <tensor/vector.hpp>
|
||||
|
||||
#include <math/mean.hpp>
|
||||
#include <math/abs.hpp>
|
||||
#include <math/add.hpp>
|
||||
#include <math/mul.hpp>
|
||||
#include <math/sum.hpp>
|
||||
|
||||
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
@@ -52,7 +56,7 @@
|
||||
* Small vectors and matrices are kept serial because the overhead of starting
|
||||
* worker threads can be larger than the work itself.
|
||||
*/
|
||||
static const panic::types::uint_t loss_omp_min_size = 500;
|
||||
static const panic::types::uint_t loss_omp_min_size = 250;
|
||||
//---------------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
|
||||
@@ -115,6 +119,43 @@ bool loss::calculate(
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::loss::regularization_loss
|
||||
//
|
||||
// Description:
|
||||
// Calculates the regularization loss of a trainable layer.
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool loss::regularization_loss(const trainable_layer& layer){
|
||||
|
||||
regularization_loss_value = 0;
|
||||
|
||||
if (layer.weight_regularizer_l1 > 0){
|
||||
regularization_loss_value += panic::math::sum(panic::math::abs(layer.weights))*layer.weight_regularizer_l1;
|
||||
}
|
||||
|
||||
if (layer.weight_regularizer_l2 > 0){
|
||||
regularization_loss_value += panic::math::sum(panic::math::mul(layer.weights,layer.weights))*layer.weight_regularizer_l2;
|
||||
}
|
||||
|
||||
if (layer.bias_regularizer_l1 > 0){
|
||||
regularization_loss_value += panic::math::sum(panic::math::abs(layer.biases))*layer.bias_regularizer_l1;
|
||||
}
|
||||
|
||||
if (layer.bias_regularizer_l2 > 0){
|
||||
regularization_loss_value += panic::math::sum(panic::math::mul(layer.biases,layer.biases))*layer.bias_regularizer_l2;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::loss::forward
|
||||
//
|
||||
|
||||
@@ -251,12 +251,20 @@ bool model::add_trainable_layer(trainable_layer* new_layer){
|
||||
// Example:
|
||||
// model.add_layer_dense(100, 64);
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::add_layer_dense(
|
||||
panic::types::uint_t input_size,
|
||||
panic::types::uint_t neuron_count){
|
||||
bool model::add_layer_dense(panic::types::uint_t input_size,
|
||||
panic::types::uint_t neurons,
|
||||
panic::types::real_t weight_regularizer_l1,
|
||||
panic::types::real_t weight_regularizer_l2,
|
||||
panic::types::real_t bias_regularizer_l1,
|
||||
panic::types::real_t bias_regularizer_l2){
|
||||
|
||||
// Create the dense layer.
|
||||
layer_dense* new_layer = new layer_dense(input_size, neuron_count);
|
||||
layer_dense* new_layer = new layer_dense(input_size,
|
||||
neurons,
|
||||
weight_regularizer_l1,
|
||||
weight_regularizer_l2,
|
||||
bias_regularizer_l1,
|
||||
bias_regularizer_l2);
|
||||
|
||||
if (new_layer == 0){
|
||||
return false;
|
||||
@@ -709,8 +717,6 @@ bool model::optimize(){
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
|
||||
if (! optimizer_function->update_params(*trainable_layers[i]) ){
|
||||
return false;
|
||||
}
|
||||
@@ -726,6 +732,41 @@ bool model::optimize(){
|
||||
}
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::calculate_regularization_loss
|
||||
//
|
||||
// Description:
|
||||
// Calculates regulaization for parameters on trainable layers
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
bool model::calculate_regularization_loss(panic::types::real_t& regularization_loss){
|
||||
|
||||
regularization_loss = 0;
|
||||
|
||||
if (loss_function == 0){
|
||||
return false;
|
||||
}
|
||||
|
||||
for(panic::types::uint_t i = 0; i < trainable_layer_count; ++i){
|
||||
|
||||
// Validate the stored layer pointers before using them.
|
||||
if (trainable_layers[i] == 0){
|
||||
return false;
|
||||
}
|
||||
|
||||
if (! loss_function->regularization_loss(*trainable_layers[i])){
|
||||
return false;
|
||||
}
|
||||
|
||||
regularization_loss += loss_function->regularization_loss_value;
|
||||
|
||||
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
//--------------------------------------------------------------------------------------------------------------------------
|
||||
// Function Name : panic::neural_network::model::train
|
||||
@@ -745,13 +786,21 @@ bool model::train(const panic::tensor::real_matrix& X_train,
|
||||
for (panic::types::uint_t epoch = 0; epoch < epochs+1; ++epoch){
|
||||
|
||||
|
||||
forward(X_train);
|
||||
|
||||
if (!forward(X_train)){
|
||||
return false;
|
||||
}
|
||||
// Calculate loss from the model's final output.
|
||||
if (!loss_function->calculate(outputs, y_train)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// Caluclates regula
|
||||
panic::types::real_t regularization_loss = 0;
|
||||
|
||||
if (! calculate_regularization_loss(regularization_loss)){
|
||||
return false;
|
||||
}
|
||||
|
||||
// There must be one output row for every target label.
|
||||
if (outputs.rows() != y_train.size() || outputs.cols() == static_cast<panic::types::uint_t>(0)){
|
||||
return false;
|
||||
@@ -780,6 +829,8 @@ bool model::train(const panic::tensor::real_matrix& X_train,
|
||||
std::cout << "Epoch: " << epoch;
|
||||
std::cout << " lr: " << optimizer_function->current_learning_rate;
|
||||
std::cout << " data loss: " << loss_function->data_loss;
|
||||
std::cout << " reg loss: " << regularization_loss;
|
||||
std::cout << " total loss: " << loss_function->data_loss + regularization_loss;
|
||||
std::cout << " acc: " << accuracy << std::endl;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user