Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 9 additions & 9 deletions src/NeuralNetworks/ExtremeLearningMachine.cs
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ public class ExtremeLearningMachine<T> : NeuralNetworkBase<T>
/// where each team member will be randomly assigned what to look for.
/// </para>
/// </remarks>
public ExtremeLearningMachine(NeuralNetworkArchitecture<T> architecture, int hiddenLayerSize, ILossFunction<T>? lossFunction = null)
public ExtremeLearningMachine(NeuralNetworkArchitecture<T> architecture, int hiddenLayerSize, ILossFunction<T>? lossFunction = null)
: base(architecture, lossFunction ?? NeuralNetworkHelper<T>.GetDefaultLossFunction(architecture.TaskType))
{
_hiddenLayerSize = hiddenLayerSize;
Expand Down Expand Up @@ -225,8 +225,8 @@ public override void Train(Tensor<T> input, Tensor<T> expectedOutput)
}

// STEP 2: Calculate the optimal output weights using pseudo-inverse
// We'll use the Moore-Penrose pseudoinverse: OutputWeights = (H? × T)
// where H? is the pseudoinverse of H (hidden activations) and T is the target output
// We'll use the Moore-Penrose pseudoinverse: OutputWeights = (H+ × T)
// where H+ is the pseudoinverse of H (hidden activations) and T is the target output

// Convert hidden activations and expected output to matrices for the calculation
Matrix<T> H = hiddenActivations.ConvertToMatrix();
Expand Down Expand Up @@ -257,7 +257,7 @@ public override void Train(Tensor<T> input, Tensor<T> expectedOutput)
/// This method calculates the Moore-Penrose pseudoinverse of a matrix, which is a generalization of the matrix inverse
/// for non-square matrices. The pseudoinverse is used in the ELM training algorithm to analytically solve
/// for the optimal output layer weights. For computational efficiency, this implementation uses the formula:
/// A? = (A^T × A)^(-1) × A^T for full column rank matrices.
/// A+ = (A^T × A)^(-1) × A^T for full column rank matrices.
/// </para>
/// <para><b>For Beginners:</b> This calculates a special type of matrix inverse used in ELM training.
///
Expand All @@ -268,7 +268,7 @@ public override void Train(Tensor<T> input, Tensor<T> expectedOutput)
/// </remarks>
private Matrix<T> CalculatePseudoInverse(Matrix<T> matrix)
{
// Calculate the pseudoinverse using the formula: A? = (A^T × A)^(-1) × A^T
// Calculate the pseudoinverse using the formula: A+ = (A^T × A)^(-1) × A^T
// This works well for matrices with full column rank, which is common in ELMs with
// more data samples than hidden neurons

Expand All @@ -288,7 +288,7 @@ private Matrix<T> CalculatePseudoInverse(Matrix<T> matrix)

// Note: In a production implementation, you might want to use singular value decomposition (SVD)
// for better numerical stability, or use a regularized version like:
// A? = (A^T × A + ?I)^(-1) × A^T where ? is a small regularization parameter
// A+ = (A^T × A + λI)^(-1) × A^T where λ is a small regularization parameter
}

/// <summary>
Expand Down Expand Up @@ -442,7 +442,7 @@ protected override void DeserializeNetworkSpecificData(BinaryReader reader)
/// <para>
/// This method implements a regularized version of the ELM training algorithm. It adds a regularization term
/// to the pseudoinverse calculation, which helps prevent overfitting. The formula becomes:
/// OutputWeights = (H^T * H + ?I)^(-1) * H^T * T, where ? is the regularization factor, I is the identity matrix,
/// OutputWeights = (H^T * H + λI)^(-1) * H^T * T, where λ is the regularization factor, I is the identity matrix,
/// H is the hidden layer activations, and T is the target output.
/// </para>
/// <para><b>For Beginners:</b> This is a more robust training method that helps prevent overfitting.
Expand Down Expand Up @@ -479,14 +479,14 @@ public void TrainWithRegularization(Tensor<T> input, Tensor<T> expectedOutput, d
Matrix<T> H = hiddenActivations.ConvertToMatrix();
Matrix<T> T = expectedOutput.ConvertToMatrix();

// Calculate regularized pseudoinverse: (H^T * H + ?I)^(-1) * H^T
// Calculate regularized pseudoinverse: (H^T * H + λI)^(-1) * H^T
Matrix<T> transposeH = H.Transpose();
Matrix<T> hTh = transposeH.Multiply(H);

// Create identity matrix for regularization
Matrix<T> identity = Matrix<T>.CreateIdentity(hTh.Rows);

// Apply regularization: hTh + ?I
// Apply regularization: hTh + λI
T regFactor = NumOps.FromDouble(regularizationFactor);
for (int i = 0; i < identity.Rows; i++)
{
Expand Down
24 changes: 21 additions & 3 deletions src/NeuralNetworks/HopfieldNetwork.cs
Original file line number Diff line number Diff line change
Expand Up @@ -200,7 +200,7 @@ private void InitializeWeights()
/// better recall. This process is different from training in most neural networks because:
/// - It happens in one pass, not through repeated iterations
/// - It doesn't use backpropagation or gradients
/// - It has limited capacity (can only store approximately 0.14 � network size patterns reliably)
/// - It has limited capacity (can only store approximately 0.14 * network size patterns reliably)
/// </para>
/// </remarks>
public void Train(List<Vector<T>> patterns)
Expand Down Expand Up @@ -314,8 +314,26 @@ public Vector<T> Recall(Vector<T> input, int maxIterations = 100)
/// </remarks>
public override void UpdateParameters(Vector<T> parameters)
{
// Hopfield networks typically don't use gradient-based updates
throw new InvalidOperationException("Hopfield networks do not support gradient-based parameter updates. Use the Train(List<Vector<T>> patterns) method instead.");
// Number of off-diagonal entries in a symmetric NxN matrix
int expectedLength = (_size * (_size - 1)) / 2;

if (parameters.Length != expectedLength)
{
throw new ArgumentException($"Parameter vector length mismatch. Expected {expectedLength} parameters but got {parameters.Length}.", nameof(parameters));
}

int paramIndex = 0;
// Fill upper triangle and mirror to lower triangle; keep diagonal zero
for (int i = 0; i < _size; i++)
{
_weights[i, i] = NumOps.Zero;
for (int j = i + 1; j < _size; j++)
{
var w = parameters[paramIndex++];
_weights[i, j] = w;
_weights[j, i] = w;
}
}
}

/// <summary>
Expand Down
24 changes: 19 additions & 5 deletions src/NeuralNetworks/SelfOrganizingMap.cs
Original file line number Diff line number Diff line change
Expand Up @@ -141,7 +141,7 @@ public class SelfOrganizingMap<T> : NeuralNetworkBase<T>
/// When creating a new SOM:
/// - The architecture tells us how many input dimensions we have (how many attributes each data point has)
/// - The architecture also suggests how many total positions we want on our map
/// - The constructor tries to make the map as square as possible (e.g., 10×10 rather than 5×20)
/// - The constructor tries to make the map as square as possible (e.g., 10×10 rather than 5×20)
/// - It may adjust the total map size slightly to make a perfect square if needed
Comment thread
coderabbitai[bot] marked this conversation as resolved.
///
/// Once the dimensions are set, it creates weight values for each position on the map.
Expand Down Expand Up @@ -331,7 +331,7 @@ private int FindBestMatchingUnit(Vector<T> input)
/// For example, if comparing two books with attributes for page count and publication year:
/// - Book 1: 300 pages, published in 2010
/// - Book 2: 400 pages, published in 2020
/// - The calculation would be: v[(300-400)² + (2010-2020)²] = v(10,100 + 100) = v10,200 ˜ 101
/// - The calculation would be: sqrt[(300-400)² + (2010-2020)²] = sqrt(10,100 + 100) = sqrt(10,200) ≈ 101
///
/// A smaller distance means the data points are more similar.
/// </para>
Expand Down Expand Up @@ -538,7 +538,7 @@ private T CalculateInfluence(T distance, T radius)
/// - Changes are proportional to learning rate and influence
/// - The BMU moves more than distant positions
///
/// The formula (learningRate × influence × (input - weight)) moves each weight
/// The formula (learningRate * influence * (input - weight)) moves each weight
/// some fraction of the way toward matching the corresponding input value.
/// </para>
/// </remarks>
Expand Down Expand Up @@ -576,8 +576,22 @@ private Vector<T> CalculateWeightDelta(Vector<T> input, Vector<T> weight, T lear
/// </remarks>
public override void UpdateParameters(Vector<T> parameters)
{
// This method is not typically used in SOMs
throw new InvalidOperationException("UpdateParameters is not implemented for Self-Organizing Maps. SOMs use competitive learning and update weights through the Train method instead.");
int expectedLength = (_mapWidth * _mapHeight) * _inputDimension;

if (parameters.Length != expectedLength)
{
throw new ArgumentException($"Parameter vector length mismatch. Expected {expectedLength} parameters but got {parameters.Length}.", nameof(parameters));
}

int paramIndex = 0;

for (int i = 0; i < _mapWidth * _mapHeight; i++)
{
for (int j = 0; j < _inputDimension; j++)
{
_weights[i, j] = parameters[paramIndex++];
}
}
}

/// <summary>
Expand Down
Loading