diff --git a/src/NeuralNetworks/ExtremeLearningMachine.cs b/src/NeuralNetworks/ExtremeLearningMachine.cs index e7e82bfda6..fb4a182b0b 100644 --- a/src/NeuralNetworks/ExtremeLearningMachine.cs +++ b/src/NeuralNetworks/ExtremeLearningMachine.cs @@ -72,7 +72,7 @@ public class ExtremeLearningMachine : NeuralNetworkBase /// where each team member will be randomly assigned what to look for. /// /// - public ExtremeLearningMachine(NeuralNetworkArchitecture architecture, int hiddenLayerSize, ILossFunction? lossFunction = null) + public ExtremeLearningMachine(NeuralNetworkArchitecture architecture, int hiddenLayerSize, ILossFunction? lossFunction = null) : base(architecture, lossFunction ?? NeuralNetworkHelper.GetDefaultLossFunction(architecture.TaskType)) { _hiddenLayerSize = hiddenLayerSize; @@ -225,8 +225,8 @@ public override void Train(Tensor input, Tensor expectedOutput) } // STEP 2: Calculate the optimal output weights using pseudo-inverse - // We'll use the Moore-Penrose pseudoinverse: OutputWeights = (H? × T) - // where H? is the pseudoinverse of H (hidden activations) and T is the target output + // We'll use the Moore-Penrose pseudoinverse: OutputWeights = (H+ × T) + // where H+ is the pseudoinverse of H (hidden activations) and T is the target output // Convert hidden activations and expected output to matrices for the calculation Matrix H = hiddenActivations.ConvertToMatrix(); @@ -257,7 +257,7 @@ public override void Train(Tensor input, Tensor expectedOutput) /// This method calculates the Moore-Penrose pseudoinverse of a matrix, which is a generalization of the matrix inverse /// for non-square matrices. The pseudoinverse is used in the ELM training algorithm to analytically solve /// for the optimal output layer weights. For computational efficiency, this implementation uses the formula: - /// A? = (A^T × A)^(-1) × A^T for full column rank matrices. + /// A+ = (A^T × A)^(-1) × A^T for full column rank matrices. /// /// For Beginners: This calculates a special type of matrix inverse used in ELM training. /// @@ -268,7 +268,7 @@ public override void Train(Tensor input, Tensor expectedOutput) /// private Matrix CalculatePseudoInverse(Matrix matrix) { - // Calculate the pseudoinverse using the formula: A? = (A^T × A)^(-1) × A^T + // Calculate the pseudoinverse using the formula: A+ = (A^T × A)^(-1) × A^T // This works well for matrices with full column rank, which is common in ELMs with // more data samples than hidden neurons @@ -288,7 +288,7 @@ private Matrix CalculatePseudoInverse(Matrix matrix) // Note: In a production implementation, you might want to use singular value decomposition (SVD) // for better numerical stability, or use a regularized version like: - // A? = (A^T × A + ?I)^(-1) × A^T where ? is a small regularization parameter + // A+ = (A^T × A + λI)^(-1) × A^T where λ is a small regularization parameter } /// @@ -442,7 +442,7 @@ protected override void DeserializeNetworkSpecificData(BinaryReader reader) /// /// This method implements a regularized version of the ELM training algorithm. It adds a regularization term /// to the pseudoinverse calculation, which helps prevent overfitting. The formula becomes: - /// OutputWeights = (H^T * H + ?I)^(-1) * H^T * T, where ? is the regularization factor, I is the identity matrix, + /// OutputWeights = (H^T * H + λI)^(-1) * H^T * T, where λ is the regularization factor, I is the identity matrix, /// H is the hidden layer activations, and T is the target output. /// /// For Beginners: This is a more robust training method that helps prevent overfitting. @@ -479,14 +479,14 @@ public void TrainWithRegularization(Tensor input, Tensor expectedOutput, d Matrix H = hiddenActivations.ConvertToMatrix(); Matrix T = expectedOutput.ConvertToMatrix(); - // Calculate regularized pseudoinverse: (H^T * H + ?I)^(-1) * H^T + // Calculate regularized pseudoinverse: (H^T * H + λI)^(-1) * H^T Matrix transposeH = H.Transpose(); Matrix hTh = transposeH.Multiply(H); // Create identity matrix for regularization Matrix identity = Matrix.CreateIdentity(hTh.Rows); - // Apply regularization: hTh + ?I + // Apply regularization: hTh + λI T regFactor = NumOps.FromDouble(regularizationFactor); for (int i = 0; i < identity.Rows; i++) { diff --git a/src/NeuralNetworks/HopfieldNetwork.cs b/src/NeuralNetworks/HopfieldNetwork.cs index c93af626a0..3e6e403f6d 100644 --- a/src/NeuralNetworks/HopfieldNetwork.cs +++ b/src/NeuralNetworks/HopfieldNetwork.cs @@ -200,7 +200,7 @@ private void InitializeWeights() /// better recall. This process is different from training in most neural networks because: /// - It happens in one pass, not through repeated iterations /// - It doesn't use backpropagation or gradients - /// - It has limited capacity (can only store approximately 0.14 � network size patterns reliably) + /// - It has limited capacity (can only store approximately 0.14 * network size patterns reliably) /// /// public void Train(List> patterns) @@ -314,8 +314,26 @@ public Vector Recall(Vector input, int maxIterations = 100) /// public override void UpdateParameters(Vector parameters) { - // Hopfield networks typically don't use gradient-based updates - throw new InvalidOperationException("Hopfield networks do not support gradient-based parameter updates. Use the Train(List> patterns) method instead."); + // Number of off-diagonal entries in a symmetric NxN matrix + int expectedLength = (_size * (_size - 1)) / 2; + + if (parameters.Length != expectedLength) + { + throw new ArgumentException($"Parameter vector length mismatch. Expected {expectedLength} parameters but got {parameters.Length}.", nameof(parameters)); + } + + int paramIndex = 0; + // Fill upper triangle and mirror to lower triangle; keep diagonal zero + for (int i = 0; i < _size; i++) + { + _weights[i, i] = NumOps.Zero; + for (int j = i + 1; j < _size; j++) + { + var w = parameters[paramIndex++]; + _weights[i, j] = w; + _weights[j, i] = w; + } + } } /// diff --git a/src/NeuralNetworks/SelfOrganizingMap.cs b/src/NeuralNetworks/SelfOrganizingMap.cs index 2670475ba3..dd383646ed 100644 --- a/src/NeuralNetworks/SelfOrganizingMap.cs +++ b/src/NeuralNetworks/SelfOrganizingMap.cs @@ -141,7 +141,7 @@ public class SelfOrganizingMap : NeuralNetworkBase /// When creating a new SOM: /// - The architecture tells us how many input dimensions we have (how many attributes each data point has) /// - The architecture also suggests how many total positions we want on our map - /// - The constructor tries to make the map as square as possible (e.g., 10�10 rather than 5�20) + /// - The constructor tries to make the map as square as possible (e.g., 10×10 rather than 5×20) /// - It may adjust the total map size slightly to make a perfect square if needed /// /// Once the dimensions are set, it creates weight values for each position on the map. @@ -331,7 +331,7 @@ private int FindBestMatchingUnit(Vector input) /// For example, if comparing two books with attributes for page count and publication year: /// - Book 1: 300 pages, published in 2010 /// - Book 2: 400 pages, published in 2020 - /// - The calculation would be: v[(300-400)� + (2010-2020)�] = v(10,100 + 100) = v10,200 � 101 + /// - The calculation would be: sqrt[(300-400)² + (2010-2020)²] = sqrt(10,100 + 100) = sqrt(10,200) ≈ 101 /// /// A smaller distance means the data points are more similar. /// @@ -538,7 +538,7 @@ private T CalculateInfluence(T distance, T radius) /// - Changes are proportional to learning rate and influence /// - The BMU moves more than distant positions /// - /// The formula (learningRate � influence � (input - weight)) moves each weight + /// The formula (learningRate * influence * (input - weight)) moves each weight /// some fraction of the way toward matching the corresponding input value. /// /// @@ -576,8 +576,22 @@ private Vector CalculateWeightDelta(Vector input, Vector weight, T lear /// public override void UpdateParameters(Vector parameters) { - // This method is not typically used in SOMs - throw new InvalidOperationException("UpdateParameters is not implemented for Self-Organizing Maps. SOMs use competitive learning and update weights through the Train method instead."); + int expectedLength = (_mapWidth * _mapHeight) * _inputDimension; + + if (parameters.Length != expectedLength) + { + throw new ArgumentException($"Parameter vector length mismatch. Expected {expectedLength} parameters but got {parameters.Length}.", nameof(parameters)); + } + + int paramIndex = 0; + + for (int i = 0; i < _mapWidth * _mapHeight; i++) + { + for (int j = 0; j < _inputDimension; j++) + { + _weights[i, j] = parameters[paramIndex++]; + } + } } ///