Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
164 commits
Select commit Hold shift + click to select a range
235f10a
fix: RBMLayer Forward uses FusedLinear for tape-tracked gradients (#1…
ooples Apr 6, 2026
beaa596
fix: rewrite HyperbolicLinearLayer Forward with batched tape-tracked …
ooples Apr 6, 2026
acdcca8
fix: HopeNetwork deterministic Predict + modern Hopfield recall (#1086)
ooples Apr 6, 2026
cec3733
perf: eliminate all tensor allocations in HyperbolicLinearLayer Forward
ooples Apr 6, 2026
e463bf1
fix: address 11 PR review comments — perf, correctness, configurability
ooples Apr 6, 2026
8bcfbce
fix: address 8 review comments — cache invalidation, validation, stat…
ooples Apr 6, 2026
f67868c
Merge branch 'master' into fix/nn-test-failures-1086
ooples Apr 6, 2026
3920d95
fix: remove dead blend/buffer-scan code from AssociativeMemory.Retrieve
ooples Apr 6, 2026
cd77623
Merge branch 'fix/nn-test-failures-1086' of https://github.com/ooples…
ooples Apr 6, 2026
f8e03bb
Merge branch 'master' into fix/nn-test-failures-1086
ooples Apr 6, 2026
e75fc9b
fix: address 5 review comments — cache, validation, dead code, API co…
ooples Apr 6, 2026
8efc422
Merge branch 'fix/nn-test-failures-1086' of https://github.com/ooples…
ooples Apr 6, 2026
17f4e74
Merge branch 'master' into fix/nn-test-failures-1086
ooples Apr 6, 2026
d4c16b5
fix: address 5 review comments + add website vercel ignoreCommand
ooples Apr 6, 2026
ab5cb94
Merge branch 'fix/nn-test-failures-1086' of https://github.com/ooples…
ooples Apr 6, 2026
addf8e7
fix: restore GetAssociationMatrix as computed view, fix test build er…
ooples Apr 6, 2026
275c9f6
refactor: add NestedLearningBase<T> with Engine/NumOps, fix bias shape
ooples Apr 6, 2026
e9b8816
Merge branch 'master' into fix/nn-test-failures-1086
ooples Apr 6, 2026
fb63de7
fix: address 9 review comments — validation, tests, base class, dupli…
ooples Apr 6, 2026
06d30e3
Merge branch 'fix/nn-test-failures-1086' of https://github.com/ooples…
ooples Apr 6, 2026
5f9db05
fix: address 8 review comments — engine acceleration, validation, docs
ooples Apr 6, 2026
dbe8219
fix: address 10 review comments — math correctness, determinism, perf
ooples Apr 6, 2026
30c0824
fix: resolve 30 CI test failures across 6 categories
franklinic Apr 6, 2026
0e36cb0
fix: resolve test failures across 4 categories in PR #1088
franklinic Apr 7, 2026
5e18411
fix: wrap Serving model load operations in InternalOperation scope
franklinic Apr 7, 2026
a2e9ed2
fix: MoE CombineExpertOutputs use TensorGather instead of TensorSlice…
franklinic Apr 7, 2026
fbc3015
chore: remove temporary test failure report
franklinic Apr 7, 2026
32f6f8f
fix: resolve 6 PR review comments on #1088
franklinic Apr 7, 2026
5abfe13
fix: add SetTrainingMode toggle to Physics/ScientificML Train overrides
franklinic Apr 7, 2026
c39b8a4
fix: address 7 new review comments on PR #1088
franklinic Apr 7, 2026
45bc996
fix: make GetOrCreateBaseOptimizer protected for subclass access
ooples Apr 7, 2026
e0eeb12
fix: address 15 review comments — Poincaré projection, test coverage,…
ooples Apr 7, 2026
8261519
fix: named constants, fail-fast gradient validation, clean tensor cre…
ooples Apr 7, 2026
49f8fe5
fix: address 8 review comments — c==0 guard, pre-mutation validation,…
ooples Apr 7, 2026
84bad91
fix: reset contextFlow in Predict, efficient constant tensor creation
ooples Apr 7, 2026
179d863
Merge branch 'master' into fix/nn-test-failures-1086
ooples Apr 7, 2026
20262cb
fix: use configured optimizer, preserve metaState, cache MoE index te…
ooples Apr 7, 2026
fe10363
Merge branch 'fix/nn-test-failures-1086' of https://github.com/ooples…
ooples Apr 7, 2026
b4accb7
fix: project Euclidean input into Poincaré ball before Möbius matmul
ooples Apr 7, 2026
a347917
fix: MoE use GetSliceAlongDimension instead of TensorGather axis=1
ooples Apr 8, 2026
d646596
fix: FNO pass configured optimizer, SparseLinearLayer skip tape regis…
ooples Apr 8, 2026
88b988a
fix: DGP variance estimation, HDBSCAN tree walk, MoE gather workaround
ooples Apr 8, 2026
4d9677b
fix: HDBSCAN stability computation, clustering test configurations
ooples Apr 8, 2026
72f4425
fix: add blame-hang-timeout for slow test detection + reporter upgrade
ooples Apr 8, 2026
d4c20b7
test: add per-test xunit timeouts to all 1156 test files
ooples Apr 8, 2026
29d7a8c
fix: resolve 3 FederatedCoordinator test failures
ooples Apr 8, 2026
89e629f
fix: address 6 PR review comments — docs, training flags, perf
ooples Apr 8, 2026
771c64b
revert: remove per-test xunit timeouts — incompatible with sync tests
ooples Apr 8, 2026
cca2db3
test: add per-test xunit timeouts with async task conversion
ooples Apr 8, 2026
0da3b1b
ci: fix build timeout — exclude obj/ from artifacts, bump to 40min
ooples Apr 8, 2026
4b64704
fix: call base.Dispose in VoyageAI test mock override
ooples Apr 8, 2026
d7767b8
fix: resolve 27 CI test failures across 9 categories
franklinic Apr 8, 2026
c005906
chore: remove temp failure report
franklinic Apr 8, 2026
b856b4a
fix: MoE gradient flow — use TensorMatMul for tape-tracked column ext…
franklinic Apr 8, 2026
ea44465
merge: resolve conflicts with master (ResearchPaper rename)
franklinic Apr 8, 2026
1144d16
fix: resolve 6 new PR review comments
franklinic Apr 8, 2026
d126974
fix: resolve round 3 regressions — Serving JSON + TFT shape mismatch
franklinic Apr 9, 2026
2278949
fix: resolve round 3 remaining — VideoCLIP, DeepGP, HDBSCAN, MappedRF…
franklinic Apr 9, 2026
07b6f14
fix: VideoCLIP tape-tracked forward + NBEATS TrainWithTape migration
franklinic Apr 9, 2026
6847ac9
perf: replace scalar loops with Engine ops in shared layers + networks
franklinic Apr 9, 2026
a153270
perf: Engine acceleration for BGE, SoundStorm, MATCHA + shared layers
franklinic Apr 9, 2026
b30649a
fix: paper-aligned fixes for NBEATS, VideoCLIP, DiT/Oasis
franklinic Apr 9, 2026
1e60615
fix: SimCSE outputs [CLS] embeddings per Gao et al. 2021, not vocab l…
franklinic Apr 9, 2026
51f6517
fix: TFT full paper rewrite per Lim et al. 2021
franklinic Apr 9, 2026
1e27864
fix: ControlNet encoder uses ConvolutionalLayer per Zhang et al. 2023
franklinic Apr 9, 2026
ec25a83
fix: Hopfield metadata, Hyperbolic NN gradients, Tanh test tolerance
franklinic Apr 9, 2026
26e9139
fix: ViLBERT numHeads 12→16 per Lu et al. 2019
franklinic Apr 9, 2026
3bb1bd6
fix: paper-aligned fixes for HDBSCAN, WaveRNN, MoE, VGG Clone
franklinic Apr 9, 2026
d2f189c
fix: NBEATS use MAE loss per Oreshkin et al. 2020
franklinic Apr 9, 2026
541e262
fix: add TFT constructor validation per paper constraints
franklinic Apr 9, 2026
ab2ab72
fix: replace remaining TensorSliceAxis calls in VideoCLIP
franklinic Apr 9, 2026
0fb39ed
revert: undo HDBSCAN changes that caused 3 new test failures
franklinic Apr 9, 2026
d2ca0f9
fix: remove duplicate SetTrainableParameters conflicting with generator
franklinic Apr 9, 2026
9ce3d4c
fix: use Engine.Reshape for tape-tracked gradient flow
franklinic Apr 10, 2026
9dc48b7
fix: tape-tracked Engine.Reshape in NBEATS training loop
franklinic Apr 10, 2026
f0d2e7d
fix: NBEATS uses observed data for in-sample prediction per paper
franklinic Apr 10, 2026
ef1ce26
fix: VideoCLIP tape-tracked Engine.Reshape for gradient flow
franklinic Apr 10, 2026
951d3b4
fix: Engine.Reshape in DenseLayer, BatchNorm, TransformerEncoder, GRU
franklinic Apr 10, 2026
f442ed1
fix: Engine.Reshape in LSTMLayer for tape-tracked gradient flow
franklinic Apr 10, 2026
40f0ad0
fix: Engine.Reshape in MultiHeadAttentionLayer for gradient flow
franklinic Apr 10, 2026
86355d4
fix: Engine.Reshape in AttentionLayer and CrossAttentionLayer
franklinic Apr 10, 2026
9a76327
fix: restore correct numHeads defaults per each model's paper
franklinic Apr 10, 2026
105ef02
fix: TFT tape-based training + Tensors 0.31.0 + Engine.Reshape throug…
franklinic Apr 10, 2026
7360d63
fix: TFT optimizer uses Tensor<T> generic params for tape compatibility
franklinic Apr 10, 2026
0acb488
fix: Engine.Reshape in ConvolutionalLayer for gradient tape flow
franklinic Apr 10, 2026
a970e37
fix: Tensors 0.31.1 + ConvLayer Engine.Reshape + Adam gradient reshape
franklinic Apr 10, 2026
b319593
fix: HDBSCAN AllowSingleCluster in EOM + default true per paper
franklinic Apr 10, 2026
81f8183
fix: NBEATS CollectParameters uses version -1 to avoid stale cache
franklinic Apr 10, 2026
07008cc
revert: undo HDBSCAN AllowSingleCluster changes (caused regressions)
franklinic Apr 10, 2026
a61e2c1
fix: HDBSCAN stability per scikit-learn + AllowSingleCluster support
franklinic Apr 10, 2026
d1534d9
fix: N-BEATS generic blocks use learned basis per Oreshkin et al. 2020
ooples Apr 11, 2026
7906e23
perf: replace SPSA with tape-based training for diffusion models
ooples Apr 11, 2026
d50733e
perf: enable xunit parallel test collections
ooples Apr 11, 2026
20b1d66
fix: make [Fact(Timeout)] actually fire on fake-async tests
ooples Apr 11, 2026
31bce28
perf: tighten coverlet instrumentation to cut per-shard finalization
ooples Apr 11, 2026
b4a1686
fix: moe combineexpertoutputs uses tensorgather axis=1 directly
ooples Apr 11, 2026
d412e37
perf: wire up tensorarena at diffusion + nn training entry points
ooples Apr 11, 2026
1a6a010
perf: scope each modelfamily test in a tensorarena
ooples Apr 11, 2026
b30c59d
Revert "perf: wire up tensorarena at diffusion + nn training entry po…
ooples Apr 11, 2026
b37b912
fix: diffusion tape training actually runs (3 real perf bugs)
ooples Apr 11, 2026
46738ab
fix: diffusion tape training — 6 tape-break sites in forward path
ooples Apr 11, 2026
c456633
fix: more tape-break sites across core layer forward paths
ooples Apr 11, 2026
8c5dbb6
fix: more activations override activate(tensor<t>) via engine ops
ooples Apr 11, 2026
f464152
fix: decompose 13 exotic activations into tape-tracked engine primitives
ooples Apr 11, 2026
2a3e897
perf: tape-direct diffusion train, flashattention tensor<t> migration
ooples Apr 11, 2026
f682acc
fix: 5 more critical layers tape-tracked via engine ops
ooples Apr 11, 2026
5548960
fix: tape-track more layer forwards — convlstm, capsule, graph, etc.
ooples Apr 11, 2026
aa2a187
fix: tape-track forward in 40 ssm layers (mamba, rwkv, hyena, etc.)
ooples Apr 11, 2026
4a1c0e0
fix: tape-track forward in 45 more neuralnetworks layers
ooples Apr 11, 2026
1037f3e
perf: zero-alloc shape access + tape-track neuralnetworkbase train
ooples Apr 11, 2026
d450e07
perf: zero-alloc shape in DETR/DINO/RTDETR forward methods
ooples Apr 11, 2026
192295a
fix: tape-track chained engine.op().reshape() calls in layer forwards
ooples Apr 11, 2026
df0c86b
perf: replace Shape.ToArray() with zero-alloc _shape field access
ooples Apr 11, 2026
50bf8e4
fix: route remaining forward-path reshapes through Engine for tape tr…
ooples Apr 11, 2026
d9903c4
fix: more forward-path tape breaks (bias reshapes, concat, conv slice)
ooples Apr 11, 2026
75ac8ed
perf: hoist bias add in Mamba/GatedDeltaNet conv1d forward loops
ooples Apr 11, 2026
1cbf97c
fix: tape-track 5 more activations via engine primitives
ooples Apr 11, 2026
ba79f7b
fix: tape-track 6 hard activations via engine primitives
ooples Apr 11, 2026
28ebd69
refactor: delete orphaned manual-backward helpers (ConvLSTM, LSTM, STN)
ooples Apr 11, 2026
ae9efe6
refactor: delete orphaned manual-backward helpers across 12 layers
ooples Apr 11, 2026
01cb8d2
fix: use native engine.tensorsquash instead of hand-decomposed formula
ooples Apr 11, 2026
887979e
feat: add prelulayer with learnable per-channel alpha
ooples Apr 11, 2026
5a9a96b
refactor: delete dead spsa fallback from flashattentionlayer
ooples Apr 11, 2026
22da9cd
feat: fno 2d spectral conv uses engine.fft2d on the gradient tape
ooples Apr 11, 2026
6ff6dbe
feat: fno n-d spectral conv uses separable engine.fft on the tape
ooples Apr 11, 2026
c40a981
feat: fno training runs on gradient tape via tapetrainstep
ooples Apr 11, 2026
f893737
refactor: delete orphaned fno fourierlayer backward + fft helpers
ooples Apr 11, 2026
4f77a5f
refactor: delete fno legacy complex-tensor scaffolding
ooples Apr 11, 2026
932fc6a
refactor: fno pointwise mixing runs on the gradient tape
ooples Apr 11, 2026
3e04f05
fix: use net471-compatible reference comparer in diffusion collectparams
ooples Apr 11, 2026
7954817
perf: zero-alloc engine acceleration for EBM, AdaBoost, NGBoost, Easy…
ooples Apr 12, 2026
c4c0858
fix: remove broken LSTM Train override that skipped backward pass
ooples Apr 12, 2026
94c16d1
fix: gracefully handle parameter size changes from lazy layer initial…
ooples Apr 12, 2026
7a8b73d
fix: use net471-compatible reference comparer in diffusion collectparams
ooples Apr 12, 2026
6ce81be
fix: HDBSCAN EOM cluster selection matched to Campello et al. 2013
ooples Apr 12, 2026
28e944d
perf: DeepANT FC forward uses Engine.DotProduct instead of scalar loop
ooples Apr 12, 2026
317733e
ci: trigger CI after force-push
ooples Apr 12, 2026
16cf443
perf: keep parameter buffer across training iterations instead of reb…
ooples Apr 12, 2026
edfa009
perf: zero-alloc TensorAllocator for DenseLayer and EmbeddingLayer we…
ooples Apr 12, 2026
1b86c42
perf: TensorArena in TrainWithTape + bulk TensorCopy + unskip trainin…
ooples Apr 12, 2026
2105acb
fix: DenseNet outputs raw logits per Huang et al. 2017 + Clone preser…
ooples Apr 12, 2026
a5a59a9
fix: AVEL layer factory produces correct layer types matching Initial…
ooples Apr 12, 2026
bdf7ae4
chore: update AiDotNet.Tensors to 0.35.2
ooples Apr 12, 2026
896fa12
merge: resolve conflicts with master — keep Engine ops, take new back…
ooples Apr 12, 2026
9d6bb5d
fix: RBFLayer register widths with Biases role to avoid role-based dedup
ooples Apr 12, 2026
2c92465
fix: skip redundant model.Train() + lazy stats + in-place updates (Is…
ooples Apr 13, 2026
f21269b
fix: clone tensor shape arrays before mutation to prevent source corr…
ooples Apr 13, 2026
e650893
fix: address 12 PR review comments — mask validation, activation bugs…
ooples Apr 13, 2026
01f0248
fix: in-place SetParameters for 4 layers + GRU validation + mask dispose
ooples Apr 13, 2026
995645f
fix: hdbscan — iterative descendant walk, leaf fallback, root cluster…
ooples Apr 13, 2026
a0a4009
fix: t5 conditioner allocs + controlnet stride formula
ooples Apr 13, 2026
c02421b
fix: hyperbolic layer alloc cache, lstm bias hoist, hope predict/train
ooples Apr 13, 2026
539e029
fix: autodiff numerical gradient uses output shape + controlnet xml docs
ooples Apr 13, 2026
f2909b0
fix: net471 build errors — tensor shape types, interface default impl
ooples Apr 13, 2026
cc4fb4f
fix: mixture of experts uses non-deterministic noise in gating
ooples Apr 13, 2026
cc4b7f2
fix: narrow coverlet exclusion pattern for serving interfaces
ooples Apr 13, 2026
2b3f6cc
fix: 6 more review comments — rbf role, slow-test script, builder, la…
ooples Apr 13, 2026
356c8ac
fix: reservoir layer engine ops for tape tracking + dense/flux alloc fix
ooples Apr 13, 2026
4b56523
fix: rbflayer in-place setparameters + rrelu tape-tracked training path
ooples Apr 13, 2026
2b848e9
fix: add CrossEntropyWithLogitsLoss + DenseNet logits fix + builder c…
ooples Apr 13, 2026
57791f2
feat: add ParallelStreamsLayer for dual-stream architectures
ooples Apr 13, 2026
784309a
fix: PR #1124 review comments — T5 encoder rewrite, BCE-with-logits, …
franklinic Apr 13, 2026
27cd710
Merge origin/master into fix/issue-1123-optimizer-perf
franklinic Apr 13, 2026
f7955d0
fix: PR #1124 round-2 review comments — 32 issues across optimizer, s…
franklinic Apr 13, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion coverlet.runsettings
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,17 @@
File exclusions mirror sonar.coverage.exclusions in sonarcloud.yml — Sonar discards
these anyway, so there is no point instrumenting or serializing them.
-->
<ExcludeByFile>**/ProgramSynthesis/Models/**/*.cs,**/ProgramSynthesis/Execution/**/*.cs,**/ProgramSynthesis/Requests/**/*.cs,**/ProgramSynthesis/Results/**/*.cs,**/ProgramSynthesis/Interfaces/**/*.cs,**/ProgramSynthesis/Enums/**/*.cs,**/AiDotNet.Serving/Configuration/**/*.cs,**/AiDotNet.Serving/Models/**/*.cs,**/AiDotNet.Serving/Persistence/Entities/**/*.cs,**/AiDotNet.Serving/Persistence/Migrations/**/*.cs,**/AiDotNet.Serving/**/I*.cs,**/Interfaces/**/*.cs</ExcludeByFile>
<!--
Do NOT add a broad **/Interfaces/**/*.cs exclusion here: several files under
src/Interfaces are not pure interfaces (e.g. SubModel.cs contains a concrete
class; IParameterizable.cs has default interface method bodies). Excluding
the whole directory would silently hide executable code from coverage.
Configuration/DTO directories with no executable code are listed explicitly.
For the AiDotNet.Serving project we use the more aggressive **/I*.cs pattern
(per master's narrower-than-the-folder convention) to catch I-prefixed
interface files anywhere under that project tree.
-->
<ExcludeByFile>**/ProgramSynthesis/Models/**/*.cs,**/ProgramSynthesis/Execution/**/*.cs,**/ProgramSynthesis/Requests/**/*.cs,**/ProgramSynthesis/Results/**/*.cs,**/ProgramSynthesis/Interfaces/**/*.cs,**/ProgramSynthesis/Enums/**/*.cs,**/AiDotNet.Serving/Configuration/**/*.cs,**/AiDotNet.Serving/Models/**/*.cs,**/AiDotNet.Serving/Persistence/Entities/**/*.cs,**/AiDotNet.Serving/Persistence/Migrations/**/*.cs,**/AiDotNet.Serving/**/I*.cs</ExcludeByFile>
<!-- SingleHit: record only whether each line was hit, not the hit count. Smaller XML, faster write. -->
<SingleHit>true</SingleHit>
<!-- SkipAutoProps: don't instrument trivial compiler-generated property accessors. -->
Expand Down
48 changes: 38 additions & 10 deletions src/ActivationFunctions/BinarySpikingActivation.cs
Original file line number Diff line number Diff line change
Expand Up @@ -238,20 +238,48 @@ public override Matrix<T> Derivative(Vector<T> input)
/// </remarks>
public override Tensor<T> Activate(Tensor<T> input)
{
// Binary spiking is non-differentiable by design (Heaviside step), so
// surrogate-gradient learning approximates the forward with a steep
// sigmoid. This is the standard approach in SNN literature (Neftci et
// al., "Surrogate Gradient Learning in SNNs", 2019) and lets every op
// land on the gradient tape.
// Straight-through estimator (STE) per Neftci et al. 2019,
// "Surrogate Gradient Learning in Spiking Neural Networks":
// * Forward value = Heaviside(input - threshold) — true binary spike.
// * Backward value = sigmoid'(slope·(input - threshold)) — smooth surrogate.
//
// f(x) ≈ sigmoid(slope * (x - threshold))
// The trick: build the output as
// output = sigmoid + (heaviside - sigmoid_detached)
// where the sigmoid term is a tape-tracked Engine.Sigmoid so the gradient
// flows through the sigmoid derivative, and the (heaviside - sigmoid_detached)
// correction is a tape-detached constant offset that simply subtracts the
// sigmoid value and adds the Heaviside value, yielding the correct binary
// forward output. Because the correction tensors have no GradFn, the tape
// never tries to backprop through them — so the gradient of `output` w.r.t.
// `input` is exactly the sigmoid derivative, the surrogate prescribed by
// the SNN literature.
//
// Larger `derivativeSlope` sharpens the approximation toward the true
// Heaviside and narrows the non-zero gradient window, matching the
// intent of the scalar Derivative(T) rectangle.
// This preserves the paper-faithful Heaviside output (true binary spikes
// at inference and on the forward pass during training) while keeping every
// step on the gradient tape so optimizer backward works automatically.
var shifted = Engine.TensorSubtractScalar(input, _threshold);
var scaled = Engine.TensorMultiplyScalar(shifted, _derivativeSlope);
return Engine.Sigmoid(scaled);
var sigmoidValue = Engine.Sigmoid(scaled); // tape-tracked

// Heaviside snapshot — detached from tape, plain Tensor<T>.
var heaviside = new Tensor<T>(input._shape);
for (int i = 0; i < input.Length; i++)
heaviside[i] = NumOps.GreaterThanOrEquals(input[i], _threshold) ? NumOps.One : NumOps.Zero;

// Detached snapshot of sigmoidValue — copy the values out so the resulting
// tensor has no GradFn link back to `scaled` / `input`. The cheapest way to
// do this without engine support is to materialize a fresh Tensor<T> from
// the value buffer.
var sigmoidDetached = new Tensor<T>(input._shape);
for (int i = 0; i < sigmoidValue.Length; i++)
sigmoidDetached[i] = sigmoidValue[i];

// correction = heaviside - sigmoidDetached (both detached → result detached)
// output = sigmoidValue + correction (sigmoidValue is tape-tracked)
// Forward value collapses to: sigmoid + heaviside - sigmoid = heaviside ✓
// Backward gradient: only sigmoidValue contributes → sigmoid' ✓
var correction = Engine.TensorSubtract(heaviside, sigmoidDetached);
return Engine.TensorAdd(sigmoidValue, correction);
}

/// <summary>
Expand Down
16 changes: 13 additions & 3 deletions src/ActivationFunctions/ScaledTanhActivation.cs
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,12 @@ public class ScaledTanhActivation<T> : ActivationFunctionBase<T>
private readonly T _beta;
private readonly double _saturationThreshold;

// Precomputed constants for the tensor hot path to avoid repeated NumOps.FromDouble
// conversions and scalar arithmetic per Activate(Tensor<T>) call.
private readonly T _halfBeta;
private readonly T _negSaturationThreshold;
private readonly T _posSaturationThreshold;

/// <summary>
/// Initializes a new instance of the ScaledTanhActivation class with the specified steepness parameter.
/// </summary>
Expand All @@ -64,6 +70,9 @@ public ScaledTanhActivation(double beta = 1.0, double saturationThreshold = 20.0
{
_beta = NumOps.FromDouble(beta);
_saturationThreshold = saturationThreshold;
_halfBeta = NumOps.Multiply(_beta, NumOps.FromDouble(0.5));
_negSaturationThreshold = NumOps.FromDouble(-saturationThreshold);
_posSaturationThreshold = NumOps.FromDouble(saturationThreshold);
}

/// <summary>
Expand Down Expand Up @@ -119,9 +128,10 @@ public override T Activate(T input)
/// </summary>
public override Tensor<T> Activate(Tensor<T> input)
{
var halfBeta = NumOps.Multiply(_beta, NumOps.FromDouble(0.5));
var scaled = Engine.TensorMultiplyScalar(input, halfBeta);
return Engine.Tanh(scaled);
var scaled = Engine.TensorMultiplyScalar(input, _halfBeta);
// Match scalar path: clamp to saturation threshold to prevent overflow
var clamped = Engine.TensorClamp(scaled, _negSaturationThreshold, _posSaturationThreshold);
return Engine.Tanh(clamped);
}

/// <summary>
Expand Down
17 changes: 16 additions & 1 deletion src/ActivationFunctions/SphericalSoftmaxActivation.cs
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,14 @@ namespace AiDotNet.ActivationFunctions;
[ActivationProperty(IsMonotonic = false, ZeroPreserving = false, IsBounded = true, IsVectorActivation = true, Cost = ComputeCost.High)]
public class SphericalSoftmaxActivation<T> : ActivationFunctionBase<T>
{
/// <summary>
/// Numerical-stability epsilon added to the squared-norm before the sqrt to avoid
/// division-by-zero when the input vector is exactly zero. The value 1e-12 is well
/// below single-precision rounding error and is the standard guard used in
/// L2-normalization layers (e.g. PyTorch's <c>F.normalize(eps=1e-12)</c>).
/// </summary>
private static readonly T NormalizationEpsilon = MathHelper.GetNumericOperations<T>().FromDouble(1e-12);

/// <summary>
/// Indicates whether this activation function supports scalar operations.
/// </summary>
Expand Down Expand Up @@ -111,10 +119,17 @@ public override Vector<T> Activate(Vector<T> input)
/// </summary>
public override Tensor<T> Activate(Tensor<T> input)
{
if (input is null)
throw new ArgumentNullException(nameof(input));
if (input.Shape.Length == 0)
throw new ArgumentException("SphericalSoftmax requires at least one tensor dimension.", nameof(input));

int lastAxis = input.Shape.Length - 1;
var squared = Engine.TensorMultiply(input, input);
var sumSquared = Engine.ReduceSum(squared, new[] { lastAxis }, keepDims: true);
var norm = Engine.TensorSqrt(sumSquared);
var eps = Tensor<T>.CreateDefault(sumSquared._shape, NormalizationEpsilon);
var safeSum = Engine.TensorAdd(sumSquared, eps);
var norm = Engine.TensorSqrt(safeSum);
var normalized = Engine.TensorBroadcastDivide(input, norm);
return Engine.Softmax(normalized, axis: lastAxis);
}
Expand Down
25 changes: 24 additions & 1 deletion src/AiModelBuilder.cs
Original file line number Diff line number Diff line change
Expand Up @@ -2469,7 +2469,13 @@ T ObjectiveFunction(Dictionary<string, object> trialHyperparameters)
// train/test split) since cluster structure depends on having all data points.
// Note: preprocessedX/preprocessedY may only contain the training split in the
// standard (non-federated) path. For clustering we need the complete dataset.
bool useFullData = model is Clustering.Base.ClusteringBase<T>;
//
// Probe the *unwrapped* model: by this point the local `model` variable may be a
// DDP/FSDP/ZeRO* distributed-training wrapper (see lines ~1921-1953), and a wrapper
// is never a ClusteringBase<T> so the check would silently flip to false and route
// clustering training back through the train/test split — which is exactly the bug
// this clustering-data path was added to prevent.
bool useFullData = _model is Clustering.Base.ClusteringBase<T>;
// Clustering models need ALL data points for correct density estimation.
// Use preparedX/preparedY (the full dataset before train/test split) when
// the preprocessing pipeline is not configured. When it IS configured,
Expand Down Expand Up @@ -2497,6 +2503,23 @@ T ObjectiveFunction(Dictionary<string, object> trialHyperparameters)
}
var directX = fullX;
var directY = fullY;

// Override the earlier split-based sample counts logged at lines ~2222-2224 so
// experiment run metadata reflects the dataset actually fed to Train(). Without
// this, clustering runs would be reported with a train/val/test split that was
// never used (we just trained on the full dataset).
if (useFullData && experimentRun is not null)
{
int effectiveSamples = InputHelper<T, TInput>.GetInputSize(directX);
experimentRun.LogParameters(new Dictionary<string, object>
{
["training_samples"] = effectiveSamples,
["validation_samples"] = 0,
["test_samples"] = 0,
["full_data_used"] = true,
});
}

model.Train(directX, directY);

// Compute evaluation metrics
Expand Down
1 change: 1 addition & 0 deletions src/Audio/Generation/VoiceCraft.cs
Original file line number Diff line number Diff line change
Expand Up @@ -247,6 +247,7 @@ protected override void InitializeLayers()
if (!_useNativeMode) return;
if (Architecture.Layers is not null && Architecture.Layers.Count > 0) Layers.AddRange(Architecture.Layers);
else Layers.AddRange(LayerHelper<T>.CreateDefaultVoiceCraftLayers(
Architecture,
hiddenDim: _options.HiddenDim, numLayers: _options.NumLayers,
numHeads: _options.NumHeads, codebookSize: _options.CodebookSize,
dropoutRate: _options.DropoutRate));
Expand Down
7 changes: 4 additions & 3 deletions src/Augmentation/Image/ImageTensor.cs
Original file line number Diff line number Diff line change
Expand Up @@ -266,8 +266,9 @@ public ImageTensor(int batchSize, int height, int width, int channels = 3, Chann
/// <returns>A new ImageTensor with copied data.</returns>
public ImageTensor<T> Clone()
{
// Create a copy of the tensor data
var clonedData = new Tensor<T>((int[])_data._shape);
// Create a copy of the tensor data — defensive-copy the shape array
// so the clone never aliases the original's mutable shape metadata.
var clonedData = new Tensor<T>((int[])_data._shape.Clone());
for (int i = 0; i < _data.Length; i++)
{
clonedData[i] = _data[i];
Expand Down Expand Up @@ -419,7 +420,7 @@ public ImageTensor<T> Crop(int x, int y, int width, int height)
/// <returns>The dimensions array.</returns>
public int[] GetDimensions()
{
return (int[])_data._shape;
return (int[])_data._shape.Clone();
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.

/// <summary>
Expand Down
28 changes: 22 additions & 6 deletions src/Autodiff/Testing/TensorOperationsVerification.cs
Original file line number Diff line number Diff line change
Expand Up @@ -111,13 +111,21 @@ public NumericalGradient<T>.ComparisonResult VerifyUnaryOperation(
{
// Compute autodiff gradient via ComputationNode backward
Tensor<T> autodiffGradient;
int[] outputShape;
{
var inputNode = TensorOperations<T>.Variable(input.Clone(), "input", requiresGradient: true);

var outputNode = operation(inputNode);
if (outputNode.Value is null)
{
throw new InvalidOperationException(
$"Operation '{operationName}' produced a ComputationNode with a null Value. " +
"Cannot verify gradients against an unevaluated node.");
}
outputShape = outputNode.Value._shape;

// Create output gradient (ones)
var outputGradient = CreateOnes(outputNode.Value._shape);
var outputGradient = CreateOnes(outputShape);
Comment thread
ooples marked this conversation as resolved.
outputNode.Gradient = outputGradient;

// Run backward pass
Expand All @@ -126,8 +134,8 @@ public NumericalGradient<T>.ComparisonResult VerifyUnaryOperation(
autodiffGradient = inputNode.Gradient ?? new Tensor<T>(input._shape);
}

// Compute numerical gradient
var outputGrad = CreateOnes(input._shape);
// Compute numerical gradient using the operation's output shape for the seed gradient
var outputGrad = CreateOnes(outputShape);
var numericalGradient = NumericalGradient<T>.ComputeForOperation(
input.Clone(),
outputGrad,
Expand Down Expand Up @@ -182,14 +190,22 @@ public NumericalGradient<T>.ComparisonResult VerifyUnaryOperation(
{
// Compute autodiff gradients via ComputationNode backward
Tensor<T> autodiffGrad1, autodiffGrad2;
int[] outputShape;
{
var node1 = TensorOperations<T>.Variable(input1.Clone(), "input1", requiresGradient: true);
var node2 = TensorOperations<T>.Variable(input2.Clone(), "input2", requiresGradient: true);

var outputNode = operation(node1, node2);
if (outputNode.Value is null)
{
throw new InvalidOperationException(
$"Operation '{operationName}' produced a ComputationNode with a null Value. " +
"Cannot verify gradients against an unevaluated node.");
}
outputShape = outputNode.Value._shape;

// Create output gradient (ones)
var outputGradient = CreateOnes(outputNode.Value._shape);
var outputGradient = CreateOnes(outputShape);
Comment thread
ooples marked this conversation as resolved.
outputNode.Gradient = outputGradient;

// Run backward pass
Expand All @@ -199,8 +215,8 @@ public NumericalGradient<T>.ComparisonResult VerifyUnaryOperation(
autodiffGrad2 = node2.Gradient ?? new Tensor<T>(input2._shape);
}

// Compute numerical gradients
var outputGrad = CreateOnes(input1._shape);
// Compute numerical gradients using the operation's output shape for the seed gradient
var outputGrad = CreateOnes(outputShape);
var (numericalGrad1, numericalGrad2) = NumericalGradient<T>.ComputeForBinaryOperation(
input1.Clone(),
input2.Clone(),
Expand Down
Loading
Loading