Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
53 commits
Select commit Hold shift + click to select a range
78550e5
fix(modelfamily): correct generator InputShape for VL + voice-cloning…
ooples Jun 5, 2026
6b7719f
fix(modelfamily): correct generator InputShape for PointNet++ and VFIT
ooples Jun 5, 2026
fb0997b
fix(video): correct RIFE per-frame channel count and RAFT flow-grid s…
ooples Jun 5, 2026
07957e6
fix(video): train RIFE/RAFT through their real forward graph
ooples Jun 5, 2026
0fdb21a
fix(video): VideoMAE temporal tubelet pooling + train through real graph
ooples Jun 5, 2026
88d0c63
fix(synthetic): stop CTAB-GAN+/CopulaGAN Train from throwing; scope g…
ooples Jun 5, 2026
bd5cc23
feat(nn): add Conv1DLayer; fix SileroVad to use 1-D convs (paper-fait…
ooples Jun 5, 2026
de78de5
fix(audio): tape-connect SileroVad conv frontend via Engine permute/s…
ooples Jun 5, 2026
d9b0aa2
fix(audio): SileroVad GetNamedLayerActivations follows real forward path
ooples Jun 5, 2026
d074b91
fix(audio,nn): SileroVad now 25/25 — preprocess, LSTM init, layer-ref…
ooples Jun 5, 2026
f436d36
fix(video): re-extract RIFE sub-layer refs after deserialize (Clone r…
ooples Jun 5, 2026
4a06c3f
fix(video): make RIFE warp/slice/upsample tape-differentiable (traini…
ooples Jun 6, 2026
c7f9645
fix(vlm): SigLIP2 25/25 — train the vision-encoder output, stable LR
ooples Jun 6, 2026
9375f43
feat(seg): paper-faithful SegMamba reimplementation (3D Mamba, Xing e…
ooples Jun 6, 2026
29caf98
perf(mamba): route MambaBlock selective-scan through fused engine kernel
ooples Jun 6, 2026
7b3f628
fix(helix): repair dual-system layer-chain dim mismatch + profiling h…
ooples Jun 6, 2026
63f6a61
feat(helix): add opt-in weight-streaming via HelixOptions.WeightOfflo…
ooples Jun 6, 2026
a8ebeb6
feat(training): default-on streaming training path (optimizer-in-back…
ooples Jun 6, 2026
fe0b990
perf(training): vectorize streaming 8-bit Adam epilogue (raw double f…
ooples Jun 6, 2026
da2d5cd
test(helix,gpt4point): reduced-scale float ModelFamily tests (Janus p…
ooples Jun 6, 2026
094024e
test(training): streaming-path integration tests (optimizer-in-backwa…
ooples Jun 6, 2026
6e43bb6
build: pin AiDotNet.Tensors to 0.92.0 (gradient-streaming release, Ai…
ooples Jun 6, 2026
5d0b01b
Merge remote-tracking branch 'origin/master' into fix/modelfamily-inp…
franklinic Jun 7, 2026
488f2d6
Merge remote-tracking branch 'origin/master' into fix/modelfamily-inp…
franklinic Jun 8, 2026
1bde6a2
fix(nn): guard GC.GetGCMemoryInfo behind NET5_0_OR_GREATER for net471…
franklinic Jun 8, 2026
2b7f054
fix(reviews): address 13 unresolved comments on PR #1514
franklinic Jun 8, 2026
64a977c
merge: resolve Directory.Packages.props comment conflict with master
franklinic Jun 8, 2026
c74d889
fix(streaming): clip layer-owned params only (parity with eager path)
franklinic Jun 8, 2026
f909fe0
fix(raft): paper-faithful convex upsampling — _upsampleConv now drive…
franklinic Jun 8, 2026
c530e02
test(metadata): assert AdditionalInfo is populated (catch empty-shell…
franklinic Jun 8, 2026
7281b14
Merge remote-tracking branch 'origin/master' into fix-1514
ooples Jun 8, 2026
5239e9d
chore(deps): bump Tensors 0.91.12 → 0.92.0 (gradient-streaming core #…
ooples Jun 8, 2026
019d417
fix(helix): size leading robotics vision LayerNorm to visionDim
ooples Jun 8, 2026
38abd51
fix(#1514 review): remove dead duplicate Conv1D deserialize branch + …
ooples Jun 9, 2026
fe775a7
feat(layers): Conv1DTransposeLayer — PyTorch-parity 1-D temporal upsa…
ooples Jun 9, 2026
74e9e78
feat(vocoders): paper-faithful HiFi-GAN (ConvTranspose1d upsampling +…
ooples Jun 9, 2026
d3ae830
feat(serialize): deserialization round-trip for Conv1DTranspose + HiF…
ooples Jun 9, 2026
1107b5a
feat(scaffold): per-category conv1d vocoder shapes for real upsampling
ooples Jun 9, 2026
bfcaf72
fix(vocoders): populate ModelMetadata.AdditionalInfo in 10 vocoder ov…
ooples Jun 9, 2026
782b406
fix(scaffold): spectral vocoders use T=2 test input for stable MoreDa…
ooples Jun 9, 2026
6dd1f62
fix(vocoders): deterministic position-seeded init for composite block…
ooples Jun 9, 2026
5afe04e
fix(scaffold): evaluate vocoder MoreData in the stable early-training…
ooples Jun 9, 2026
19ac285
docs(#1514 review): reconcile VITS decoder header with its deliberate…
ooples Jun 9, 2026
4eb15f4
fix(tts): populate ModelMetadata.AdditionalInfo in VITS/NaturalSpeech…
ooples Jun 9, 2026
92842f0
fix(tts): residual Transformer blocks in CodecLM + FlowMatching TTS h…
ooples Jun 9, 2026
5d476f3
Revert "fix(tts): residual Transformer blocks in CodecLM + FlowMatchi…
ooples Jun 9, 2026
212ee90
Reapply "fix(tts): residual Transformer blocks in CodecLM + FlowMatch…
ooples Jun 9, 2026
695281d
fix(tts): GPT-SoVITS token-input contract + deep-TTS MoreData window …
ooples Jun 9, 2026
afa9ac8
fix(tts): pin deterministic init for init-sensitive TTS models (VITS …
ooples Jun 9, 2026
58e48ef
fix(tts): populate ModelMetadata.AdditionalInfo for 5 end-to-end TTS …
ooples Jun 9, 2026
c962c20
fix(tts): address CodeRabbit review on vocoder/Conv1DTranspose (#1514)
ooples Jun 9, 2026
3cb9146
fix(tts): UnivNet native init fail-fast on unwired NumLMBlocks/Dropou…
ooples Jun 9, 2026
78b43af
Merge remote-tracking branch 'origin/master' into vocoder-1514
ooples Jun 9, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 9 additions & 8 deletions Directory.Packages.props
Original file line number Diff line number Diff line change
Expand Up @@ -106,7 +106,6 @@
over input channels at batch=1 instead of collapsing to serial). This is
the Tensors-side lever #1463 called for; the memory/OOM half is handled by
the cache-clearing + diffusion re-shard in this PR (#1485).

Bumped Tensors 0.91.2 -> 0.91.11 (latest PUBLISHED on nuget.org) for the
GPU-resident optimizer step (host-read-free Adam for cudaGraph capture) that
this PR (#1501) depends on, plus the proximal-L1 (ISTA) and 8-bit Adam GPU
Expand All @@ -115,14 +114,16 @@
TensorAllocator.RentPinnedOnGpu, QuantizedTapeState GPU buffers) is all present
in 0.91.11 and the solution builds clean against it; the device-side CUDA
kernels are exercised only under AIDOTNET_GPU_ADAM=1 on a real GPU (off in CI).
Do NOT pin an unpublished version (0.91.14/0.91.16/0.91.36 broke restore with
NU1102); re-bump once a newer Tensors is actually released. The AiDotNet.Native
packages stay at 0.91.2 (no native change needed).
Do NOT pin an unpublished version (0.91.14/0.91.16/0.91.36/0.92.0 broke restore
with NU1102); re-bump once a newer Tensors is actually released. The
AiDotNet.Native packages bump with the Tensors lockstep when available.

Bumped 0.91.12 -> 0.92.0: ships BOTH Tensors #564 (memory-bounded gradient-
streaming core — this PR's streaming-training path) and Tensors #574 (FP16
optimizer-agnostic MixedPrecisionCompiledPlan.ComputeGradients — the
all-fused-optimizer FP16 path). AiDotNet.Native packages coreleased in lockstep
at 0.92.0. -->

Bumped 0.91.12 -> 0.92.0: ships the FP16 optimizer-agnostic
MixedPrecisionCompiledPlan.ComputeGradients (Tensors #574, consumed by the
all-fused-optimizer FP16 path here) and the memory-bounded gradient-streaming
core (Tensors #564). AiDotNet.Native packages coreleased in lockstep at 0.92.0. -->
<PackageVersion Include="AiDotNet.Tensors" Version="0.92.0" />
<PackageVersion Include="AiDotNet.Native.OneDNN" Version="0.92.0" />
<PackageVersion Include="AiDotNet.Native.OpenBLAS" Version="0.92.0" />
Expand Down
15 changes: 15 additions & 0 deletions src/ActivationFunctions/LeakyReLUActivation.cs
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,21 @@ public LeakyReLUActivation(double alpha = 0.01)
_alpha = NumOps.FromDouble(alpha);
}

/// <summary>
/// Initializes a new instance of the Leaky ReLU activation function with the
/// default slope (alpha = 0.01).
/// </summary>
/// <remarks>
/// An explicit parameterless constructor is required so the layer
/// (de)serialization layer, which reflectively reconstructs activation
/// functions via <see cref="System.Activator"/>, can recreate this activation
/// on a clone / load round-trip. A constructor with an all-defaulted parameter
/// is not treated as parameterless by <c>Activator.CreateInstance(Type)</c>.
/// </remarks>
public LeakyReLUActivation() : this(0.01)
{
}

/// <summary>
/// Indicates whether this activation function can operate on individual scalar values.
/// </summary>
Expand Down
378 changes: 347 additions & 31 deletions src/AiDotNet.Generators/TestScaffoldGenerator.cs

Large diffs are not rendered by default.

201 changes: 162 additions & 39 deletions src/Audio/VoiceActivity/SileroVad.cs
Original file line number Diff line number Diff line change
Expand Up @@ -323,25 +323,69 @@ protected override void InitializeLayers()
numLstmLayers: _numLstmLayers, lstmHiddenDim: _lstmHiddenDim).ToList();

Layers.Clear();
Layers.AddRange(layers);

ExtractLayerReferences();
}

/// <summary>
/// (Re)populates the conv / LSTM / output sub-layer references from the
/// canonical <see cref="NeuralNetworkBase{T}.Layers"/> list and materializes
/// any lazy weights.
/// </summary>
/// <remarks>
/// Called both after <see cref="InitializeLayers"/> builds the layers and
/// after deserialization rebuilds <c>Layers</c>. Deserialization replaces the
/// <c>Layers</c> list with freshly reconstructed layers but does not know
/// about SileroVad's cached <c>_convLayers</c>/<c>_lstmLayers</c>/<c>_outputLayer</c>
/// references — without re-extracting them, a deserialized/cloned model would
/// keep running the constructor's randomly-initialized layers in Forward while
/// the loaded weights sit unused in <c>Layers</c> (the cause of
/// Clone_ShouldProduceIdenticalOutput diverging). Idempotent.
/// </remarks>
private void ExtractLayerReferences()
{
_convLayers.Clear();
_lstmLayers.Clear();
Layers.AddRange(layers);

// Assign internal references for forward pass (3 conv + numLstm LSTM + 1 output)
int expectedCount = 3 + _numLstmLayers + 1;
if (layers.Count < expectedCount)
if (Layers.Count < expectedCount)
{
throw new ArgumentException(
$"Layer list must have at least {expectedCount} layers " +
$"(3 conv + {_numLstmLayers} LSTM + 1 output), but got {layers.Count}.",
"Architecture.Layers");
$"(3 conv + {_numLstmLayers} LSTM + 1 output), but got {Layers.Count}.",
nameof(Layers));
}

for (int i = 0; i < 3; i++)
_convLayers.Add(layers[i]);
_convLayers.Add(Layers[i]);
for (int i = 0; i < _numLstmLayers; i++)
_lstmLayers.Add(layers[3 + i]);
_outputLayer = layers[3 + _numLstmLayers];
_lstmLayers.Add(Layers[3 + i]);
_outputLayer = Layers[3 + _numLstmLayers];

// Materialize the lazy LSTM weights now (their feature dim is known:
// convFilters into the first LSTM, lstmHiddenDim thereafter). Without
// this the LSTM weights stay at zero size until the first forward, so a
// clone/serialize of a never-yet-run model would capture no weights and
// the clone would resolve fresh random weights — diverging from the
// original (Clone_ShouldProduceIdenticalOutput).
int lstmInputDim = _convFilters;
foreach (var lstm in _lstmLayers)
{
if (lstm is LayerBase<T> lb && !lb.IsShapeResolved)
{
lb.ResolveFromShape(new[] { 1, 1, lstmInputDim });
}
lstmInputDim = _lstmHiddenDim;
}

// The output dense layer is also lazy (input size = lstmHiddenDim, the
// last-timestep feature width). Resolve it now for the same reason as
// the LSTMs.
if (_outputLayer is LayerBase<T> outLb && !outLb.IsShapeResolved)
{
outLb.ResolveFromShape(new[] { 1, _lstmHiddenDim });
}
}

#endregion
Expand Down Expand Up @@ -522,36 +566,24 @@ void IVoiceActivityDetector<T>.ResetState()
/// <inheritdoc/>
protected override Tensor<T> PreprocessAudio(Tensor<T> rawAudio)
{
// Normalize audio to [-1, 1] range
// Silero VAD (Silero Team, 2021) consumes float PCM already scaled to
// [-1, 1]; it does NOT re-normalize each chunk by its own peak amplitude.
// A per-chunk max-abs normalization would make the model amplitude-blind
// (a loud and a quiet copy of the same clip would map to identical
// features) and collapse a constant signal to all-ones, which is neither
// paper-faithful nor desirable. Reshape the waveform to the
// [batch, channels, samples] layout the 1-D conv frontend expects and
// leave the sample values intact.
var samples = rawAudio.ToVector().ToArray();
double maxAbs = 0;

for (int i = 0; i < samples.Length; i++)
{
double absVal = Math.Abs(NumOps.ToDouble(samples[i]));
if (absVal > maxAbs) maxAbs = absVal;
}

var normalizedSamples = new T[samples.Length];
if (maxAbs > 0)
{
for (int i = 0; i < samples.Length; i++)
{
double normalized = NumOps.ToDouble(samples[i]) / maxAbs;
normalizedSamples[i] = NumOps.FromDouble(normalized);
}
}
else
{
Array.Copy(samples, normalizedSamples, samples.Length);
}

// Reshape to [batch, channels, samples] for Conv
var result = new Tensor<T>([1, 1, samples.Length]);
var resultVector = result.ToVector();
for (int i = 0; i < normalizedSamples.Length; i++)
// Write directly into the tensor's backing storage. Tensor.ToVector()
// returns a COPY, so assigning into that copy would leave `result` all
// zeros (the bug that made the conv frontend see a zero signal and emit
// a constant 0.5 for every input).
var resultSpan = result.Data.Span;
for (int i = 0; i < samples.Length; i++)
{
resultVector[i] = normalizedSamples[i];
resultSpan[i] = samples[i];
}

return result;
Expand Down Expand Up @@ -590,29 +622,62 @@ protected override Tensor<T> Forward(Tensor<T> input)

var output = input;

// Pass through conv layers
// Pass through 1-D conv layers: [batch, 1, samples] -> [batch, convFilters, T].
foreach (var layer in _convLayers)
{
output = layer.Forward(output);
}

// The conv stack emits [batch, channels, time]; the LSTM consumes a
// sequence [batch, time, features]. Transpose the channel and time axes
// so each timestep's convFilters-dim feature vector becomes the LSTM
// input (Silero Team, 2021 — conv frontend feeding a recurrent core).
// Use the engine's tape-aware permute so the gradient flows back into
// the conv frontend during training (a plain Tensor.Transpose would
// detach the tape and leave the convs/LSTM untrained).
if (output.Rank == 3)
{
output = Engine.TensorPermute(output, [0, 2, 1]);
}

// Pass through LSTM layers
foreach (var layer in _lstmLayers)
{
output = layer.Forward(output);
}

// Take the last timestep output and pass through dense layer
// Take the last timestep and pass through the dense layer. Use the
// tape-aware axis slice ([batch, seq, hidden] -> [batch, hidden]) so the
// gradient propagates back through the LSTM and conv frontend.
if (_outputLayer is not null)
{
// Get last timestep: shape [batch, hidden] from [batch, seq, hidden]
var lastTimestep = ExtractLastTimestep(output);
var lastTimestep = output.Rank == 3
? Engine.TensorSliceAxis(output, axis: 1, index: output.Shape[1] - 1)
: output;
output = _outputLayer.Forward(lastTimestep);
}

return output;
}

/// <inheritdoc/>
/// <remarks>
/// SileroVad's forward is the conv frontend → axis transpose → LSTM →
/// last-timestep → dense pipeline in <see cref="Forward"/>, not a sequential
/// pass over the flat <c>Layers</c> list (which would feed the conv output
/// straight into the LSTM with the wrong axis order and skip the
/// last-timestep reduction). Route the training forward through the real
/// pipeline so the gradient tape records the actual operations.
/// </remarks>
public override Tensor<T> ForwardForTraining(Tensor<T> input)
{
// Mirror Predict: normalize + reshape the raw waveform to [1, 1, samples]
// before running the conv frontend. Without this the training forward
// would feed the un-preprocessed input straight into the first 1-D conv
// (channel-count mismatch).
return Forward(PreprocessAudio(input));
}

/// <summary>
/// Extracts the last timestep from a sequence tensor.
/// </summary>
Expand All @@ -638,6 +703,55 @@ private Tensor<T> ExtractLastTimestep(Tensor<T> sequenceOutput)
return result;
}

/// <inheritdoc/>
/// <remarks>
/// The base implementation runs the flat <c>Layers</c> list sequentially on
/// the raw input, which neither preprocesses the waveform nor applies the
/// conv→LSTM axis transpose, so it fails on SileroVad's custom pipeline.
/// Capture activations along the model's actual forward path instead.
/// </remarks>
public override Dictionary<string, Tensor<T>> GetNamedLayerActivations(Tensor<T> input)
{
var activations = new Dictionary<string, Tensor<T>>();
if (!_useNativeMode)
{
return activations;
}

var current = PreprocessAudio(input);
int idx = 0;

foreach (var layer in _convLayers)
{
current = layer.Forward(current);
activations[$"Layer_{idx}_{layer.GetType().Name}"] = current.Clone();
idx++;
}

if (current.Rank == 3)
{
current = Engine.TensorPermute(current, [0, 2, 1]);
}

foreach (var layer in _lstmLayers)
{
current = layer.Forward(current);
activations[$"Layer_{idx}_{layer.GetType().Name}"] = current.Clone();
idx++;
}

if (_outputLayer is not null)
{
var lastTimestep = current.Rank == 3
? Engine.TensorSliceAxis(current, axis: 1, index: current.Shape[1] - 1)
: current;
current = _outputLayer.Forward(lastTimestep);
activations[$"Layer_{idx}_{_outputLayer.GetType().Name}"] = current.Clone();
}

return activations;
}

/// <inheritdoc/>
public override void Train(Tensor<T> input, Tensor<T> expectedOutput)
{
Expand Down Expand Up @@ -722,6 +836,15 @@ protected override void DeserializeNetworkSpecificData(BinaryReader reader)
_ = reader.ReadInt32(); // _convFilters
_ = reader.ReadInt32(); // _lstmHiddenDim
_ = reader.ReadInt32(); // _numLstmLayers

// Deserialization has already rebuilt the canonical Layers list with the
// loaded weights. Re-point the cached conv/LSTM/output references at those
// layers; otherwise Forward would keep running the constructor's
// randomly-initialized layers and ignore the loaded weights.
if (_useNativeMode && Layers.Count >= 3 + _numLstmLayers + 1)
{
ExtractLayerReferences();
}
}

/// <inheritdoc/>
Expand Down
Loading
Loading