Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
77 commits
Select commit Hold shift + click to select a range
9fcda46
feat: add inference optimization operators
claude Nov 8, 2025
a8ea885
refactor: move simdkernels and platformdetector to aidotnet.tensors
ooples Dec 14, 2025
95848e0
refactor: move optimization utilities to aidotnet.tensors
ooples Dec 15, 2025
85219a3
fix: update tensor api from dimensions to shape in kernels
ooples Dec 15, 2025
c51f42f
chore: remove outdated examples from inferenceoptimization
ooples Dec 15, 2025
ebfba14
fix: correct simd api usage for runtime intrinsics compatibility
ooples Dec 15, 2025
9e0676c
feat: add data property for direct array access in tensor and vector
ooples Dec 15, 2025
e253be1
fix: resolve nullable reference type warnings in inference optimization
ooples Dec 15, 2025
a4cdc75
fix: enable ilgpu algorithms extension for roundtoeven support
ooples Dec 15, 2025
125271a
fix: correct gpu stress test performance assertion and update readme …
ooples Dec 15, 2025
4c1ba03
fix: address pr review comments for code quality improvements
ooples Dec 15, 2025
f13db53
fix: address pr review comments for inference optimization
ooples Dec 15, 2025
ff1b681
fix: remove unused scope stack from performanceprofiler
ooples Dec 15, 2025
204ea2c
fix: remove stubs and fix net471 compatibility issues
ooples Dec 15, 2025
3b6d719
refactor: use mathhelper.clamp for net471 compatibility in inferenceo…
ooples Dec 15, 2025
f7295c0
fix: update performance profiler
ooples Dec 15, 2025
b124ce0
fix: update custom operator registry
ooples Dec 15, 2025
7fcaecf
fix: update convolution kernel
ooples Dec 15, 2025
0627ef6
fix: integrate PR433 inference optimizations + address review
ooples Dec 16, 2025
dd0b0eb
feat: add speculation policy + continuous batcher support
ooples Dec 16, 2025
11869e6
fix: add inference diagnostics and stability guardrails
ooples Dec 16, 2025
50d8555
feat: optimize self-attention via cached attention rewrite
ooples Dec 16, 2025
946335d
feat: add kv-cache fp16 option
ooples Dec 16, 2025
13a2f98
fix: make cloning preserve layer parameters
ooples Dec 16, 2025
19d7e5e
fix: make speculative decoding draft selection non-throwing
ooples Dec 16, 2025
b532b9d
feat: add dynamic speculative decoding backoff
ooples Dec 16, 2025
e1a59e2
feat: add int8 kv-cache quantization option
ooples Dec 16, 2025
aa182ee
feat: route serving requests via adapter header
ooples Dec 16, 2025
69aabed
docs: update inference MVP plan with implemented hooks
ooples Dec 16, 2025
25c58c8
test: tighten speculative draft fallback assertion
ooples Dec 16, 2025
54a41a9
perf: group SIMD benchmarks by category
ooples Dec 16, 2025
67da3b2
docs: add phase mapping table for MVP sequencing
ooples Dec 16, 2025
e7889f7
fix: improve adapter model lookup error
ooples Dec 16, 2025
4df6b5e
fix: guard large unbatched predict requests
ooples Dec 16, 2025
758d799
fix: bound paged kv-cache sequence allocation retries
ooples Dec 16, 2025
54f6c91
fix: rescale int8 kv-cache across all batches
ooples Dec 16, 2025
70e0df0
fix: mark paged cached attention as inference-only
ooples Dec 16, 2025
9645c13
docs: document paged cached attention batch limitation
ooples Dec 16, 2025
47f7d60
perf: cache paged attention weights and reuse buffers
ooples Dec 16, 2025
5de9728
perf: use optimized output projection in paged attention fallback
ooples Dec 16, 2025
b38ba88
perf: use matmul for paged attention qkv
ooples Dec 16, 2025
9bc8312
fix: harden attention kernel shape validation
ooples Dec 16, 2025
65d719b
fix: validate conv2d kernel in-channels
ooples Dec 16, 2025
b94d8d3
fix: round-trip inference optimization config
ooples Dec 16, 2025
31ca20f
fix: stabilize paged attention allocation and tests
ooples Dec 16, 2025
6bc1f05
feat: add speculation policies and method hooks
ooples Dec 16, 2025
dfcf343
feat: add weight-only int8 dense quantization
ooples Dec 16, 2025
8976320
docs: address PR433 review feedback
ooples Dec 16, 2025
c0a12f8
feat: support Multi-LoRA deep clone and session isolation
ooples Dec 16, 2025
60e1211
test: cover int8 KV-cache quantization
ooples Dec 16, 2025
62babc6
docs: add strict PR433 phase audit and gap plan
ooples Dec 16, 2025
7afd67a
fix: avoid swallowing unexpected deserialization errors
ooples Dec 16, 2025
88c0b72
feat: integrate tree speculation and paged attention WOQ
ooples Dec 17, 2025
3ee1aaa
test: add phase 5/7/8 coverage
ooples Dec 17, 2025
ca9e6ea
fix: make inference diagnostics runtime-toggleable
ooples Dec 17, 2025
35377d0
test: close remaining PR433 mvp gaps
ooples Dec 17, 2025
1bb1a75
docs: update PR433 phase audit
ooples Dec 17, 2025
8de2fd7
fix: address PR433 review feedback
ooples Dec 17, 2025
d77dbcb
test: serialize diagnostics env var tests
ooples Dec 17, 2025
e1906ad
fix: tighten deserialization and speculation safety
ooples Dec 17, 2025
c07b191
fix: satisfy CodeQL unused-collection
ooples Dec 17, 2025
cb44170
fix: remove unsafe and random security hotspots
ooples Dec 18, 2025
ea38600
test: cover inference optimization kernels
ooples Dec 18, 2025
3ab670f
fix: validate convolution stride and kernel dims
ooples Dec 18, 2025
ca4bd00
fix: validate depthwise convolution params
ooples Dec 18, 2025
e2f6102
chore: trigger pr workflows
ooples Dec 18, 2025
f26c741
fix: avoid float equality in attention kernel
ooples Dec 18, 2025
c9c7bc8
ci: expand sonar coverage test projects
ooples Dec 18, 2025
bac85a9
ci: build tensors tests in sonar run
ooples Dec 19, 2025
8d40e54
fix: correct ARM dot product capability detection
ooples Dec 19, 2025
9b429fc
test: improve coverage for deserialization and platform detection
ooples Dec 19, 2025
83d1c0b
test: stabilize performance profiler singleton usage
ooples Dec 19, 2025
23592b0
refactor: move layer serialization metadata to LayerBase
ooples Dec 19, 2025
79857cb
docs: remove pr-specific planning markdowns
ooples Dec 19, 2025
acf2591
fix: propagate inference optimization config to meta learning results
ooples Dec 19, 2025
85bc75f
test: cover tensors cache, loop, and simd optimizers
ooples Dec 19, 2025
0f49c4a
perf: reduce quantization overhead in paged attention
ooples Dec 19, 2025
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion .github/workflows/sonarcloud.yml
Original file line number Diff line number Diff line change
Expand Up @@ -185,7 +185,9 @@ jobs:

- name: Run tests with coverage (net8.0)
run: |
dotnet test -c Release --framework net8.0 --no-build --filter "Category!=GPU&Category!=Integration" --collect:"XPlat Code Coverage" --settings coverlet.runsettings --logger "trx;LogFileName=test-results-net8.trx" --results-directory ./TestResults
dotnet test tests/AiDotNet.Tests/AiDotNetTests.csproj -c Release --framework net8.0 --no-build --filter "Category!=GPU&Category!=Integration" --collect:"XPlat Code Coverage" --settings coverlet.runsettings --logger "trx;LogFileName=test-results-aidotnet-net8.trx" --results-directory ./TestResults
dotnet test tests/AiDotNet.Serving.Tests/AiDotNet.Serving.Tests.csproj -c Release --framework net8.0 --no-build --filter "Category!=GPU&Category!=Integration" --collect:"XPlat Code Coverage" --settings coverlet.runsettings --logger "trx;LogFileName=test-results-serving-net8.trx" --results-directory ./TestResults
dotnet test tests/AiDotNet.Tensors.Tests/AiDotNet.Tensors.Tests.csproj -c Release --framework net8.0 --filter "Category!=GPU&Category!=Integration" --collect:"XPlat Code Coverage" --settings coverlet.runsettings --logger "trx;LogFileName=test-results-tensors-net8.trx" --results-directory ./TestResults

- name: End SonarCloud analysis
if: github.event_name != 'pull_request' || github.event.pull_request.changed_files <= 250
Expand Down
1 change: 1 addition & 0 deletions AiDotNetBenchmarkTests/AiDotNetBenchmarkTests.csproj
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
<ImplicitUsings>enable</ImplicitUsings>
<Nullable>enable</Nullable>
<LangVersion>latest</LangVersion>
<AllowUnsafeBlocks>true</AllowUnsafeBlocks>
<!-- CA1822: BenchmarkDotNet requires instance methods for benchmarks -->
<NoWarn>$(NoWarn);CA1822</NoWarn>
</PropertyGroup>
Expand Down
144 changes: 144 additions & 0 deletions AiDotNetBenchmarkTests/InferenceOptimization/AttentionBenchmark.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
using BenchmarkDotNet.Attributes;
using BenchmarkDotNet.Jobs;
using AiDotNet.InferenceOptimization;
using AiDotNet.InferenceOptimization.Kernels;
using AiDotNet.LinearAlgebra;
using System;

namespace AiDotNetBenchmarkTests.InferenceOptimization
{
/// <summary>
/// Benchmarks for fused attention kernel
/// </summary>
[SimpleJob(RuntimeMoniker.Net80)]
[MemoryDiagnoser]
[CsvExporter]
[HtmlExporter]
public class AttentionBenchmark
{
private Tensor<float> _q;
private Tensor<float> _k;
private Tensor<float> _v;
private AttentionKernel _attentionKernel;

[Params(64, 128, 256)]
public int SequenceLength { get; set; }

[Params(32, 64)]
public int FeatureDim { get; set; }

[GlobalSetup]
public void Setup()
{
OptimizationInitializer.Initialize(enableProfiling: false);

_attentionKernel = new AttentionKernel();

// Initialize Q, K, V tensors
_q = new Tensor<float>(new[] { 1, SequenceLength, FeatureDim });
_k = new Tensor<float>(new[] { 1, SequenceLength, FeatureDim });
_v = new Tensor<float>(new[] { 1, SequenceLength, FeatureDim });

for (int i = 0; i < _q.Data.Length; i++)
{
_q.Data[i] = DeterministicValue(i);
}

for (int i = 0; i < _k.Data.Length; i++)
{
_k.Data[i] = DeterministicValue(i + 1_000_000);
}

for (int i = 0; i < _v.Data.Length; i++)
{
_v.Data[i] = DeterministicValue(i + 2_000_000);
}
}

private static float DeterministicValue(int i)
{
// Stable deterministic value in [0, 1) without PRNG APIs (avoids security hotspot noise in analysis).
unchecked
{
uint x = (uint)(i * 1664525 + 1013904223);
return (x & 0x00FFFFFF) / 16777216f;
}
}

[Benchmark(Baseline = true)]
public Tensor<float> NaiveAttention()
{
// Naive implementation: QK^T, softmax, multiply by V
float scale = 1.0f / MathF.Sqrt(FeatureDim);

// Compute attention scores
var scores = new float[SequenceLength * SequenceLength];

for (int i = 0; i < SequenceLength; i++)
{
for (int j = 0; j < SequenceLength; j++)
{
float score = 0.0f;
for (int k = 0; k < FeatureDim; k++)
{
score += _q.Data[i * FeatureDim + k] * _k.Data[j * FeatureDim + k];
}
scores[i * SequenceLength + j] = score * scale;
}
}

// Apply softmax
for (int i = 0; i < SequenceLength; i++)
{
float maxVal = float.NegativeInfinity;
for (int j = 0; j < SequenceLength; j++)
{
if (scores[i * SequenceLength + j] > maxVal)
maxVal = scores[i * SequenceLength + j];
}

float sum = 0.0f;
for (int j = 0; j < SequenceLength; j++)
{
scores[i * SequenceLength + j] = MathF.Exp(scores[i * SequenceLength + j] - maxVal);
sum += scores[i * SequenceLength + j];
}

for (int j = 0; j < SequenceLength; j++)
{
scores[i * SequenceLength + j] /= sum;
}
}

// Multiply by V
var result = new Tensor<float>(new[] { 1, SequenceLength, FeatureDim });

for (int i = 0; i < SequenceLength; i++)
{
for (int j = 0; j < FeatureDim; j++)
{
float sum = 0.0f;
for (int k = 0; k < SequenceLength; k++)
{
sum += scores[i * SequenceLength + k] * _v.Data[k * FeatureDim + j];
}
result.Data[i * FeatureDim + j] = sum;
}
}

return result;
}

[Benchmark]
public Tensor<float> OptimizedAttention()
{
return _attentionKernel.Execute(_q, _k, _v);
}

[Benchmark]
public Tensor<float> MultiHeadAttention()
{
return _attentionKernel.MultiHeadAttention(_q, _k, _v, numHeads: 8);
}
}
}
92 changes: 92 additions & 0 deletions AiDotNetBenchmarkTests/InferenceOptimization/GemmBenchmark.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
using BenchmarkDotNet.Attributes;
using BenchmarkDotNet.Jobs;
using AiDotNet.InferenceOptimization;
using AiDotNet.InferenceOptimization.Kernels;
using AiDotNet.LinearAlgebra;
using System;

namespace AiDotNetBenchmarkTests.InferenceOptimization
{
/// <summary>
/// Benchmarks for GEMM (General Matrix Multiplication) kernel
/// Tests optimized implementation against naive implementation
/// </summary>
[SimpleJob(RuntimeMoniker.Net80)]
[MemoryDiagnoser]
[CsvExporter]
[HtmlExporter]
public class GemmBenchmark
{
private Tensor<float> _matrixA;
private Tensor<float> _matrixB;
private GemmKernel _gemmKernel;

[Params(64, 128, 256, 512, 1024)]
public int MatrixSize { get; set; }

[GlobalSetup]
public void Setup()
{
OptimizationInitializer.Initialize(enableProfiling: false);

_gemmKernel = new GemmKernel();

// Initialize matrices with deterministic data (avoids security hotspot noise in analysis)
_matrixA = new Tensor<float>(new[] { MatrixSize, MatrixSize });
_matrixB = new Tensor<float>(new[] { MatrixSize, MatrixSize });

for (int i = 0; i < _matrixA.Data.Length; i++)
{
_matrixA.Data[i] = DeterministicValue(i);
}

for (int i = 0; i < _matrixB.Data.Length; i++)
{
_matrixB.Data[i] = DeterministicValue(i + 1_000_000);
}
}

private static float DeterministicValue(int i)
{
unchecked
{
uint x = (uint)(i * 1664525 + 1013904223);
return (x & 0x00FFFFFF) / 16777216f;
}
}

[Benchmark(Baseline = true)]
public Tensor<float> NaiveGemm()
{
// Naive triple-nested loop implementation
var result = new Tensor<float>(new[] { MatrixSize, MatrixSize });

for (int i = 0; i < MatrixSize; i++)
{
for (int j = 0; j < MatrixSize; j++)
{
float sum = 0.0f;
for (int k = 0; k < MatrixSize; k++)
{
sum += _matrixA.Data[i * MatrixSize + k] * _matrixB.Data[k * MatrixSize + j];
}
result.Data[i * MatrixSize + j] = sum;
}
}

return result;
}

[Benchmark]
public Tensor<float> OptimizedGemm()
{
return _gemmKernel.Execute(_matrixA, _matrixB);
}

[Benchmark]
public Tensor<float> OptimizedGemmTranspose()
{
return _gemmKernel.GemmTransposeB(_matrixA, _matrixB);
}
}
}
Loading
Loading