From 8028d3eb3247c701c39064867591233fd61c45e3 Mon Sep 17 00:00:00 2001 From: SatyaKumarJ Date: Thu, 27 Feb 2025 21:09:50 -0800 Subject: [PATCH 1/4] Added ReduceMean --- .../webgpu/reduction/reduction_ops.cc | 131 ++++++++++++++++++ .../webgpu/reduction/reduction_ops.h | 57 ++++++++ .../webgpu/webgpu_execution_provider.cc | 8 +- 3 files changed, 192 insertions(+), 4 deletions(-) create mode 100644 onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc create mode 100644 onnxruntime/core/providers/webgpu/reduction/reduction_ops.h diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc new file mode 100644 index 0000000000000..7131d1a764cc8 --- /dev/null +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc @@ -0,0 +1,131 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. +// Licensed under the MIT License. + +#include "core/providers/webgpu/reduction/reduction_ops.h" +#include +#include "core/providers/webgpu/shader_helper.h" +#include "core/providers/webgpu/webgpu_supported_types.h" + +namespace onnxruntime { +namespace webgpu { + +#define REGISTER_UNARY_ELEMENTWISE_VERSIONED_KERNEL(ReduceOp, begin, end) \ + ONNX_OPERATOR_VERSIONED_KERNEL_EX( \ + ReduceOp, \ + kOnnxDomain, \ + begin, end, \ + kWebGpuExecutionProvider, \ + (*KernelDefBuilder::Create()).TypeConstraint("T", WebGpuSupportedNumberTypes()), \ + ReduceOp); + +#define REGISTER_UNARY_ELEMENTWISE_KERNEL(ReduceOp, version) \ + ONNX_OPERATOR_KERNEL_EX( \ + ReduceOp, \ + kOnnxDomain, \ + version, \ + kWebGpuExecutionProvider, \ + (*KernelDefBuilder::Create()).TypeConstraint("T", WebGpuSupportedNumberTypes()).InputMemoryType(OrtMemTypeCPUInput, 1), \ + ReduceOp); + +REGISTER_UNARY_ELEMENTWISE_VERSIONED_KERNEL(ReduceMean, 1, 10); +REGISTER_UNARY_ELEMENTWISE_VERSIONED_KERNEL(ReduceMean, 11, 12); +REGISTER_UNARY_ELEMENTWISE_VERSIONED_KERNEL(ReduceMean, 13, 17); +REGISTER_UNARY_ELEMENTWISE_KERNEL(ReduceMean, 18); + +Status ReduceKernelProgram::GenerateShaderCode(ShaderHelper& shader) const { + const auto& input = shader.AddInput("input", ShaderUsage::UseUniform | ShaderUsage::UseIndicesTypeAlias | ShaderUsage::UseValueTypeAlias); + const auto& output = shader.AddOutput("output", ShaderUsage::UseUniform | ShaderUsage::UseIndicesTypeAlias | ShaderUsage::UseValueTypeAlias); + bool reduce_on_all_axes = no_op_with_empty_axes_ == false && axes_.empty(); + std::string loop_header = code_[0]; + std::string loop_body = "let current_element: input_value_t = " + input.GetByIndices("input_indices") + ";\n" + code_[1]; + std::string loop_footer = code_[2]; + const auto input_rank = input.Rank(); + for (size_t i = 0, l = 0; i < input_rank; ++i) { + if (reduce_on_all_axes || std::find(axes_.begin(), axes_.end(), i) != axes_.end()) { + if (keepdims_) { + l++; + } + std::stringstream ss; + std::string index = "i" + std::to_string(i); + ss << "for (var " << index << " : u32 = 0; " << index << " < " << input.IndicesGet("uniforms.input_shape", i) << "; " << index << "++) {\n"; + ss << input.IndicesSet("input_indices", i, index) << ";\n"; + ss << loop_body << "\n"; + ss << "}\n"; + loop_body = ss.str(); + } else { + std::stringstream ss; + ss << loop_header << "\n"; + std::string index = "i" + std::to_string(i); + ss << "let " << index << " = " << output.IndicesGet("output_indices", l) << ";\n"; + ss << input.IndicesSet("input_indices", i, index) << ";\n"; + loop_header = ss.str(); + l++; + } + } + shader.MainFunctionBody() << shader.GuardAgainstOutOfBoundsWorkgroupSizes("uniforms.output_size") + << "let output_indices: output_indices_t = " << output.OffsetToIndices("global_idx") << ";\n" + << "var input_indices: input_indices_t = input_indices_t(0);\n" + << loop_header << loop_body << loop_footer; + shader.MainFunctionBody() << output.SetByOffset("global_idx", "output_value"); + return Status::OK(); +} + +template +Status ReduceKernel::ComputeInternal(ComputeContext& context) const { + const auto* input_tensor = context.Input(0); + std::vector input_axes; + // Check if axes input is provided and copy the axes values to input_axes + if (context.InputCount() > 1) { + ORT_ENFORCE(axes_.empty(), "Axes attribute may not be specified when axes input is also provided."); + const Tensor* axes_tensor = context.Input(1); + auto size = static_cast(axes_tensor->Shape()[0]); + const auto* data = axes_tensor->Data(); + input_axes.resize(size); + std::copy(data, data + size, input_axes.begin()); + } else { + input_axes.resize(axes_.size()); + std::copy(axes_.begin(), axes_.end(), input_axes.begin()); + } + const auto code = GetOpSpecificCode(input_tensor, input_axes); + // Compute output shape + std::vector output_shape; + for (int i = 0; i < input_tensor->Shape().NumDimensions(); ++i) { + if ((input_axes.empty() && !noop_with_empty_axes_) || std::find(input_axes.begin(), input_axes.end(), i) != input_axes.end()) { + if (keepdims_) { + output_shape.push_back(1); + } + } else { + output_shape.push_back(input_tensor->Shape()[i]); + } + } + TensorShape output_tensor_shape(output_shape); + int64_t output_size = output_tensor_shape.Size(); + ReduceKernelProgram program("ReduceMean", keepdims_, noop_with_empty_axes_, input_axes, code); + program.AddInput({input_tensor, ProgramTensorMetadataDependency::TypeAndRank}) + .AddOutput({context.Output(0, output_shape), ProgramTensorMetadataDependency::TypeAndRank}) + .SetDispatchGroupSize((output_size + WORKGROUP_SIZE - 1) / WORKGROUP_SIZE) + .AddUniformVariables({{static_cast(output_size)}}); + return context.RunProgram(program); +} + +ReduceOpSpecificCode ReduceMean::GetOpSpecificCode(const Tensor* input_tensor, const std::vector& axes) const { + const TensorShape& input_shape = input_tensor->Shape(); + size_t input_rank = input_shape.NumDimensions(); + size_t size = 1; + for (size_t i = 0; i < input_rank; ++i) { + if ((axes.empty() && !noop_with_empty_axes_) || std::find(axes.begin(), axes.end(), i) != axes.end()) { + size *= input_shape[i]; + } + } + std::stringstream ss; + ss << "let output_value = output_value_t(sum / f32(" << size << "));"; + ReduceOpSpecificCode code({"var sum = f32(0);", "sum += f32(current_element);", ss.str()}); + return code; +} + +Status ReduceMean::ComputeInternal(ComputeContext& ctx) const { + return ReduceKernel::ComputeInternal(ctx); +} + +} // namespace webgpu +} // namespace onnxruntime diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h new file mode 100644 index 0000000000000..a7a47ff5f311c --- /dev/null +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h @@ -0,0 +1,57 @@ +// Copyright (c) Microsoft Corporation. All rights reserved. +// Licensed under the MIT License. + +#pragma once +#include "core/common/optional.h" +#include "core/providers/webgpu/webgpu_supported_types.h" +#include "core/providers/webgpu/webgpu_kernel.h" +#include "core/providers/cpu/reduction/reduction_kernel_base.h" +#include "core/providers/webgpu/program.h" +#include "core/providers/webgpu/shader_helper.h" +namespace onnxruntime { +namespace webgpu { +// reduceOpSpecificCode is a 3-element array of strings that represent the op specific code for the reduce operation. +// The first element is the loop header, the second element is the loop body, and the third element is the loop footer. +// The loop header is the code that is executed before the loop starts. The loop body is the code that is executed for each element in the loop. +// The loop footer is the code that is executed after the loop ends. +typedef std::array ReduceOpSpecificCode; +class ReduceKernelProgram final : public Program { + public: + ReduceKernelProgram(std::string name, bool keepdims, bool no_op_with_empty_axes, const std::vector& axes, ReduceOpSpecificCode code) : Program{name}, keepdims_(keepdims), no_op_with_empty_axes_(no_op_with_empty_axes), axes_(axes.begin(), axes.end()), code_(code) {} + Status GenerateShaderCode(ShaderHelper& wgpuShaderModuleAddRef) const override; + WEBGPU_PROGRAM_DEFINE_UNIFORM_VARIABLES({"output_size", ProgramUniformVariableDataType::Uint32}); + private: + const bool keepdims_; + const bool no_op_with_empty_axes_; + InlinedVector axes_; + ReduceOpSpecificCode code_; +}; + +template +class ReduceKernel : public WebGpuKernel, public ReduceKernelBase { + protected: + using ReduceKernelBase::axes_; + using ReduceKernelBase::noop_with_empty_axes_; + using ReduceKernelBase::keepdims_; + using ReduceKernelBase::select_last_index_; + + ReduceKernel(const OpKernelInfo& info, std::string name, optional keepdims_override = {}) + : WebGpuKernel(info), + ReduceKernelBase(info, keepdims_override), + name_(name) { + } + Status ComputeInternal(ComputeContext& ctx) const; + virtual ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, const std::vector& axes) const = 0; + private: + std::string name_; +}; + +class ReduceMean final : public ReduceKernel { + public: + ReduceMean(const OpKernelInfo& info) : ReduceKernel(info, "ReduceMean") {} + ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, const std::vector& axes) const override; + Status ComputeInternal(ComputeContext& ctx) const; +}; + +} // namespace webgpu +} // namespace onnxruntime diff --git a/onnxruntime/core/providers/webgpu/webgpu_execution_provider.cc b/onnxruntime/core/providers/webgpu/webgpu_execution_provider.cc index d44cf4674d8a3..4950d94dea4c4 100644 --- a/onnxruntime/core/providers/webgpu/webgpu_execution_provider.cc +++ b/onnxruntime/core/providers/webgpu/webgpu_execution_provider.cc @@ -516,10 +516,10 @@ std::unique_ptr RegisterKernels() { // BuildKernelCreateInfo, // BuildKernelCreateInfo, - // BuildKernelCreateInfo, - // BuildKernelCreateInfo, - // BuildKernelCreateInfo, - // BuildKernelCreateInfo, + BuildKernelCreateInfo, + BuildKernelCreateInfo, + BuildKernelCreateInfo, + BuildKernelCreateInfo, BuildKernelCreateInfo, BuildKernelCreateInfo, From 292cfc3b54eb7da5c79c31d838b6bc15e349b989 Mon Sep 17 00:00:00 2001 From: SatyaKumarJ Date: Sat, 1 Mar 2025 12:10:36 -0800 Subject: [PATCH 2/4] Use unforms in the ReduceMean code. --- .../webgpu/reduction/reduction_ops.cc | 46 +++++++++++++------ .../webgpu/reduction/reduction_ops.h | 14 ++++-- 2 files changed, 41 insertions(+), 19 deletions(-) diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc index 7131d1a764cc8..cfe4313897210 100644 --- a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc @@ -3,6 +3,7 @@ #include "core/providers/webgpu/reduction/reduction_ops.h" #include +#include #include "core/providers/webgpu/shader_helper.h" #include "core/providers/webgpu/webgpu_supported_types.h" @@ -47,7 +48,7 @@ Status ReduceKernelProgram::GenerateShaderCode(ShaderHelper& shader) const { } std::stringstream ss; std::string index = "i" + std::to_string(i); - ss << "for (var " << index << " : u32 = 0; " << index << " < " << input.IndicesGet("uniforms.input_shape", i) << "; " << index << "++) {\n"; + ss << "for (var " << index << " : u32 = 0; " << index << " < " << input.IndicesGet("uniforms.input_shape", i) << "; " << index << "++) {\n"; ss << input.IndicesSet("input_indices", i, index) << ";\n"; ss << loop_body << "\n"; ss << "}\n"; @@ -73,7 +74,17 @@ Status ReduceKernelProgram::GenerateShaderCode(ShaderHelper& shader) const { template Status ReduceKernel::ComputeInternal(ComputeContext& context) const { const auto* input_tensor = context.Input(0); - std::vector input_axes; + InlinedVector input_axes; + auto rank = input_tensor->Shape().NumDimensions(); + auto transform_axis = [rank](int64_t axis) { + if (axis < 0) { + axis += rank; + } + if (axis < 0 || static_cast(axis) >= rank) { + ORT_THROW("Axes values must be in the range [-rank, rank-1]. Got: ", axis); + } + return static_cast(axis); + }; // Check if axes input is provided and copy the axes values to input_axes if (context.InputCount() > 1) { ORT_ENFORCE(axes_.empty(), "Axes attribute may not be specified when axes input is also provided."); @@ -81,12 +92,12 @@ Status ReduceKernel::ComputeInternal(ComputeContext& context) auto size = static_cast(axes_tensor->Shape()[0]); const auto* data = axes_tensor->Data(); input_axes.resize(size); - std::copy(data, data + size, input_axes.begin()); + std::transform(data, data + size, std::back_inserter(input_axes), transform_axis); } else { input_axes.resize(axes_.size()); - std::copy(axes_.begin(), axes_.end(), input_axes.begin()); + std::transform(axes_.begin(), axes_.end(), std::back_inserter(input_axes), transform_axis); } - const auto code = GetOpSpecificCode(input_tensor, input_axes); + const auto code = GetOpSpecificCode(input_tensor, input_axes.size()); // Compute output shape std::vector output_shape; for (int i = 0; i < input_tensor->Shape().NumDimensions(); ++i) { @@ -104,21 +115,28 @@ Status ReduceKernel::ComputeInternal(ComputeContext& context) program.AddInput({input_tensor, ProgramTensorMetadataDependency::TypeAndRank}) .AddOutput({context.Output(0, output_shape), ProgramTensorMetadataDependency::TypeAndRank}) .SetDispatchGroupSize((output_size + WORKGROUP_SIZE - 1) / WORKGROUP_SIZE) - .AddUniformVariables({{static_cast(output_size)}}); + .AddUniformVariables({{static_cast(output_size)}, + {static_cast(noop_with_empty_axes_ ? 1 : 0)}, + {input_axes}}); return context.RunProgram(program); } -ReduceOpSpecificCode ReduceMean::GetOpSpecificCode(const Tensor* input_tensor, const std::vector& axes) const { +ReduceOpSpecificCode ReduceMean::GetOpSpecificCode(const Tensor* input_tensor, size_t axes_size) const { const TensorShape& input_shape = input_tensor->Shape(); size_t input_rank = input_shape.NumDimensions(); - size_t size = 1; - for (size_t i = 0; i < input_rank; ++i) { - if ((axes.empty() && !noop_with_empty_axes_) || std::find(axes.begin(), axes.end(), i) != axes.end()) { - size *= input_shape[i]; - } - } std::stringstream ss; - ss << "let output_value = output_value_t(sum / f32(" << size << "));"; + ss << "size: f32 = 1.0;\n" + << "if (uniforms.noop_with_empty_axes == 0u && uniforms.axes_size == 0) {\n" + << " for (i: i32 = 0; i < " << input_rank << "; ++i) { \n" + << " size = size * " << GetElementAt("uniforms.input_shape", "i", input_rank) << ";\n" + << " }\n" + << "} else {\n" + << " for (i: i32 = 0; i < uniforms.axes_size; ++i) { \n" + << " index: i32 = " << GetElementAt("uniforms.axes", "i", axes_size) << ";\n" + << " size = size * " << GetElementAt("uniforms.input_shape", "index", input_rank) << ";\n" + << " }\n" + << "}\n" + << "let output_value = output_value_t(sum / f32(size));"; ReduceOpSpecificCode code({"var sum = f32(0);", "sum += f32(current_element);", ss.str()}); return code; } diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h index a7a47ff5f311c..9b6aa345d2431 100644 --- a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h @@ -17,13 +17,16 @@ namespace webgpu { typedef std::array ReduceOpSpecificCode; class ReduceKernelProgram final : public Program { public: - ReduceKernelProgram(std::string name, bool keepdims, bool no_op_with_empty_axes, const std::vector& axes, ReduceOpSpecificCode code) : Program{name}, keepdims_(keepdims), no_op_with_empty_axes_(no_op_with_empty_axes), axes_(axes.begin(), axes.end()), code_(code) {} + ReduceKernelProgram(std::string name, bool keepdims, bool no_op_with_empty_axes, const InlinedVector& axes, ReduceOpSpecificCode code) : Program{name}, keepdims_(keepdims), no_op_with_empty_axes_(no_op_with_empty_axes), axes_(axes.begin(), axes.end()), code_(code) {} Status GenerateShaderCode(ShaderHelper& wgpuShaderModuleAddRef) const override; - WEBGPU_PROGRAM_DEFINE_UNIFORM_VARIABLES({"output_size", ProgramUniformVariableDataType::Uint32}); + WEBGPU_PROGRAM_DEFINE_UNIFORM_VARIABLES({"output_size", ProgramUniformVariableDataType::Uint32}, + {"no_op_with_empty_axes", ProgramUniformVariableDataType::Uint32}, + {"axes", ProgramUniformVariableDataType::Uint32}); + private: const bool keepdims_; const bool no_op_with_empty_axes_; - InlinedVector axes_; + InlinedVector axes_; ReduceOpSpecificCode code_; }; @@ -41,7 +44,8 @@ class ReduceKernel : public WebGpuKernel, public ReduceKernelBase& axes) const = 0; + virtual ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, size_t axes_size) const = 0; + private: std::string name_; }; @@ -49,7 +53,7 @@ class ReduceKernel : public WebGpuKernel, public ReduceKernelBase { public: ReduceMean(const OpKernelInfo& info) : ReduceKernel(info, "ReduceMean") {} - ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, const std::vector& axes) const override; + ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, size_t axes_size) const override; Status ComputeInternal(ComputeContext& ctx) const; }; From aecf0f2cb7b6cb3fb196313d49d6911ec50bd99c Mon Sep 17 00:00:00 2001 From: SatyaKumarJ Date: Mon, 3 Mar 2025 13:56:58 -0800 Subject: [PATCH 3/4] Handle corner cases and shader code errors. --- .../webgpu/reduction/reduction_ops.cc | 47 ++++++++++++------- .../webgpu/reduction/reduction_ops.h | 5 +- 2 files changed, 34 insertions(+), 18 deletions(-) diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc index cfe4313897210..dc0897dadc096 100644 --- a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc @@ -4,6 +4,8 @@ #include "core/providers/webgpu/reduction/reduction_ops.h" #include #include +#include "core/framework/data_transfer_manager.h" +#include "core/providers/webgpu/data_transfer.h" #include "core/providers/webgpu/shader_helper.h" #include "core/providers/webgpu/webgpu_supported_types.h" @@ -41,7 +43,7 @@ Status ReduceKernelProgram::GenerateShaderCode(ShaderHelper& shader) const { std::string loop_body = "let current_element: input_value_t = " + input.GetByIndices("input_indices") + ";\n" + code_[1]; std::string loop_footer = code_[2]; const auto input_rank = input.Rank(); - for (size_t i = 0, l = 0; i < input_rank; ++i) { + for (int i = 0, l = 0; i < input_rank; ++i) { if (reduce_on_all_axes || std::find(axes_.begin(), axes_.end(), i) != axes_.end()) { if (keepdims_) { l++; @@ -91,17 +93,34 @@ Status ReduceKernel::ComputeInternal(ComputeContext& context) const Tensor* axes_tensor = context.Input(1); auto size = static_cast(axes_tensor->Shape()[0]); const auto* data = axes_tensor->Data(); - input_axes.resize(size); + input_axes.reserve(size); std::transform(data, data + size, std::back_inserter(input_axes), transform_axis); } else { - input_axes.resize(axes_.size()); + input_axes.reserve(axes_.size()); std::transform(axes_.begin(), axes_.end(), std::back_inserter(input_axes), transform_axis); } + if (input_axes.empty()) { + if (noop_with_empty_axes_ || rank == 0) { + // If axes is empty and noop_with_empty_axes_ is true, it is a no-op according to the spec + // If input tensor is a scalar, return the input tensor as is. + // This is not correct for ReduceLogSum and ReduceSumSquare + // TODO handle these cases separately. + auto output = context.Output(0, input_tensor->Shape()); + if (output->DataRaw() != input_tensor->DataRaw()) { + ORT_RETURN_IF_ERROR(Info().GetDataTransferManager().CopyTensor(*input_tensor, *output)); + } + return Status::OK(); + } else { + // If axes is empty and noop_with_empty_axes_ is false, it is a reduction over all axes + input_axes.resize(rank); + std::iota(input_axes.begin(), input_axes.end(), 0); + } + } const auto code = GetOpSpecificCode(input_tensor, input_axes.size()); // Compute output shape std::vector output_shape; - for (int i = 0; i < input_tensor->Shape().NumDimensions(); ++i) { - if ((input_axes.empty() && !noop_with_empty_axes_) || std::find(input_axes.begin(), input_axes.end(), i) != input_axes.end()) { + for (size_t i = 0; i < input_tensor->Shape().NumDimensions(); ++i) { + if (std::find(input_axes.begin(), input_axes.end(), i) != input_axes.end()) { if (keepdims_) { output_shape.push_back(1); } @@ -117,7 +136,9 @@ Status ReduceKernel::ComputeInternal(ComputeContext& context) .SetDispatchGroupSize((output_size + WORKGROUP_SIZE - 1) / WORKGROUP_SIZE) .AddUniformVariables({{static_cast(output_size)}, {static_cast(noop_with_empty_axes_ ? 1 : 0)}, - {input_axes}}); + {input_axes}, + {static_cast(input_axes.size())}}); + return context.RunProgram(program); } @@ -125,16 +146,10 @@ ReduceOpSpecificCode ReduceMean::GetOpSpecificCode(const Tensor* input_tensor, s const TensorShape& input_shape = input_tensor->Shape(); size_t input_rank = input_shape.NumDimensions(); std::stringstream ss; - ss << "size: f32 = 1.0;\n" - << "if (uniforms.noop_with_empty_axes == 0u && uniforms.axes_size == 0) {\n" - << " for (i: i32 = 0; i < " << input_rank << "; ++i) { \n" - << " size = size * " << GetElementAt("uniforms.input_shape", "i", input_rank) << ";\n" - << " }\n" - << "} else {\n" - << " for (i: i32 = 0; i < uniforms.axes_size; ++i) { \n" - << " index: i32 = " << GetElementAt("uniforms.axes", "i", axes_size) << ";\n" - << " size = size * " << GetElementAt("uniforms.input_shape", "index", input_rank) << ";\n" - << " }\n" + ss << "var size: u32 = 1;\n" + << "for (var i: u32 = 0; i < uniforms.axes_size; i += 1) { \n" + << " let index = " << GetElementAt("uniforms.axes", "i", axes_size) << ";\n" + << " size = size * " << GetElementAt("uniforms.input_shape", "index", input_rank) << ";\n" << "}\n" << "let output_value = output_value_t(sum / f32(size));"; ReduceOpSpecificCode code({"var sum = f32(0);", "sum += f32(current_element);", ss.str()}); diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h index 9b6aa345d2431..e93eb06f20886 100644 --- a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.h @@ -21,7 +21,8 @@ class ReduceKernelProgram final : public Program { Status GenerateShaderCode(ShaderHelper& wgpuShaderModuleAddRef) const override; WEBGPU_PROGRAM_DEFINE_UNIFORM_VARIABLES({"output_size", ProgramUniformVariableDataType::Uint32}, {"no_op_with_empty_axes", ProgramUniformVariableDataType::Uint32}, - {"axes", ProgramUniformVariableDataType::Uint32}); + {"axes", ProgramUniformVariableDataType::Uint32}, + {"axes_size", ProgramUniformVariableDataType::Uint32}); private: const bool keepdims_; @@ -54,7 +55,7 @@ class ReduceMean final : public ReduceKernel { public: ReduceMean(const OpKernelInfo& info) : ReduceKernel(info, "ReduceMean") {} ReduceOpSpecificCode GetOpSpecificCode(const Tensor* input_tensor, size_t axes_size) const override; - Status ComputeInternal(ComputeContext& ctx) const; + Status ComputeInternal(ComputeContext& ctx) const override; }; } // namespace webgpu From cdd0723a2a3a29c8608414dd3836db48b6b6f47a Mon Sep 17 00:00:00 2001 From: SatyaKumarJ Date: Mon, 3 Mar 2025 21:09:25 -0800 Subject: [PATCH 4/4] Added code to initialize input_indices in shader code. --- .../core/providers/webgpu/reduction/reduction_ops.cc | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc index dc0897dadc096..eb7903e7903b6 100644 --- a/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc +++ b/onnxruntime/core/providers/webgpu/reduction/reduction_ops.cc @@ -3,7 +3,6 @@ #include "core/providers/webgpu/reduction/reduction_ops.h" #include -#include #include "core/framework/data_transfer_manager.h" #include "core/providers/webgpu/data_transfer.h" #include "core/providers/webgpu/shader_helper.h" @@ -65,9 +64,14 @@ Status ReduceKernelProgram::GenerateShaderCode(ShaderHelper& shader) const { l++; } } + std::stringstream input_indices_init_value; + for (int i = 0; i < input_rank - 1; ++i) { + input_indices_init_value << "0, "; + } + input_indices_init_value << "0"; shader.MainFunctionBody() << shader.GuardAgainstOutOfBoundsWorkgroupSizes("uniforms.output_size") << "let output_indices: output_indices_t = " << output.OffsetToIndices("global_idx") << ";\n" - << "var input_indices: input_indices_t = input_indices_t(0);\n" + << "var input_indices: input_indices_t = input_indices_t(" << input_indices_init_value.str() << ");\n" << loop_header << loop_body << loop_footer; shader.MainFunctionBody() << output.SetByOffset("global_idx", "output_value"); return Status::OK();