Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 25 additions & 0 deletions onnxruntime/core/optimizer/nchwc_transformer.cc
Original file line number Diff line number Diff line change
Expand Up @@ -358,7 +358,7 @@
}

const size_t nchwc_block_size = MlasNchwcGetBlockSize();
const int64_t nchwc_output_channels = (output_channels + nchwc_block_size - 1) & ~(nchwc_block_size - 1);

Check warning on line 361 in onnxruntime/core/optimizer/nchwc_transformer.cc

View workflow job for this annotation

GitHub Actions / build_x86_release

'~': zero extending 'size_t' to 'int64_t' of greater size

bool do_reorder_input = true;
bool reorder_filter_OIHWBo = false;
Expand Down Expand Up @@ -391,7 +391,7 @@
if ((input_channels % channel_alignment) != 0) {
return;
}
filter_input_channels = (input_channels + nchwc_block_size - 1) & ~(nchwc_block_size - 1);

Check warning on line 394 in onnxruntime/core/optimizer/nchwc_transformer.cc

View workflow job for this annotation

GitHub Actions / build_x86_release

'~': zero extending 'size_t' to 'int64_t' of greater size
}
}

Expand Down Expand Up @@ -533,6 +533,31 @@
return;
}

// Bail out for AveragePool with ceil_mode==1 && count_include_pad==1. The default float CPU
// AveragePool kernel was fixed (PR #29629) to divide the trailing ceil_mode window by its
// clamped size instead of the full kernel size. NchwcAveragePool routes to
// MlasAveragePoolingIncludePad, which still uses the full-kernel-size divisor and would
// silently reintroduce the wrong-average bug for optimized NCHWc graphs. Leaving this combo
// unconverted keeps it on the fixed CPU path; every other AveragePool (and all MaxPool /
// global pooling) conversion is unaffected, so there is no perf impact on the common case.
if (node.OpType() == "AveragePool") {
const NodeAttributes& attrs = node.GetAttributes();
const auto get_int_attr = [&attrs](const char* name) -> int64_t {
const auto it = attrs.find(name);
return (it != attrs.end() && it->second.type() == ONNX_NAMESPACE::AttributeProto_AttributeType_INT)
? it->second.i()
: 0;
};
// Gate count_include_pad as (!= 0) to match how PoolBase and the float (pool.cc) / fp16
// kernels treat it: any nonzero value enables include-pad. Using == 1 here would let an
// out-of-spec count_include_pad=2 model escape the bail-out, convert to NchwcAveragePool,
// and silently hit the buggy full-kernel divisor in the optimized graph. ceil_mode stays
// == 1 because the kernels gate the ceil-window fix on exactly ceil_mode == 1.
if (get_int_attr("ceil_mode") == 1 && get_int_attr("count_include_pad") != 0) {
return;
}
}

const size_t nchwc_block_size = MlasNchwcGetBlockSize();

const auto* input_type = input_defs[0]->TypeAsProto();
Expand Down Expand Up @@ -995,7 +1020,7 @@
bn_B.sub(bn_mean);

const size_t nchwc_block_size = MlasNchwcGetBlockSize();
const int64_t nchwc_channels = (channels + nchwc_block_size - 1) & ~(nchwc_block_size - 1);

Check warning on line 1023 in onnxruntime/core/optimizer/nchwc_transformer.cc

View workflow job for this annotation

GitHub Actions / build_x86_release

'~': zero extending 'size_t' to 'int64_t' of greater size

InlinedVector<float> padded_buffer(gsl::narrow<size_t>(nchwc_channels));

Expand Down
119 changes: 116 additions & 3 deletions onnxruntime/core/providers/cpu/fp16/fp16_pool.cc
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@

#include "core/framework/op_kernel.h"
#include "core/util/math.h"
#include "core/providers/cpu/nn/pool.h"
#include "core/providers/cpu/nn/pool_attributes.h"
#include "core/platform/threadpool.h"
#include "core/common/safeint.h"
Expand Down Expand Up @@ -36,6 +37,22 @@ class PoolFp16 : public OpKernel {
PoolAttributes pool_attrs_;
bool is_max_pool_; // either max pool or average pool
bool channels_last_;

private:
// Correct reference fallback for ceil_mode==1 && count_include_pad==1 AveragePool. The MLAS
// fp16 im2col path divides by the full kernel_size and cannot drop the ceil_mode phantom tail
// cells. Rather than duplicate the pooling loop, this casts the tensor MLFloat16->float,
// delegates to the shared float ComputeAveragePoolReferenceCompute (cpu/nn/pool.cc), then casts
// float->MLFloat16 with a single round at store. Accumulation happens in float inside the shared
// helper, so the result is identical to accumulate-in-float + round-once. Handles both NCHW and
// channels-last (NHWC) layouts (transposing to NCHW for the shared helper) and 1D/2D/3D. Zero
// MLAS edits.
Status ComputeAveragePoolFp16Reference(OpKernelContext* context,
const Tensor* X,
const TensorShapeVector& output_dims,
const TensorShapeVector& pads,
int64_t N,
int64_t C) const;
};

Status PoolFp16::Compute(OpKernelContext* context) const {
Expand Down Expand Up @@ -115,6 +132,17 @@ Status PoolFp16::Compute(OpKernelContext* context) const {
output_dims.push_back(C);
}

// The MLAS fp16 NHWC avg-pool (MlasNhwcAvgPool -> mlas/lib/pooling_fp16.cpp) divides every
// window by the full kernel_size and cannot drop the ceil_mode phantom tail cells, giving a
// too-small average for ceil_mode==1 && count_include_pad. Route only that combo to a correct
// float-accumulating reference loop; every other case keeps the fast MLAS im2col path. This
// mirrors the float fix in pool.cc (Pool<float, AveragePool>::Compute guard) so the fp16 and
// float AveragePool paths stay dtype-consistent.
if (!is_max_pool_ && pool_attrs_.ceil_mode == 1 && pool_attrs_.count_include_pad &&
!pool_attrs_.global_pooling) {
return ComputeAveragePoolFp16Reference(context, X, output_dims, pads, N, C);
}

const bool need_padding = !is_max_pool_ && pool_attrs_.count_include_pad;
std::vector<MLFloat16> padding_data;
if (need_padding) {
Expand Down Expand Up @@ -217,9 +245,94 @@ Status PoolFp16::Compute(OpKernelContext* context) const {
return Status::OK();
}

//
// Operator definitions
//
// fp16 AveragePool reference fallback for ceil_mode + count_include_pad. Casts the input
// MLFloat16->float (transposing NHWC->NCHW when channels_last_), runs the shared float pooling
// core (ComputeAveragePoolReferenceCompute in cpu/nn/pool.cc), then casts float->MLFloat16 into
// the output (transposing NCHW->NHWC back when channels_last_). A single round happens at the
// float->fp16 store; all accumulation is in float inside the shared core, so this is numerically
// equivalent to accumulate-in-float + round-once and stays in lockstep with the float path's
// clamped-divisor semantics. Reuses the float loop per xadupre's review request.
Status PoolFp16::ComputeAveragePoolFp16Reference(OpKernelContext* context,
const Tensor* X,
const TensorShapeVector& output_dims,
const TensorShapeVector& pads,
int64_t N,
int64_t C) const {
const TensorShape& input_shape = X->Shape();
const size_t spatial_dims = output_dims.size() - 2;
const size_t spatial_dim_start = channels_last_ ? 1 : 2;

// Build NCHW spatial extents plus flat spatial sizes for input and output.
int64_t input_image_size = 1;
int64_t output_image_size = 1;
TensorShapeVector x_nchw_dims({N, C});
TensorShapeVector out_nchw_dims({N, C});
for (size_t d = 0; d < spatial_dims; ++d) {
const int64_t in_d = input_shape[d + spatial_dim_start];
const int64_t out_d = output_dims[d + spatial_dim_start];
x_nchw_dims.push_back(in_d);
out_nchw_dims.push_back(out_d);
input_image_size *= in_d;
output_image_size *= out_d;
}
const TensorShape x_nchw_shape(x_nchw_dims);

AllocatorPtr alloc;
ORT_RETURN_IF_ERROR(context->GetTempSpaceAllocator(&alloc));

const size_t x_float_count = static_cast<size_t>(N) * static_cast<size_t>(C) *
static_cast<size_t>(input_image_size);
const size_t y_float_count = static_cast<size_t>(N) * static_cast<size_t>(C) *
static_cast<size_t>(output_image_size);
auto x_float = IAllocator::MakeUniquePtr<float>(alloc, x_float_count);
auto y_float = IAllocator::MakeUniquePtr<float>(alloc, y_float_count);

// Cast MLFloat16 -> float into an NCHW buffer (transpose from NHWC when channels_last_).
const auto* Xdata = X->Data<MLFloat16>();
float* x_ptr = x_float.get();
if (channels_last_) {
for (int64_t n = 0; n < N; ++n) {
for (int64_t c = 0; c < C; ++c) {
for (int64_t s = 0; s < input_image_size; ++s) {
x_ptr[(n * C + c) * input_image_size + s] =
Xdata[(n * input_image_size + s) * C + c].ToFloat();
}
}
}
} else {
for (size_t i = 0; i < x_float_count; ++i) {
x_ptr[i] = Xdata[i].ToFloat();
}
}

// Run the shared float pooling core. The guard in Compute() excludes global_pooling, so
// pool_attrs_ carries fully-populated kernel_shape/strides/dilations that match this call.
concurrency::ThreadPool* tp = context->GetOperatorThreadPool();
ORT_RETURN_IF_ERROR(ComputeAveragePoolReferenceCompute(x_ptr, y_float.get(), x_nchw_shape,
out_nchw_dims, pads, pool_attrs_, tp));

// Cast float -> MLFloat16 into the output (transpose back to NHWC when channels_last_).
// MLFloat16(float) is the single, final rounding step.
auto* Y = context->Output(0, output_dims);
auto* Ydata = Y->MutableData<MLFloat16>();
const float* y_ptr = y_float.get();
if (channels_last_) {
for (int64_t n = 0; n < N; ++n) {
for (int64_t c = 0; c < C; ++c) {
for (int64_t s = 0; s < output_image_size; ++s) {
Ydata[(n * output_image_size + s) * C + c] =
MLFloat16(y_ptr[(n * C + c) * output_image_size + s]);
}
}
}
} else {
for (size_t i = 0; i < y_float_count; ++i) {
Ydata[i] = MLFloat16(y_ptr[i]);
}
}

return Status::OK();
}
ONNX_CPU_OPERATOR_VERSIONED_TYPED_KERNEL(
MaxPool, 8, 11,
MLFloat16,
Expand Down
192 changes: 121 additions & 71 deletions onnxruntime/core/providers/cpu/nn/pool.cc
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,117 @@ inline static void RunLoop(concurrency::ThreadPool* tp, std::ptrdiff_t total_cha
concurrency::ThreadPool::TryParallelFor(tp, total_channels, task.Cost(), task);
}

// Core (non-MLAS) average-pooling loop, operating on caller-provided NCHW float buffers and
// pre-computed output_dims/pads. Correct for ceil_mode + count_include_pad because the
// AveragePool{1,2,3}DTask functors clamp the window end to input+tail_pad, excluding the
// ceil-mode phantom cells that MLAS's full-kernel divisor would count.
//
// This is the single source of truth for the reference average, shared by:
// - the float production path (AveragePoolV19 and the Pool<float, AveragePool> opset 7-18
// ceil_mode+count_include_pad fallback), via the ComputeAveragePoolReference wrapper below;
// - the fp16 fallback (PoolFp16::ComputeAveragePoolFp16Reference), which casts the tensor
// MLFloat16->float, calls this, then casts float->MLFloat16 (single round at store).
// Accumulation is in float regardless of caller dtype, so the fp16 path is numerically
// equivalent to accumulate-in-float + round-once.
Status ComputeAveragePoolReferenceCompute(const float* X_data, float* Y_data,
const TensorShape& x_shape,
gsl::span<const int64_t> output_dims,
gsl::span<const int64_t> pads,
const PoolAttributes& pool_attrs,
concurrency::ThreadPool* tp) {
const auto& kernel_shape = pool_attrs.kernel_shape;

const int64_t channels = x_shape[1];
const int64_t height = x_shape[2];
const int64_t width = kernel_shape.size() > 1 ? x_shape[3] : 1;
const int64_t depth = kernel_shape.size() > 2 ? x_shape[4] : 1;
const int64_t pooled_height = output_dims[2];
const int64_t pooled_width = kernel_shape.size() > 1 ? output_dims[3] : 1;
const int64_t pooled_depth = kernel_shape.size() > 2 ? output_dims[4] : 1;
const int64_t total_channels = x_shape[0] * channels;

const int64_t stride_h = pool_attrs.strides[0];
const int64_t stride_w = kernel_shape.size() > 1 ? pool_attrs.strides[1] : 1;
const int64_t stride_d = kernel_shape.size() > 2 ? pool_attrs.strides[2] : 1;

// p (LpPool p-norm) is unused by AveragePool functors; pass 0.
constexpr int64_t p = 0;

switch (kernel_shape.size()) {
case 1: {
int64_t x_step = height;
int64_t y_step = pooled_height;
const int64_t dilation_h = pool_attrs.dilations[0];
RunLoop<AveragePool1DTask<float>>(tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step, dilation_h, pooled_height, stride_h,
height, kernel_shape, pads, pool_attrs.count_include_pad, p});
break;
}
case 2: {
int64_t x_step = height * width;
int64_t y_step = pooled_height * pooled_width;
const int64_t dilation_h = pool_attrs.dilations[0];
const int64_t dilation_w = pool_attrs.dilations[1];
RunLoop<AveragePool2DTask<float>>(
tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step, dilation_h, dilation_w, pooled_height, pooled_width, stride_h,
stride_w, height, width, kernel_shape, pads, pool_attrs.count_include_pad, p});
break;
}
case 3: {
int64_t x_step = height * width * depth;
int64_t y_step = pooled_height * pooled_width * pooled_depth;
const int64_t dilation_h = pool_attrs.dilations[0];
const int64_t dilation_w = pool_attrs.dilations[1];
const int64_t dilation_d = pool_attrs.dilations[2];
RunLoop<AveragePool3DTask<float>>(tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step, dilation_h, dilation_w, dilation_d,
pooled_height, pooled_width, pooled_depth, stride_h, stride_w, stride_d,
height, width, depth, kernel_shape, pads, pool_attrs.count_include_pad, p});
break;
}
default:
return Status(ONNXRUNTIME, INVALID_ARGUMENT,
"Unsupported kernel dimension : " + std::to_string(kernel_shape.size()));
}
return Status::OK();
}

// Context wrapper for the float production path (AveragePoolV19, opset >= 19; and the
// Pool<float, AveragePool> opset 7-18 ceil_mode+count_include_pad fallback). Pulls X/Y from
// the kernel context, mirrors PoolBase's rank checks, computes output_dims/pads, then runs the
// shared float compute. Behavior-preserving: identical geometry + functor dispatch as before.
template <typename T>
Status ComputeAveragePoolReference(OpKernelContext* context,
const PoolAttributes& pool_attrs,
concurrency::ThreadPool* tp) {
ORT_ENFORCE(!pool_attrs.global_pooling,
"ComputeAveragePoolReference requires non-global pooling; "
"strides/dilations are not populated for global pooling.");

const auto* X = context->Input<Tensor>(0);
const TensorShape& x_shape = X->Shape();
ORT_RETURN_IF_NOT(x_shape.NumDimensions() >= 3, "Input dimension cannot be less than 3.");

// Mirror the rank checks PoolBase::Compute performs before SetOutputSize. This function is
// reached directly from the Pool<float, AveragePool>::Compute ceil_mode+count_include_pad
// guard (and from AveragePoolV19), bypassing PoolBase::Compute, so without these a
// rank-mismatched model would reach SetOutputSize/InferOutputSize and read out of bounds.
const size_t pooling_dims = x_shape.NumDimensions() - 2;
if (pooling_dims > 3) {
return Status(ONNXRUNTIME, INVALID_ARGUMENT, "Unsupported pooling size.");
}
ORT_RETURN_IF_NOT(pooling_dims == pool_attrs.kernel_shape.size(),
"kernel_shape num_dims is not compatible with X num_dims.");

auto pads = pool_attrs.pads;
auto output_dims = pool_attrs.SetOutputSize(x_shape, x_shape[1], &pads);
Tensor* Y = context->Output(0, output_dims);

return ComputeAveragePoolReferenceCompute(X->Data<T>(), Y->MutableData<T>(), x_shape,
output_dims, pads, pool_attrs, tp);
}

template <typename T, typename PoolType>
Status Pool<T, PoolType>::Compute(OpKernelContext* context) const {
concurrency::ThreadPool* tp = context->GetOperatorThreadPool();
Expand Down Expand Up @@ -152,6 +263,15 @@ Status Pool<float, MaxPool<1 /*VERSION*/>>::Compute(OpKernelContext* context) co

template <>
Status Pool<float, AveragePool>::Compute(OpKernelContext* context) const {
// MLAS's include-pad path divides by the full kernel size and cannot drop the
// ceil_mode phantom tail cells, giving a wrong average for
// ceil_mode=1 && count_include_pad. Route only that combo to the reference loop,
// which clamps the window to input+tail_pad. Everything else keeps the MLAS fast path.
// Global pooling is excluded because it has no ceil/pads (so it is already correct) and
// its strides[] are not populated, which ComputeAveragePoolReference requires.
if (pool_attrs_.ceil_mode == 1 && pool_attrs_.count_include_pad && !pool_attrs_.global_pooling) {
return ComputeAveragePoolReference<float>(context, pool_attrs_, context->GetOperatorThreadPool());
}
return PoolBase::Compute(context,
pool_attrs_.count_include_pad ? MlasAveragePoolingIncludePad : MlasAveragePoolingExcludePad);
}
Expand Down Expand Up @@ -251,77 +371,7 @@ Status MaxPoolV8::ComputeImpl(OpKernelContext* context) const {

template <typename T>
Status AveragePoolV19<T>::Compute(OpKernelContext* context) const {
concurrency::ThreadPool* tp = context->GetOperatorThreadPool();
bool need_dilation = false;
for (auto n : pool_attrs_.dilations) {
need_dilation |= n > 1;
}

const auto* X = context->Input<Tensor>(0);
const TensorShape& x_shape = X->Shape();

ORT_RETURN_IF_NOT(x_shape.NumDimensions() >= 3, "Input dimension cannot be less than 3.");

auto pads = pool_attrs_.pads;
auto kernel_shape = pool_attrs_.kernel_shape;

auto output_dims = pool_attrs_.SetOutputSize(x_shape, x_shape[1], &pads);
Tensor* Y = context->Output(0, output_dims);

const auto* X_data = X->Data<T>();
auto* Y_data = Y->MutableData<T>();

// The main loop
int64_t channels = x_shape[1];
int64_t height = x_shape[2];
int64_t width = kernel_shape.size() > 1 ? x_shape[3] : 1;
int64_t depth = kernel_shape.size() > 2 ? x_shape[4] : 1;
int64_t pooled_height = output_dims[2];
int64_t pooled_width = kernel_shape.size() > 1 ? output_dims[3] : 1;
int64_t pooled_depth = kernel_shape.size() > 2 ? output_dims[4] : 1;
const int64_t total_channels = x_shape[0] * channels;

switch (kernel_shape.size()) {
case 1: {
int64_t x_step = height;
int64_t y_step = pooled_height;
const int64_t dilation_h = pool_attrs_.dilations[0];

RunLoop<AveragePool1DTask<T>>(tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step, dilation_h, pooled_height, stride_h(),
height, kernel_shape, pads, pool_attrs_.count_include_pad, p_});
break;
}

case 2: {
int64_t x_step = height * width;
int64_t y_step = pooled_height * pooled_width;
const int64_t dilation_h = pool_attrs_.dilations[0];
const int64_t dilation_w = pool_attrs_.dilations[1];
RunLoop<AveragePool2DTask<T>>(
tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step, dilation_h, dilation_w, pooled_height, pooled_width, stride_h(),
stride_w(), height, width, kernel_shape, pads, pool_attrs_.count_include_pad, p_});
break;
}
case 3: {
int64_t x_step = height * width * depth;
int64_t y_step = pooled_height * pooled_width * pooled_depth;
const int64_t dilation_h = pool_attrs_.dilations[0];
const int64_t dilation_w = pool_attrs_.dilations[1];
const int64_t dilation_d = pool_attrs_.dilations[2];
RunLoop<AveragePool3DTask<T>>(tp, onnxruntime::narrow<size_t>(total_channels),
{X_data, Y_data, x_step, y_step,
dilation_h, dilation_w, dilation_d, pooled_height, pooled_width,
pooled_depth, stride_h(), stride_w(), stride_d(), height,
width, depth, kernel_shape, pads, pool_attrs_.count_include_pad, p_});
break;
}
default:
return Status(ONNXRUNTIME, INVALID_ARGUMENT, "Unsupported kernel dimension : " + std::to_string(kernel_shape.size()));
}

return Status::OK();
return ComputeAveragePoolReference<T>(context, pool_attrs_, context->GetOperatorThreadPool());
}

template <typename T>
Expand Down
Loading
Loading