Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .github/workflows/android.yml
Original file line number Diff line number Diff line change
Expand Up @@ -78,8 +78,8 @@ jobs:
run: |
set -e -x
BINARY_SIZE_THRESHOLD_ARGS=""
echo "Binary size threshold in bytes: 1440768"
BINARY_SIZE_THRESHOLD_ARGS="--threshold_size_in_bytes 1440768"
echo "Binary size threshold in bytes: 1443840"
BINARY_SIZE_THRESHOLD_ARGS="--threshold_size_in_bytes 1443840"

# Ensure ANDROID_NDK_HOME is available and get its real path
if [ -z "$ANDROID_NDK_HOME" ]; then
Expand Down
5 changes: 5 additions & 0 deletions .github/workflows/linux_minimal_build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,11 @@ jobs:
with:
node-version: 20

# This job builds with --use_coreml. The coremltools modelpackage sources include <uuid/uuid.h>,
# which is provided by uuid-dev on Ubuntu.
- name: Install libuuid development files
run: sudo apt-get update -y && sudo apt-get install -y uuid-dev

- name: Setup CCache
uses: actions/cache@v5
with:
Expand Down
7 changes: 7 additions & 0 deletions include/onnxruntime/core/framework/op_kernel.h
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,13 @@ class OpKernel {
return false;
}

// Only control flow kernels (If/Loop/Scan) legitimately carry subgraphs. SessionState
// finalization uses this to gate the downcast to controlflow::IControlFlowKernel without
// RTTI (onnxruntime_DISABLE_RTTI is ON by default, so dynamic_cast is unavailable).
[[nodiscard]] virtual bool IsControlFlowKernel() const {
return false;
}

[[nodiscard]] virtual Status ComputeAsync(_Inout_ OpKernelContext*, DoneCallback) const {
ORT_NOT_IMPLEMENTED(__FUNCTION__, " is not implemented");
}
Expand Down
31 changes: 27 additions & 4 deletions onnxruntime/core/framework/allocation_planner.cc
Original file line number Diff line number Diff line change
Expand Up @@ -496,14 +496,27 @@ class PlannerImpl {
return true;
}

/*! \brief Given a tensor-type, return the size of an element of the tensor.
/*! \brief Given a tensor-type, return the primitive element type of the tensor.
*/
static size_t GetElementSize(const DataType& tensor_type) {
static MLDataType GetPrimitiveElementType(const DataType& tensor_type) {
MLDataType ml_data_type = DataTypeImpl::GetDataType(*tensor_type);
const TensorTypeBase* tensor_type_base = ml_data_type->AsTensorType();
ORT_ENFORCE(nullptr != tensor_type_base);
MLDataType elt_type = tensor_type_base->GetElementType();
return elt_type->Size();
return tensor_type_base->GetElementType();
}

/*! \brief Given a tensor-type, return the size in bytes of the C++ carrier used for an element.
*/
static size_t GetElementSize(const DataType& tensor_type) {
return GetPrimitiveElementType(tensor_type)->Size();
}

/*! \brief Given a tensor-type, return how many logical (sub-byte) elements are packed into one
* carrier element. Returns 1 for regular types and >1 for packed sub-byte types (e.g. 2 for int4/uint4).
*/
static int32_t GetSubElemCount(const DataType& tensor_type) {
const auto* prim_type = GetPrimitiveElementType(tensor_type)->AsPrimitiveDataType();
return prim_type != nullptr ? prim_type->GetNumSubElems() : 1;
}

static bool SameSize(const TensorShapeProto& shape1, const onnxruntime::NodeArg& arg1,
Expand All @@ -515,6 +528,16 @@ class PlannerImpl {
bool is_type1_string = arg1.TypeAsProto()->tensor_type().elem_type() == ONNX_NAMESPACE::TensorProto_DataType_STRING;
bool is_type2_string = arg2.TypeAsProto()->tensor_type().elem_type() == ONNX_NAMESPACE::TensorProto_DataType_STRING;

// Packed sub-byte types (e.g. int4/uint4) share the same one-byte C++ carrier size as int8/uint8, but a
// carrier stores GetNumSubElems() logical elements, so the physical storage is ceil(N / sub_elems) bytes.
// Two tensors with equal logical shape and equal carrier size can therefore have different storage sizes
// (e.g. uint4[1024] needs 512 bytes while uint8[1024] needs 1024 bytes). Reusing the smaller buffer for the
// larger tensor produces a heap buffer overflow when the tensor is later written. Only treat the tensors as
// the same size when the sub-element packing density also matches, which guarantees identical storage bytes.
if (GetSubElemCount(ptype1) != GetSubElemCount(ptype2)) {
return false;
}

// sizeof(std::string) = sizeof(double) on gcc 4.8.x on CentOS. This causes the allocation planner to reuse
// a tensor of type double. This won't work for string tensors since they need to be placement new'ed.
// If either of the tensors is a string, don't treat them the same. Moreover, reusing a string tensor for a string
Expand Down
16 changes: 16 additions & 0 deletions onnxruntime/core/framework/execution_frame.cc
Original file line number Diff line number Diff line change
Expand Up @@ -697,6 +697,22 @@ Status ExecutionFrame::AllocateMLValueTensorPreAllocateBuffer(OrtValue& ort_valu
return ORT_MAKE_STATUS(ONNXRUNTIME, FAIL, message);
}
}

// Defense in depth: equal logical element counts do not guarantee equal physical storage. Packed sub-byte
// types (e.g. uint4[N] needs ceil(N/2) bytes) share the same carrier size as full-byte types (uint8[N] needs
// N bytes), so a reused buffer sized for the packed type is too small for the full-byte tensor and a later
// write would overflow it. Reject the reuse whenever the buffer cannot physically hold the requested tensor.
size_t required_storage_bytes = 0;
ORT_RETURN_IF_ERROR(
Tensor::CalculateTensorStorageSize(element_type, shape, /*alignment*/ 0, required_storage_bytes));
const size_t buffer_storage_bytes = reuse_tensor->SizeInBytes();
if (required_storage_bytes > buffer_storage_bytes) {
return ORT_MAKE_STATUS(
ONNXRUNTIME, FAIL, "Cannot re-use buffer: requested tensor needs ", required_storage_bytes,
" bytes of storage but the buffer being reused only has ", buffer_storage_bytes,
" bytes (buffer shape ", reuse_tensor->Shape(), ", requested shape ", shape,
"). This can happen when a packed sub-byte tensor is reused for a full-byte tensor of the same shape.");
}
}

void* reuse_buffer = reuse_tensor->MutableDataRaw();
Expand Down
4 changes: 2 additions & 2 deletions onnxruntime/core/framework/plugin_data_transfer.cc
Original file line number Diff line number Diff line change
Expand Up @@ -51,14 +51,14 @@ Status DataTransfer::CopyTensors(const std::vector<SrcDstPair>& src_dst_pairs) c
}

// optimized version for a single copy. see comments above in CopyTensors regarding the OrtValue usage and const_cast
Status DataTransfer::CopyTensorImpl(const Tensor& src_tensor, Tensor& dst_tensor, onnxruntime::Stream* /*stream*/) const {
Status DataTransfer::CopyTensorImpl(const Tensor& src_tensor, Tensor& dst_tensor, onnxruntime::Stream* stream) const {
OrtValue src, dst;
Tensor* src_tensor_ptr = const_cast<Tensor*>(&src_tensor);
src.Init(static_cast<void*>(src_tensor_ptr), ml_tensor_type, no_op_deleter);
dst.Init(static_cast<void*>(&dst_tensor), ml_tensor_type, no_op_deleter);
const OrtValue* src_ptr = &src;
OrtValue* dst_ptr = &dst;
OrtSyncStream* stream_ptr = nullptr; // static_cast<OrtSyncStream*>(stream);
OrtSyncStream* stream_ptr = reinterpret_cast<OrtSyncStream*>(stream);
auto* status = impl_.CopyTensors(&impl_, &src_ptr, &dst_ptr, &stream_ptr, 1);

return ToStatusAndRelease(status);
Expand Down
65 changes: 51 additions & 14 deletions onnxruntime/core/framework/session_state.cc
Original file line number Diff line number Diff line change
Expand Up @@ -1486,22 +1486,51 @@ static Status OuterScopeNodeArgLocationAccumulator(const SequentialExecutionPlan
// Process explicit inputs to the node
// (they are passed through as explicit subgraph inputs and hence requires a re-mapping of names
// to their corresponding names in the inner nested subgraph(s) held by the node)
const auto& subgraph_inputs = subgraph.GetInputs();
//
// Only nodes whose inputs map one-to-one onto the explicit subgraph inputs (Loop, Scan>=9) reach the
// re-mapping below; for other control flow nodes (e.g. If, or Scan opset 8) there is no such positional
// mapping and nothing is accumulated here.
if (IsNodeWhereNodeInputsAreSameAsExplicitSubgraphInputs(parent_node)) {
// The parent node's explicit inputs map positionally onto the subgraph's declared inputs.
//
// GetInputs() (inputs that are not backed by an initializer) is the list every downstream Loop/Scan path
// validates and indexes against - see Loop::Info, scan::detail::Info and scan_8/scan_9's
// ValidateSubgraphInput - so it is also the list used for the re-mapping here. Those downstream checks are
// ORT_ENFORCE based, which abort() in ORT_NO_EXCEPTIONS builds, so the mismatch has to be rejected with a
// clean status here before we get that far.
//
// A malformed/hostile model can declare a subgraph input that is also an initializer of the subgraph. Such
// an input is dropped from GetInputs() while remaining in GetInputsIncludingInitializers(), making
// GetInputs() shorter than the parent's input list. Without this check, indexing the shorter vector by the
// parent's input index below is an out-of-bounds read.
const auto num_parent_inputs = parent_node.InputDefs().size();
const auto& subgraph_inputs = subgraph.GetInputs();
if (subgraph_inputs.size() != num_parent_inputs) {
const char* initializer_hint =
subgraph.GetInputsIncludingInitializers().size() != subgraph_inputs.size()
? " The subgraph declares a graph input that is also one of its initializers, which is not supported"
" for the body of a Loop or Scan node."
: "";
return ORT_MAKE_STATUS(ONNXRUNTIME, INVALID_GRAPH,
"Subgraph input count does not match the number of inputs provided by the parent node '",
parent_node.Name(), "' (OpType: ", parent_node.OpType(), "). Parent provides ",
num_parent_inputs, " inputs but the subgraph has ", subgraph_inputs.size(),
" inputs that are not also initializers.", initializer_hint);
}

auto process_input = [&plan, &ort_value_name_to_idx_map, &outer_scope_arg_to_location_map,
&subgraph_inputs](const NodeArg& input, size_t arg_idx) {
const auto& name = input.Name();
OrtValueIndex index = -1;
ORT_RETURN_IF_ERROR(Index(ort_value_name_to_idx_map, name, index));
auto process_input = [&plan, &ort_value_name_to_idx_map, &outer_scope_arg_to_location_map,
&subgraph_inputs](const NodeArg& input, size_t arg_idx) {
const auto& name = input.Name();
OrtValueIndex index = -1;
ORT_RETURN_IF_ERROR(Index(ort_value_name_to_idx_map, name, index));

// Store the location of the outer scope value in the map using the subgraph input as the key
// as that will be the referenced name in the subgraph (i.e.) re-mapping of names is required
outer_scope_arg_to_location_map.insert({subgraph_inputs[arg_idx]->Name(), plan.GetLocation(index)});
// Store the location of the outer scope value in the map using the subgraph input as the key
// as that will be the referenced name in the subgraph (i.e.) re-mapping of names is required
outer_scope_arg_to_location_map.insert({subgraph_inputs[arg_idx]->Name(), plan.GetLocation(index)});

return Status::OK();
};
return Status::OK();
};

if (IsNodeWhereNodeInputsAreSameAsExplicitSubgraphInputs(parent_node)) {
return Node::ForEachWithIndex(parent_node.InputDefs(), process_input);
}

Expand Down Expand Up @@ -1801,8 +1830,16 @@ Status SessionState::FinalizeSessionStateImpl(const std::basic_string<PATH_CHAR_
auto* p_op_kernel = GetMutableKernel(node.Index());
ORT_ENFORCE(p_op_kernel);

// Downcast is safe, since only control flow nodes have subgraphs
// (node.GetAttributeNameToMutableSubgraphMap() is non-empty)
// A node carries a subgraph whenever it has a GRAPH-typed attribute, which Node::Init
// materializes with no schema gate. The downcast below is only valid for control flow
// kernels, so reject any other kernel that reached here with a subgraph instead of
// reading an out-of-bounds vtable slot.
ORT_RETURN_IF_NOT(p_op_kernel->IsControlFlowKernel(), "Node '", node.Name(),
"' (OpType: ", node.OpType(),
") has a subgraph but is not a control flow node.");

// Downcast is safe: only control flow nodes reach here (guarded above), and
// node.GetAttributeNameToMutableSubgraphMap() is non-empty.
auto& control_flow_kernel = static_cast<controlflow::IControlFlowKernel&>(*p_op_kernel);
ORT_RETURN_IF_ERROR(control_flow_kernel.SetupSubgraphExecutionInfo(*this, attr_name, subgraph_session_state));
}
Expand Down
35 changes: 26 additions & 9 deletions onnxruntime/core/graph/contrib_ops/bert_defs.cc
Original file line number Diff line number Diff line change
Expand Up @@ -234,7 +234,8 @@ void MultiHeadAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& c
void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx,
int past_key_index = -1,
int use_max_past_present_buffer = -1,
int output_qk_index = -1) {
int output_qk_index = -1,
int total_sequence_length_index = -1) {
// Type inference for outputs
ONNX_NAMESPACE::propagateElemTypeFromInputToOutput(ctx, 0, 0); // output

Expand Down Expand Up @@ -296,9 +297,13 @@ void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceConte

if (ctx.getNumOutputs() >= 3) { // has present output
int64_t total_sequence_length_value = 0;
const auto* total_sequence_length_data = ctx.getInputData(6);
const auto* total_sequence_length_data =
total_sequence_length_index >= 0 ? ctx.getInputData(total_sequence_length_index) : nullptr;
if (total_sequence_length_data != nullptr) {
const auto& data = ParseData<int32_t>(total_sequence_length_data);
if (data.size() != 1) {
fail_shape_inference("total_sequence_length input must contain a single element");
}
total_sequence_length_value = static_cast<int64_t>(data[0]);
}

Expand Down Expand Up @@ -447,13 +452,17 @@ void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceConte
void GroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx, int past_key_index, int qk_output_index) {
// TODO(aciddelgado): propagate output shapes depending if kv-share buffer is on or not
constexpr int use_max_past_present_buffer = -1;
BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index);
constexpr int total_sequence_length_index = 6;
BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index,
total_sequence_length_index);
}

void SparseAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx, int past_key_index) {
constexpr int use_max_past_present_buffer = 1;
constexpr int qk_output_index = -1;
BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index);
constexpr int total_sequence_length_index = 7;
BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index,
total_sequence_length_index);
}

constexpr const char* Attention_ver1_doc = R"DOC(
Expand Down Expand Up @@ -2316,6 +2325,11 @@ ONNX_MS_OPERATOR_SET_SCHEMA(
propagateElemTypeFromInputToOutput(ctx, 0, 0);
propagateElemTypeFromInputToOutput(ctx, 0, 1);

const int64_t ndim = getAttribute(ctx, "ndim", 1);
if (ndim < 1 || ndim > 3) {
fail_shape_inference("CausalConvWithState: ndim must be 1, 2, or 3, got ", ndim);
}

// Output 0: same shape as input (batch_size, channels, ...)
propagateShapeFromInputToOutput(ctx, 0, 0);

Expand All @@ -2326,13 +2340,16 @@ ONNX_MS_OPERATOR_SET_SCHEMA(
if (hasInputShape(ctx, 0) && hasInputShape(ctx, 1)) {
auto& input_shape = getInputShape(ctx, 0);
auto& weight_shape = getInputShape(ctx, 1);
if (input_shape.dim_size() < 2) {
fail_shape_inference("CausalConvWithState: input must have rank >= 2");
// Both are channels-first with rank ndim + 2: weight is (channels, 1, k_1, ..., k_ndim) and
// input is (batch_size, channels, d_1, ..., d_ndim).
if (weight_shape.dim_size() != ndim + 2) {
fail_shape_inference("CausalConvWithState: weight must have rank ndim + 2 (",
ndim + 2, "), got rank ", weight_shape.dim_size());
}
if (weight_shape.dim_size() < 2) {
fail_shape_inference("CausalConvWithState: weight must have rank >= 2");
if (input_shape.dim_size() != ndim + 2) {
fail_shape_inference("CausalConvWithState: input must have rank ndim + 2 (",
ndim + 2, "), got rank ", input_shape.dim_size());
}
int64_t ndim = getAttribute(ctx, "ndim", 1);
TensorShapeProto state_shape;
*state_shape.add_dim() = input_shape.dim(0); // batch_size
*state_shape.add_dim() = input_shape.dim(1); // channels
Expand Down
6 changes: 6 additions & 0 deletions onnxruntime/core/graph/contrib_ops/contrib_defs.cc
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,12 @@ void convTransposeWithDynamicPadsShapeInference(InferenceContext& ctx) {
}
kernel_shape.push_back(second_input_shape.dim(i).dim_value());
}
// A longer kernel_shape (W rank > X rank) overruns `dilations` in the loop right below;
// a shorter one (W rank < X rank) leaves `effective_kernel_shape` too short for the
// output-shape loop further below.
if (kernel_shape.size() != n_input_dims) {
return;
}
}

std::vector<int64_t> effective_kernel_shape = kernel_shape;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -77,12 +77,12 @@ void EmbedLayerNormalizationShapeInference(::ONNX_NAMESPACE::InferenceContext& c
!gamma_dims[0].has_dim_value() ||
gamma_shape.dim(0).dim_value() != hidden_size) {
fail_shape_inference(
"gamma should have 2 dimension, dimension size known, "
"gamma should have 1 dimension, dimension size known, "
"and same hidden size as word_embedding.");
}

auto& beta_shape = getInputShape(ctx, 6);
auto& beta_dims = gamma_shape.dim();
auto& beta_dims = beta_shape.dim();
if (beta_dims.size() != 1 ||
!beta_dims[0].has_dim_value() ||
beta_shape.dim(0).dim_value() != hidden_size) {
Expand Down
Loading
Loading