From 0e2ad1c961811f8968fa023eef3090377f576868 Mon Sep 17 00:00:00 2001 From: Wei Wang Date: Thu, 17 Sep 2026 15:17:41 +0800 Subject: [PATCH 01/11] Fix out-of-bounds read in EmbedLayerNormalization shape inference for scalar beta (#32610) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Description `EmbedLayerNormalizationShapeInference` validated the `beta` input's rank using the wrong shape: `beta_dims` was aliased to `gamma_shape.dim()` instead of `beta_shape.dim()`. As a result, a rank-1 `gamma` would let the rank check pass even when `beta` is a 0-dim scalar, and execution reached `beta_shape.dim(0).dim_value()` — an unchecked protobuf repeated-field access on an empty `dim` list, causing an out-of-bounds read during session load, before provider/kernel assignment. ### Fix - Alias `beta_dims` to `beta_shape.dim()` so the rank check guards the correct input; a scalar `beta` is now rejected via `fail_shape_inference` instead of reaching the OOB access. This matches the adjacent `gamma` validation. - Correct the `gamma` error message ("2 dimension" → "1 dimension") to reflect the actual check (cherry picked from commit 7054657436874a0ddcdf95e4eb8aebd040f223f4) --- .../core/graph/contrib_ops/shape_inference_functions.cc | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/onnxruntime/core/graph/contrib_ops/shape_inference_functions.cc b/onnxruntime/core/graph/contrib_ops/shape_inference_functions.cc index f7e22608002ff..bbe99d27074d4 100644 --- a/onnxruntime/core/graph/contrib_ops/shape_inference_functions.cc +++ b/onnxruntime/core/graph/contrib_ops/shape_inference_functions.cc @@ -77,12 +77,12 @@ void EmbedLayerNormalizationShapeInference(::ONNX_NAMESPACE::InferenceContext& c !gamma_dims[0].has_dim_value() || gamma_shape.dim(0).dim_value() != hidden_size) { fail_shape_inference( - "gamma should have 2 dimension, dimension size known, " + "gamma should have 1 dimension, dimension size known, " "and same hidden size as word_embedding."); } auto& beta_shape = getInputShape(ctx, 6); - auto& beta_dims = gamma_shape.dim(); + auto& beta_dims = beta_shape.dim(); if (beta_dims.size() != 1 || !beta_dims[0].has_dim_value() || beta_shape.dim(0).dim_value() != hidden_size) { From 022c715d1a2ceddf327bd143a9a234d58da22926 Mon Sep 17 00:00:00 2001 From: shiyi Date: Fri, 18 Sep 2026 01:00:24 +0800 Subject: [PATCH 02/11] Reject nodes that feed inputs to a zero-input operator schema (#32633) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Description Add an explicit validation in `Node::UpdateInputArgCount()` to reject a node that supplies actual inputs while its bound operator/function schema declares no formal input parameters. ### Motivation and Context When `op.inputs()` is empty but the node has ≥1 input, the arg-count adjustment loop was skipped yet the trailing `input_arg_count.push_back(arg_count_left)` still ran unconditionally, producing `InputArgCount().size() == 1` against `op.inputs().size() == 0`. This size-invariant violation later caused an out-of-bounds read at `op.inputs()[i]` in `InferAndVerifyTypeMatch` during `Graph::Resolve()`. This is reachable only via a model-local function in a custom domain: for registered ops, `OpSchema::Verify` in the ONNX checker rejects the extra inputs first, but for custom-domain function calls the checker performs no arity validation. (cherry picked from commit 613bc03d7f73f5f43ba5ada2cb68a2d39e537a8e) --- onnxruntime/core/graph/graph.cc | 27 ++++++++++------- onnxruntime/test/framework/function_test.cc | 33 +++++++++++++++++++++ 2 files changed, 49 insertions(+), 11 deletions(-) diff --git a/onnxruntime/core/graph/graph.cc b/onnxruntime/core/graph/graph.cc index ed4ee2951adb4..f936ee6a681a2 100644 --- a/onnxruntime/core/graph/graph.cc +++ b/onnxruntime/core/graph/graph.cc @@ -1113,6 +1113,16 @@ Status Node::UpdateInputArgCount() { // Verify size of node arg count is same as input number in // operator definition. if (op.inputs().size() != definitions_.input_arg_count.size()) { + // A node cannot feed actual inputs to an operator/function whose schema + // declares no formal input parameters. + if (op.inputs().empty()) { + return ORT_MAKE_STATUS(ONNXRUNTIME, FAIL, + "This is an invalid model. Node (", name_, + ") has ", total_arg_count, + " input(s) but its operator schema (", op.Name(), + ") declares no inputs."); + } + // Adjust input arg count array with op definition // The adjustment will work as below, // In total, there're inputs, which @@ -1126,21 +1136,16 @@ Status Node::UpdateInputArgCount() { size_t m = 0; auto arg_count_left = total_arg_count; - if (!op.inputs().empty()) { - for (; m < op.inputs().size() - 1; ++m) { - if (arg_count_left > 0) { - input_arg_count.push_back(1); - arg_count_left--; - } else { - input_arg_count.push_back(0); - } + for (; m < op.inputs().size() - 1; ++m) { + if (arg_count_left > 0) { + input_arg_count.push_back(1); + arg_count_left--; + } else { + input_arg_count.push_back(0); } } // Set the arg count for the last input formal parameter. - // NOTE: in the case that there's no .input(...) defined - // in op schema, all input args will be fed as one input - // of the operator. input_arg_count.push_back(arg_count_left); graph_->SetGraphResolveNeeded(); diff --git a/onnxruntime/test/framework/function_test.cc b/onnxruntime/test/framework/function_test.cc index 351bfb6e2b0a8..cf52a27be7238 100644 --- a/onnxruntime/test/framework/function_test.cc +++ b/onnxruntime/test/framework/function_test.cc @@ -346,6 +346,39 @@ TEST(FunctionTest, CallInConditional) { Check(code, "x", {1.0, 2.0, 3.0}, "y", {6.0, 12.0, 18.0}); } +// A model-local function that declares zero inputs must not be invoked with +// actual inputs. +TEST(FunctionTest, RejectsZeroInputFunctionCalledWithInput) { + const char* code = R"( + < + ir_version: 8, + opset_import: [ "" : 16, "local" : 1 ] + > + agraph (float[N] x) => (float[1] y) + { + y = local.zerofun (x) + } + + < + opset_import: [ "" : 16 ], + domain: "local" + > + zerofun () => (ly) { + ly = Constant () + } + )"; + + std::string serialized_model; + ParseOnnxSource(code, serialized_model); + + SessionOptions session_options; + InferenceSession session_object{session_options, GetEnvironment()}; + std::stringstream sstr(serialized_model); + const auto status = session_object.Load(sstr); + ASSERT_FALSE(status.IsOK()); + EXPECT_THAT(status.ErrorMessage(), testing::HasSubstr("declares no inputs")); +} + TEST(FunctionTest, RejectsSelfRecursiveLocalFunction) { const char* code = R"( < From d69a6b807576fb5f2811e851ef1862e25015ec1e Mon Sep 17 00:00:00 2001 From: Bin Miao Date: Fri, 18 Sep 2026 01:21:52 +0800 Subject: [PATCH 03/11] Reject non-control-flow nodes that carry subgraphs during session state finalization (#32641) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Description `SessionState::FinalizeSessionStateImpl` iterates over the subgraphs attached to each node and unconditionally downcasts the node's kernel to `controlflow::IControlFlowKernel`, then calls `SetupSubgraphExecutionInfo` on it: ```cpp // Downcast is safe, since only control flow nodes have subgraphs auto& control_flow_kernel = static_cast(*p_op_kernel); ``` The "only control flow nodes have subgraphs" invariant is not enforced anywhere. Node::Init materializes a subgraph for any attribute of type GRAPH, with no schema gate, so an ordinary (non-control-flow) node can carry a subgraph. When it does, the downcast above is invalid: IControlFlowKernel adds a vtable slot that a plain OpKernel does not have, so the call reads an out-of-bounds vtable slot — an undefined-behavior type confusion that crashes (observed as an access violation inside FinalizeSessionStateImpl). This is reachable by loading a crafted/malformed model whose non-control-flow node has a GRAPH-typed attribute (for example, an op that permits unchecked attributes). ORT should reject such a model with a clear error instead of executing the bad cast. ### Fix Gate the downcast on a virtual predicate: - Add `OpKernel::IsControlFlowKernel()` returning `false` by default. - Override it to `true` on `IControlFlowKernel`, which covers If/Loop/Scan and their CUDA derivatives. - Check it in `FinalizeSessionStateImpl` before the cast and return an error otherwise. A virtual predicate is used rather than `dynamic_cast` because `onnxruntime_DISABLE_RTTI` is on by default. There is no behavior change for valid models: only the control-flow kernels inherit `IControlFlowKernel`, and they always return `true`. Adds `InferenceSessionTests.SubgraphAttributeOnNonControlFlowNodeIsRejected` test case, which builds a model whose non-control-flow node carries a `GRAPH` attribute and asserts that session initialization fails gracefully instead of triggering the downcast. Existing If/Loop/Scan subgraph tests cover the no-regression path. (cherry picked from commit b95cf781a26ff9b1bf71f3780281987e5d68a6bb) --- .../onnxruntime/core/framework/op_kernel.h | 7 ++ onnxruntime/core/framework/session_state.cc | 12 +- .../core/providers/cpu/controlflow/utils.h | 3 + .../test/framework/inference_session_test.cc | 103 ++++++++++++++++++ 4 files changed, 123 insertions(+), 2 deletions(-) diff --git a/include/onnxruntime/core/framework/op_kernel.h b/include/onnxruntime/core/framework/op_kernel.h index 42e8e9c5e3cbe..240ba3205189e 100644 --- a/include/onnxruntime/core/framework/op_kernel.h +++ b/include/onnxruntime/core/framework/op_kernel.h @@ -53,6 +53,13 @@ class OpKernel { return false; } + // Only control flow kernels (If/Loop/Scan) legitimately carry subgraphs. SessionState + // finalization uses this to gate the downcast to controlflow::IControlFlowKernel without + // RTTI (onnxruntime_DISABLE_RTTI is ON by default, so dynamic_cast is unavailable). + [[nodiscard]] virtual bool IsControlFlowKernel() const { + return false; + } + [[nodiscard]] virtual Status ComputeAsync(_Inout_ OpKernelContext*, DoneCallback) const { ORT_NOT_IMPLEMENTED(__FUNCTION__, " is not implemented"); } diff --git a/onnxruntime/core/framework/session_state.cc b/onnxruntime/core/framework/session_state.cc index 241eb8362ddfa..25bd50c0e56ae 100644 --- a/onnxruntime/core/framework/session_state.cc +++ b/onnxruntime/core/framework/session_state.cc @@ -1801,8 +1801,16 @@ Status SessionState::FinalizeSessionStateImpl(const std::basic_stringIsControlFlowKernel(), "Node '", node.Name(), + "' (OpType: ", node.OpType(), + ") has a subgraph but is not a control flow node."); + + // Downcast is safe: only control flow nodes reach here (guarded above), and + // node.GetAttributeNameToMutableSubgraphMap() is non-empty. auto& control_flow_kernel = static_cast(*p_op_kernel); ORT_RETURN_IF_ERROR(control_flow_kernel.SetupSubgraphExecutionInfo(*this, attr_name, subgraph_session_state)); } diff --git a/onnxruntime/core/providers/cpu/controlflow/utils.h b/onnxruntime/core/providers/cpu/controlflow/utils.h index 4dcf6025c04b4..e9a2ea5a6bea6 100644 --- a/onnxruntime/core/providers/cpu/controlflow/utils.h +++ b/onnxruntime/core/providers/cpu/controlflow/utils.h @@ -37,6 +37,9 @@ namespace controlflow { class IControlFlowKernel : public OpKernel { public: explicit IControlFlowKernel(const OpKernelInfo& info) : OpKernel(info) {} + + [[nodiscard]] bool IsControlFlowKernel() const final { return true; } + /** Setup information that is re-used each time to execute the subgraph. @param session_state SessionState for graph containing the control flow node @param attribute_name Control flow node's attribute name that contained the subgraph diff --git a/onnxruntime/test/framework/inference_session_test.cc b/onnxruntime/test/framework/inference_session_test.cc index 899f7dd6c57a0..321d438226487 100644 --- a/onnxruntime/test/framework/inference_session_test.cc +++ b/onnxruntime/test/framework/inference_session_test.cc @@ -3986,6 +3986,109 @@ TEST(InferenceSessionTests, CompileApiOutputHonorsOptimizationLevel) { EXPECT_EQ(default_counts.count("com.microsoft.BiasGelu") ? default_counts.at("com.microsoft.BiasGelu") : 0, 0); } #endif // !defined(DISABLE_CONTRIB_OPS) + +#if !defined(DISABLE_CONTRIB_OPS) +// SessionState::FinalizeSessionStateImpl unconditionally static_casts any node that carries +// a subgraph to controlflow::IControlFlowKernel and calls the control-flow-only virtual +// SetupSubgraphExecutionInfo on it. The "only control flow nodes have subgraphs" invariant it +// relies on is enforced nowhere: Node::Init builds a subgraph for ANY attribute of GRAPH type, +// with no schema gate. So a model whose ordinary node carries a GRAPH attribute reaches an +// out-of-bounds vtable slot read (IControlFlowKernel appends a vtable slot that a plain kernel +// does not have). +// +// SimplifiedLayerNormalization is a usable carrier: its schema (kOnnxDomain, since v1) sets +// AllowUncheckedAttributes(), ONNX defines no competing op of that name, and it has a plain +// (non-control-flow) CPU kernel. The CPU EP assigns it as a single node and keeps its GRAPH +// attribute, so finalize hits the bad cast. +TEST(InferenceSessionTests, SubgraphAttributeOnNonControlFlowNodeIsRejected) { + auto& logger = DefaultLoggingManager().DefaultLogger(); + + // Minimal, self-contained subgraph body: a single Constant producing one output and taking + // no inputs, so it needs no outer-scope wiring. Its contents are irrelevant to the defect. + ONNX_NAMESPACE::GraphProto forged_subgraph; + { + onnxruntime::Model sub_model("forged_subgraph", false, ModelMetaData(), PathString(), + IOnnxRuntimeOpSchemaRegistryList(), {{kOnnxDomain, 12}}, {}, logger); + Graph& sub_graph = sub_model.MainGraph(); + + ONNX_NAMESPACE::TypeProto float_tensor; + float_tensor.mutable_tensor_type()->set_elem_type(ONNX_NAMESPACE::TensorProto_DataType_FLOAT); + float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + auto& sub_out = sub_graph.GetOrCreateNodeArg("forged_sub_out", &float_tensor); + std::vector const_inputs; + std::vector const_outputs = {&sub_out}; + auto& const_node = sub_graph.AddNode("forged_const", "Constant", "", const_inputs, const_outputs); + + ONNX_NAMESPACE::TensorProto value; + value.set_name("forged_value"); + value.add_dims(1); + value.set_data_type(ONNX_NAMESPACE::TensorProto_DataType_FLOAT); + value.add_float_data(0.0f); + const_node.AddAttribute("value", value); + + std::vector sub_graph_outputs = {&sub_out}; + sub_graph.SetOutputs(sub_graph_outputs); + ASSERT_STATUS_OK(sub_graph.Resolve()); + forged_subgraph = sub_graph.ToGraphProto(); + } + + // Main graph: SimplifiedLayerNormalization(X, scale) -> Y, plus a forged GRAPH attribute that + // no schema forbids because the op allows unchecked attributes. + onnxruntime::Model model("subgraph_type_confusion", false, ModelMetaData(), PathString(), + IOnnxRuntimeOpSchemaRegistryList(), {{kOnnxDomain, 12}}, {}, logger); + Graph& graph = model.MainGraph(); + + ONNX_NAMESPACE::TypeProto x_type; + x_type.mutable_tensor_type()->set_elem_type(ONNX_NAMESPACE::TensorProto_DataType_FLOAT); + x_type.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(3); + x_type.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(4); + + ONNX_NAMESPACE::TypeProto vec_type; + vec_type.mutable_tensor_type()->set_elem_type(ONNX_NAMESPACE::TensorProto_DataType_FLOAT); + vec_type.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(4); + + auto& x = graph.GetOrCreateNodeArg("X", &x_type); + auto& scale = graph.GetOrCreateNodeArg("scale", &vec_type); + auto& y = graph.GetOrCreateNodeArg("Y", &x_type); + + std::vector inputs = {&x, &scale}; + std::vector outputs = {&y}; + auto& node = graph.AddNode("sln", "SimplifiedLayerNormalization", "carrier", inputs, outputs); + node.AddAttribute("axis", int64_t{-1}); + // The forged, non-control-flow subgraph. Node::Init materializes a Graph for it with no schema + // gate, and finalize then treats this ordinary node as a control flow kernel. + node.AddAttribute("forged_subgraph", forged_subgraph); + + std::vector graph_inputs = {&x, &scale}; + std::vector graph_outputs = {&y}; + graph.SetInputs(graph_inputs); + graph.SetOutputs(graph_outputs); + // Resolving OK proves the model is otherwise well-formed, so a later Initialize failure can + // only come from the type-confusion guard, not from a malformed graph. + ASSERT_STATUS_OK(graph.Resolve()); + + std::string serialized; + ASSERT_TRUE(model.ToProto().SerializeToString(&serialized)); + + SessionOptions so; + so.session_logid = "InferenceSessionTests.SubgraphAttributeOnNonControlFlowNodeIsRejected"; + // Keep optimizers out so the carrier node reaches finalize unchanged, mirroring the WebNN + // dispatch session which also runs with ORT_DISABLE_ALL. + so.graph_optimization_level = TransformerLevel::Default; + InferenceSession session_object{so, GetEnvironment()}; + ASSERT_STATUS_OK(session_object.RegisterExecutionProvider(DefaultCpuExecutionProvider())); + + // Mirrors the attacker's entry point (CreateSessionFromArray on attacker-controlled bytes). + ASSERT_STATUS_OK(session_object.Load(serialized.data(), static_cast(serialized.size()))); + + const auto status = session_object.Initialize(); + ASSERT_FALSE(status.IsOK()); + EXPECT_NE(status.ErrorMessage().find("has a subgraph but is not a control flow node"), + std::string::npos) + << "actual error: " << status.ErrorMessage(); +} +#endif // !defined(DISABLE_CONTRIB_OPS) #endif // !defined(ORT_MINIMAL_BUILD) } // namespace test From f50b4852b0ee810da80efb87f092b34540d13f08 Mon Sep 17 00:00:00 2001 From: Wei Wang Date: Sat, 19 Sep 2026 14:55:21 +0800 Subject: [PATCH 04/11] Fix heap buffer overflow from reusing a packed sub-byte buffer for a full-byte tensor (#32611) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Description The allocation planner's `SameSize()` decided buffer reuse by comparing the C++ carrier size of the element type (`elt_type->Size()`) plus the logical shape. Packed sub-byte types (`int4`/`uint4`) have the same 1-byte carrier size as `int8`/`uint8`, but one carrier stores 2 logical elements. As a result a `uint4[N]` tensor (physical `ceil(N/2)` bytes) and a `uint8[N]` tensor (physical `N` bytes) were treated as the same size, and the smaller `uint4` buffer was reused for the `uint8` output. Writing the `uint8` tensor into that half-sized buffer overflows it. ### Fix - `SameSize()` now also requires the sub-element packing density (`GetNumSubElems()`) to match, so tensors are only considered the same size when their physical storage bytes are actually equal. - `ExecutionFrame::AllocateMLValueTensorPreAllocateBuffer()` adds a defense-in-depth capacity check that rejects a reuse when the requested tensor needs more storage bytes than the buffer being reused (the previous check only compared logical element counts). - Added `AllocationPlannerTest.AvoidReuseOfPackedSubByteBufferForFullByteTensor`, which builds `X(float) → uint4 → float → uint8 → float` so a dead `uint4[1024]` output would be reused by a `uint8[1024]` output. Without the fix the planner reuses the buffer (`alloc_kind == kReuse`) and the output is corrupted; with the fix the test passes and the model runs correctly (cherry picked from commit 061eecbd8cda6245887687353547f328aa4a1947) --- .../core/framework/allocation_planner.cc | 31 +++++- onnxruntime/core/framework/execution_frame.cc | 16 +++ .../test/framework/allocation_planner_test.cc | 101 ++++++++++++++++++ .../test/framework/execution_frame_test.cc | 99 +++++++++++++++++ 4 files changed, 243 insertions(+), 4 deletions(-) diff --git a/onnxruntime/core/framework/allocation_planner.cc b/onnxruntime/core/framework/allocation_planner.cc index 006e6e0da8b56..5b737b1531e9d 100644 --- a/onnxruntime/core/framework/allocation_planner.cc +++ b/onnxruntime/core/framework/allocation_planner.cc @@ -496,14 +496,27 @@ class PlannerImpl { return true; } - /*! \brief Given a tensor-type, return the size of an element of the tensor. + /*! \brief Given a tensor-type, return the primitive element type of the tensor. */ - static size_t GetElementSize(const DataType& tensor_type) { + static MLDataType GetPrimitiveElementType(const DataType& tensor_type) { MLDataType ml_data_type = DataTypeImpl::GetDataType(*tensor_type); const TensorTypeBase* tensor_type_base = ml_data_type->AsTensorType(); ORT_ENFORCE(nullptr != tensor_type_base); - MLDataType elt_type = tensor_type_base->GetElementType(); - return elt_type->Size(); + return tensor_type_base->GetElementType(); + } + + /*! \brief Given a tensor-type, return the size in bytes of the C++ carrier used for an element. + */ + static size_t GetElementSize(const DataType& tensor_type) { + return GetPrimitiveElementType(tensor_type)->Size(); + } + + /*! \brief Given a tensor-type, return how many logical (sub-byte) elements are packed into one + * carrier element. Returns 1 for regular types and >1 for packed sub-byte types (e.g. 2 for int4/uint4). + */ + static int32_t GetSubElemCount(const DataType& tensor_type) { + const auto* prim_type = GetPrimitiveElementType(tensor_type)->AsPrimitiveDataType(); + return prim_type != nullptr ? prim_type->GetNumSubElems() : 1; } static bool SameSize(const TensorShapeProto& shape1, const onnxruntime::NodeArg& arg1, @@ -515,6 +528,16 @@ class PlannerImpl { bool is_type1_string = arg1.TypeAsProto()->tensor_type().elem_type() == ONNX_NAMESPACE::TensorProto_DataType_STRING; bool is_type2_string = arg2.TypeAsProto()->tensor_type().elem_type() == ONNX_NAMESPACE::TensorProto_DataType_STRING; + // Packed sub-byte types (e.g. int4/uint4) share the same one-byte C++ carrier size as int8/uint8, but a + // carrier stores GetNumSubElems() logical elements, so the physical storage is ceil(N / sub_elems) bytes. + // Two tensors with equal logical shape and equal carrier size can therefore have different storage sizes + // (e.g. uint4[1024] needs 512 bytes while uint8[1024] needs 1024 bytes). Reusing the smaller buffer for the + // larger tensor produces a heap buffer overflow when the tensor is later written. Only treat the tensors as + // the same size when the sub-element packing density also matches, which guarantees identical storage bytes. + if (GetSubElemCount(ptype1) != GetSubElemCount(ptype2)) { + return false; + } + // sizeof(std::string) = sizeof(double) on gcc 4.8.x on CentOS. This causes the allocation planner to reuse // a tensor of type double. This won't work for string tensors since they need to be placement new'ed. // If either of the tensors is a string, don't treat them the same. Moreover, reusing a string tensor for a string diff --git a/onnxruntime/core/framework/execution_frame.cc b/onnxruntime/core/framework/execution_frame.cc index 59efae597ceb2..5134bfc8d48a1 100644 --- a/onnxruntime/core/framework/execution_frame.cc +++ b/onnxruntime/core/framework/execution_frame.cc @@ -697,6 +697,22 @@ Status ExecutionFrame::AllocateMLValueTensorPreAllocateBuffer(OrtValue& ort_valu return ORT_MAKE_STATUS(ONNXRUNTIME, FAIL, message); } } + + // Defense in depth: equal logical element counts do not guarantee equal physical storage. Packed sub-byte + // types (e.g. uint4[N] needs ceil(N/2) bytes) share the same carrier size as full-byte types (uint8[N] needs + // N bytes), so a reused buffer sized for the packed type is too small for the full-byte tensor and a later + // write would overflow it. Reject the reuse whenever the buffer cannot physically hold the requested tensor. + size_t required_storage_bytes = 0; + ORT_RETURN_IF_ERROR( + Tensor::CalculateTensorStorageSize(element_type, shape, /*alignment*/ 0, required_storage_bytes)); + const size_t buffer_storage_bytes = reuse_tensor->SizeInBytes(); + if (required_storage_bytes > buffer_storage_bytes) { + return ORT_MAKE_STATUS( + ONNXRUNTIME, FAIL, "Cannot re-use buffer: requested tensor needs ", required_storage_bytes, + " bytes of storage but the buffer being reused only has ", buffer_storage_bytes, + " bytes (buffer shape ", reuse_tensor->Shape(), ", requested shape ", shape, + "). This can happen when a packed sub-byte tensor is reused for a full-byte tensor of the same shape."); + } } void* reuse_buffer = reuse_tensor->MutableDataRaw(); diff --git a/onnxruntime/test/framework/allocation_planner_test.cc b/onnxruntime/test/framework/allocation_planner_test.cc index 31eb7b7fad43b..05564384259d2 100644 --- a/onnxruntime/test/framework/allocation_planner_test.cc +++ b/onnxruntime/test/framework/allocation_planner_test.cc @@ -24,6 +24,7 @@ using json = nlohmann::json; #include "core/util/thread_utils.h" #include "test/test_environment.h" +#include "test/unittest_util/framework_test_utils.h" #include "test/util/include/asserts.h" #include "test/util/include/default_providers.h" #ifdef USE_CUDA @@ -2142,5 +2143,105 @@ TEST(AllocationPlannerTest, AvoidReuseOfBufferForNodeOutputWithNoConsumers) { } #endif +// Regression test for a heap buffer overflow caused by reusing a packed sub-byte buffer for a full-byte tensor. +// +// A packed sub-byte tensor (uint4[N]) needs ceil(N/2) storage bytes, while a full-byte tensor (uint8[N]) of the +// same logical shape needs N bytes. Both types have a one-byte C++ carrier size, so the allocation planner's +// SameSize() check used to treat them as the same size (carrier size 1 == 1 and identical shape) and let the +// uint8 output reuse the smaller uint4 buffer. Writing the uint8 tensor into that half-sized buffer then +// overflows it (CWE-131 -> CWE-787). The uint8 output must not reuse the uint4 buffer. +// +// Graph (opset 21): X(float) -Cast-> A(uint4) -Cast-> B(float) -Cast-> C(uint8) -Cast-> Y(float) +// A is fully consumed by the second Cast and freed, so before the fix the planner reused A's 512-byte buffer for +// the 1023-byte uint8 tensor C. +// +// The dimension is deliberately odd so the ceiling division in the packed storage size (ceil(1023 / 2) = 512, not +// 1023 / 2 = 511) is covered as well. +TEST(AllocationPlannerTest, AvoidReuseOfPackedSubByteBufferForFullByteTensor) { + constexpr int64_t kDim = 1023; + + auto make_tensor_type = [](TensorProto_DataType elem_type) { + TypeProto t; + t.mutable_tensor_type()->set_elem_type(elem_type); + t.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(kDim); + return t; + }; + + auto create_model = [&]() -> Model { + Model model("packed_subbyte_reuse", false, ModelMetaData(), PathString(), + IOnnxRuntimeOpSchemaRegistryList(), {{kOnnxDomain, 21}}, {}, + DefaultLoggingManager().DefaultLogger()); + Graph& graph = model.MainGraph(); + + TypeProto float_type = make_tensor_type(TensorProto_DataType_FLOAT); + TypeProto uint4_type = make_tensor_type(TensorProto_DataType_UINT4); + TypeProto uint8_type = make_tensor_type(TensorProto_DataType_UINT8); + + auto& X = graph.GetOrCreateNodeArg("X", &float_type); + auto& A = graph.GetOrCreateNodeArg("A", &uint4_type); // packed sub-byte buffer: ceil(1023/2) = 512 bytes + auto& B = graph.GetOrCreateNodeArg("B", &float_type); // consumes A -> A becomes dead here + auto& C = graph.GetOrCreateNodeArg("C", &uint8_type); // full-byte buffer: 1023 bytes + auto& Y = graph.GetOrCreateNodeArg("Y", &float_type); + + auto add_cast = [&graph](const std::string& name, NodeArg& in, NodeArg& out, TensorProto_DataType to) { + auto& node = graph.AddNode(name, "Cast", name, {&in}, {&out}); + node.AddAttribute("to", static_cast(to)); + }; + add_cast("cast_to_uint4", X, A, TensorProto_DataType_UINT4); + add_cast("cast_a_to_float", A, B, TensorProto_DataType_FLOAT); + add_cast("cast_to_uint8", B, C, TensorProto_DataType_UINT8); + add_cast("cast_c_to_float", C, Y, TensorProto_DataType_FLOAT); + + graph.SetInputs({&X}); + graph.SetOutputs({&Y}); + EXPECT_STATUS_OK(graph.Resolve()); + return model; + }; + + SessionOptions so; + // Keep memory reuse on (default) and avoid graph optimizations that could rewrite the Cast chain. + so.graph_optimization_level = TransformerLevel::Default; + InferenceSession sess{so, GetEnvironment()}; + + std::string serialized; + ASSERT_TRUE(create_model().ToProto().SerializeToString(&serialized)); + std::stringstream sstr(serialized); + ASSERT_STATUS_OK(sess.Load(sstr)); + ASSERT_STATUS_OK(sess.Initialize()); + + const auto& session_state = sess.GetSessionState(); + const auto& ort_value_index_map = session_state.GetOrtValueNameIdxMap(); + const SequentialExecutionPlan* plan = session_state.GetExecutionPlan(); + + OrtValueIndex a_index, c_index; + ASSERT_STATUS_OK(ort_value_index_map.GetIdx("A", a_index)); + ASSERT_STATUS_OK(ort_value_index_map.GetIdx("C", c_index)); + + // The uint8 tensor C must never be planned to reuse the (smaller) uint4 tensor A's buffer. + const auto& c_plan = plan->allocation_plan[c_index]; + const bool reuses_a = (c_plan.alloc_kind == AllocKind::kReuse) && (c_plan.reused_buffer == a_index); + EXPECT_FALSE(reuses_a) << "uint8[" << kDim << "] output must not reuse the uint4[" << kDim << "] buffer"; + + // The model must still run and produce correct results after round-tripping through uint4 and uint8. + constexpr float kValue = 3.0f; + std::vector dims{kDim}; + std::vector input_data(static_cast(kDim), kValue); + OrtValue input_value; + CreateMLValue(TestCPUExecutionProvider()->CreatePreferredAllocators()[0], dims, input_data, &input_value); + + NameMLValMap feeds{{"X", input_value}}; + std::vector output_names{"Y"}; + std::vector fetches; + ASSERT_STATUS_OK(sess.Run(feeds, output_names, &fetches)); + + ASSERT_EQ(fetches.size(), 1u); + const Tensor& out = fetches[0].Get(); + ASSERT_EQ(out.Shape().Size(), kDim); + const float* out_data = out.Data(); + for (int64_t i = 0; i < kDim; ++i) { + ASSERT_EQ(out_data[i], kValue) << "mismatch at index " << i; + } +} + } // namespace test } // namespace onnxruntime diff --git a/onnxruntime/test/framework/execution_frame_test.cc b/onnxruntime/test/framework/execution_frame_test.cc index bbfe22a2d4bc7..5070e90568f5a 100644 --- a/onnxruntime/test/framework/execution_frame_test.cc +++ b/onnxruntime/test/framework/execution_frame_test.cc @@ -3,6 +3,7 @@ #include "core/common/span_utils.h" #include "core/framework/execution_frame.h" +#include "core/framework/int4.h" #include "core/framework/op_kernel.h" #include "core/framework/session_state.h" #include "core/graph/model.h" @@ -188,6 +189,104 @@ TEST_F(ExecutionFrameTest, OutputShapeValidationTest) { ASSERT_STATUS_OK(frame.GetOrCreateNodeOutputMLValue(int(node->Index()), 1, &actual_shape_diff_from_input, p_ml_value, *node)); } +// Directly exercises the runtime capacity guard in ExecutionFrame::AllocateMLValueTensorPreAllocateBuffer. +// +// A packed sub-byte tensor (uint4[N]) has the same logical element count and the same one-byte C++ carrier size as +// a full-byte tensor (uint8[N]), but only needs ceil(N/2) storage bytes instead of N. Reusing the smaller uint4 +// buffer for the uint8 tensor would overflow it on the first write, so the reuse must be rejected. The opposite +// direction (a uint8 buffer reused by a uint4 tensor) is safe and must still be allowed. +// +// An odd dimension is used on purpose so the ceiling division in the storage size calculation is covered. +TEST_F(ExecutionFrameTest, PreAllocatedBufferTooSmallForSubByteTypeTest) { + constexpr int64_t kDim = 1023; + + onnxruntime::Model model("test", false, ModelMetaData(), PathString(), IOnnxRuntimeOpSchemaRegistryList(), + {{kOnnxDomain, 12}}, {}, DefaultLoggingManager().DefaultLogger()); + onnxruntime::Graph& graph = model.MainGraph(); + TypeProto tensor_float; + tensor_float.mutable_tensor_type()->set_elem_type(TensorProto_DataType_FLOAT); + onnxruntime::NodeArg input_def("X", &tensor_float), output_def("Y", &tensor_float); + + onnxruntime::Node* node = &graph.AddNode("node1", "Relu", "Relu operator", ArgMap{&input_def}, ArgMap{&output_def}); + node->SetExecutionProviderType(kCpuExecutionProvider); + ASSERT_STATUS_OK(graph.Resolve()); + + auto cpu_xp = CreateCPUExecutionProvider(); + auto xp_typ = cpu_xp->Type(); + ExecutionProviders execution_providers; + ASSERT_STATUS_OK(execution_providers.Add(xp_typ, std::move(cpu_xp))); + KernelRegistryManager kernel_registry_manager; + ASSERT_STATUS_OK(kernel_registry_manager.RegisterKernels(execution_providers)); + + DataTransferManager dtm; + ExternalDataLoaderManager edlm; + profiling::Profiler profiler; + + SessionOptions sess_options; + sess_options.enable_mem_pattern = true; + sess_options.execution_mode = ExecutionMode::ORT_SEQUENTIAL; + sess_options.use_deterministic_compute = false; + sess_options.enable_mem_reuse = true; + + SessionState state(graph, execution_providers, &tp_, nullptr, dtm, edlm, + DefaultLoggingManager().DefaultLogger(), profiler, sess_options); + + node->SetExecutionProviderType(xp_typ); + + ASSERT_STATUS_OK(state.FinalizeSessionState(ORT_TSTR(""), kernel_registry_manager)); + + const auto& memory_info = execution_providers.Get(xp_typ)->GetOrtDeviceByMemType(OrtMemTypeDefault); + const TensorShape shape(std::vector{kDim}); + MLDataType uint4_type = DataTypeImpl::GetType(); + MLDataType uint8_type = DataTypeImpl::GetType(); + + ASSERT_EQ(Tensor::CalculateTensorStorageSize(uint4_type, shape), static_cast((kDim + 1) / 2)); + ASSERT_EQ(Tensor::CalculateTensorStorageSize(uint8_type, shape), static_cast(kDim)); + + // A uint8 tensor must not re-use a uint4 buffer of the same logical shape: the buffer is only half the size. + { + vector outputs; + ExecutionFrame frame({}, {}, {}, outputs, {}, +#ifdef ORT_ENABLE_STREAM + {}, +#endif + state); + + int start_index = frame.GetNodeOffset(node->Index()); + ASSERT_EQ(start_index, 0); + + OrtValue& uint4_value = *frame.GetMutableNodeInputOrOutputMLValue(start_index); + ASSERT_STATUS_OK(frame.AllocateMLValueTensorSelfOwnBuffer(uint4_value, start_index, uint4_type, memory_info, shape)); + + OrtValue& uint8_value = *frame.GetMutableNodeInputOrOutputMLValue(start_index + 1); + const Status status = frame.AllocateMLValueTensorPreAllocateBuffer(uint8_value, start_index, uint8_type, + memory_info, shape); + ASSERT_FALSE(status.IsOK()); + EXPECT_THAT(status.ErrorMessage(), ::testing::HasSubstr("Cannot re-use buffer")); + EXPECT_FALSE(uint8_value.IsAllocated()); + } + + // The reverse direction is safe: a uint4 tensor fits in a uint8 buffer of the same logical shape. + { + vector outputs; + ExecutionFrame frame({}, {}, {}, outputs, {}, +#ifdef ORT_ENABLE_STREAM + {}, +#endif + state); + + int start_index = frame.GetNodeOffset(node->Index()); + + OrtValue& uint8_value = *frame.GetMutableNodeInputOrOutputMLValue(start_index); + ASSERT_STATUS_OK(frame.AllocateMLValueTensorSelfOwnBuffer(uint8_value, start_index, uint8_type, memory_info, shape)); + + OrtValue& uint4_value = *frame.GetMutableNodeInputOrOutputMLValue(start_index + 1); + ASSERT_STATUS_OK(frame.AllocateMLValueTensorPreAllocateBuffer(uint4_value, start_index, uint4_type, + memory_info, shape)); + EXPECT_EQ(uint4_value.Get().DataRaw(), uint8_value.Get().DataRaw()); + } +} + TEST_F(ExecutionFrameTest, FeedInDataTest) { onnxruntime::Model model("test", false, ModelMetaData(), PathString(), IOnnxRuntimeOpSchemaRegistryList(), std::unordered_map{{"", 10}}, {}, From ffbb06f9b0959d2562745ccb29b1c6487267bd88 Mon Sep 17 00:00:00 2001 From: Wei Wang Date: Sat, 19 Sep 2026 14:57:21 +0800 Subject: [PATCH 05/11] Fix out-of-bounds read when a Loop/Scan body input is also an initializer (#32609) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Problem `OuterScopeNodeArgLocationAccumulator` (session_state.cc) maps a Loop/Scan node's explicit inputs onto the subgraph's inputs using `GraphViewer::GetInputs()`, which *excludes* initializer-backed inputs. When a body declares an input that is also a body initializer, `GetInputs()` is shorter than the parent node's input list, so indexing `subgraph_inputs[arg_idx]` reads past the end of the vector and dereferences an invalid `NodeArg` pointer during session initialization. This causes a crash / heap out-of-bounds read (observable under ASan) when loading a malformed model containing such a Loop/Scan. ### Fix In the Loop/Scan≥9 branch of `OuterScopeNodeArgLocationAccumulator`: - If `subgraph.GetInputs().size()` does not equal the parent node's input count, return an `INVALID_GRAPH` status naming the parent node and both counts, instead of indexing. When the size difference is explained by initializer-backed inputs (`GetInputsIncludingInitializers()` is larger), the message adds a hint pointing at that as the cause. - Add `Loop.BodyInputAlsoInitializerIsRejected` test: builds a Loop whose body input `state_in` is also a body initializer, and asserts `InferenceSession::Initialize()` fails with `INVALID_GRAPH` and the expected message — i.e. rejected cleanly rather than crashing. The test asserts on the status code and message text on purpose, so that it keeps covering this path rather than silently falling through to the `ORT_ENFORCE` downstream. (cherry picked from commit 09dfa6ad06ed8072b2fbe57687d4f71c7914b025) --- onnxruntime/core/framework/session_state.cc | 53 +++++++-- .../providers/cpu/controlflow/loop_test.cc | 108 ++++++++++++++++++ 2 files changed, 149 insertions(+), 12 deletions(-) diff --git a/onnxruntime/core/framework/session_state.cc b/onnxruntime/core/framework/session_state.cc index 25bd50c0e56ae..5c7680f62dc97 100644 --- a/onnxruntime/core/framework/session_state.cc +++ b/onnxruntime/core/framework/session_state.cc @@ -1486,22 +1486,51 @@ static Status OuterScopeNodeArgLocationAccumulator(const SequentialExecutionPlan // Process explicit inputs to the node // (they are passed through as explicit subgraph inputs and hence requires a re-mapping of names // to their corresponding names in the inner nested subgraph(s) held by the node) - const auto& subgraph_inputs = subgraph.GetInputs(); + // + // Only nodes whose inputs map one-to-one onto the explicit subgraph inputs (Loop, Scan>=9) reach the + // re-mapping below; for other control flow nodes (e.g. If, or Scan opset 8) there is no such positional + // mapping and nothing is accumulated here. + if (IsNodeWhereNodeInputsAreSameAsExplicitSubgraphInputs(parent_node)) { + // The parent node's explicit inputs map positionally onto the subgraph's declared inputs. + // + // GetInputs() (inputs that are not backed by an initializer) is the list every downstream Loop/Scan path + // validates and indexes against - see Loop::Info, scan::detail::Info and scan_8/scan_9's + // ValidateSubgraphInput - so it is also the list used for the re-mapping here. Those downstream checks are + // ORT_ENFORCE based, which abort() in ORT_NO_EXCEPTIONS builds, so the mismatch has to be rejected with a + // clean status here before we get that far. + // + // A malformed/hostile model can declare a subgraph input that is also an initializer of the subgraph. Such + // an input is dropped from GetInputs() while remaining in GetInputsIncludingInitializers(), making + // GetInputs() shorter than the parent's input list. Without this check, indexing the shorter vector by the + // parent's input index below is an out-of-bounds read. + const auto num_parent_inputs = parent_node.InputDefs().size(); + const auto& subgraph_inputs = subgraph.GetInputs(); + if (subgraph_inputs.size() != num_parent_inputs) { + const char* initializer_hint = + subgraph.GetInputsIncludingInitializers().size() != subgraph_inputs.size() + ? " The subgraph declares a graph input that is also one of its initializers, which is not supported" + " for the body of a Loop or Scan node." + : ""; + return ORT_MAKE_STATUS(ONNXRUNTIME, INVALID_GRAPH, + "Subgraph input count does not match the number of inputs provided by the parent node '", + parent_node.Name(), "' (OpType: ", parent_node.OpType(), "). Parent provides ", + num_parent_inputs, " inputs but the subgraph has ", subgraph_inputs.size(), + " inputs that are not also initializers.", initializer_hint); + } - auto process_input = [&plan, &ort_value_name_to_idx_map, &outer_scope_arg_to_location_map, - &subgraph_inputs](const NodeArg& input, size_t arg_idx) { - const auto& name = input.Name(); - OrtValueIndex index = -1; - ORT_RETURN_IF_ERROR(Index(ort_value_name_to_idx_map, name, index)); + auto process_input = [&plan, &ort_value_name_to_idx_map, &outer_scope_arg_to_location_map, + &subgraph_inputs](const NodeArg& input, size_t arg_idx) { + const auto& name = input.Name(); + OrtValueIndex index = -1; + ORT_RETURN_IF_ERROR(Index(ort_value_name_to_idx_map, name, index)); - // Store the location of the outer scope value in the map using the subgraph input as the key - // as that will be the referenced name in the subgraph (i.e.) re-mapping of names is required - outer_scope_arg_to_location_map.insert({subgraph_inputs[arg_idx]->Name(), plan.GetLocation(index)}); + // Store the location of the outer scope value in the map using the subgraph input as the key + // as that will be the referenced name in the subgraph (i.e.) re-mapping of names is required + outer_scope_arg_to_location_map.insert({subgraph_inputs[arg_idx]->Name(), plan.GetLocation(index)}); - return Status::OK(); - }; + return Status::OK(); + }; - if (IsNodeWhereNodeInputsAreSameAsExplicitSubgraphInputs(parent_node)) { return Node::ForEachWithIndex(parent_node.InputDefs(), process_input); } diff --git a/onnxruntime/test/providers/cpu/controlflow/loop_test.cc b/onnxruntime/test/providers/cpu/controlflow/loop_test.cc index cbb7771d7f09e..09177a73615fa 100644 --- a/onnxruntime/test/providers/cpu/controlflow/loop_test.cc +++ b/onnxruntime/test/providers/cpu/controlflow/loop_test.cc @@ -830,6 +830,114 @@ TEST(Loop, Opset11WithNoVariadicInputsAndOutputs) { test.Run(OpTester::ExpectResult::kExpectSuccess, "", {kTensorrtExecutionProvider, kOpenVINOExecutionProvider}); } +// Regression test for an out-of-bounds read during session initialization when a Loop body declares an input +// that is ALSO a body initializer. Such a body has fewer entries in GraphViewer::GetInputs() (which excludes +// initializer-backed inputs) than the parent Loop's explicit inputs. Previously +// OuterScopeNodeArgLocationAccumulator() indexed GetInputs() using the parent's input index, reading past the +// end of the vector and dereferencing an invalid NodeArg pointer. +// +// The model must be rejected during InferenceSession::Initialize() with an INVALID_GRAPH status. Checking the +// status code and message matters here: the downstream Loop::Info validation of the same condition is +// ORT_ENFORCE based, which calls abort() in ORT_NO_EXCEPTIONS builds, so the test would silently stop covering +// the intended path if the rejection moved back there. +TEST(Loop, BodyInputAlsoInitializerIsRejected) { + auto create_body = []() { + Model body_model("Loop body with initializer-backed input", false, DefaultLoggingManager().DefaultLogger()); + auto& graph = body_model.MainGraph(); + + TypeProto int64_tensor; + int64_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_INT64); + int64_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + TypeProto bool_tensor; + bool_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_BOOL); + bool_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + TypeProto float_tensor; + float_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_FLOAT); + float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + // Body inputs: iter_num, cond_in, state_in (a loop-carried variable). + auto& iter_num_in = graph.GetOrCreateNodeArg("iter_num_in", &int64_tensor); + auto& cond_in = graph.GetOrCreateNodeArg("cond_in", &bool_tensor); + auto& state_in = graph.GetOrCreateNodeArg("state_in", &float_tensor); + + auto& cond_out = graph.GetOrCreateNodeArg("cond_out", &bool_tensor); + auto& state_out = graph.GetOrCreateNodeArg("state_out", &float_tensor); + + graph.AddNode("cond_identity", "Identity", "Forward cond_in to cond_out", {&cond_in}, {&cond_out}); + graph.AddNode("state_identity", "Identity", "Forward state_in to state_out", {&state_in}, {&state_out}); + + // Make state_in ALSO a body initializer. This is the malformed condition: state_in stays in + // GetInputsIncludingInitializers() but is dropped from GetInputs(). + TensorProto state_initializer; + state_initializer.set_name("state_in"); + state_initializer.set_data_type(TensorProto_DataType_FLOAT); + state_initializer.add_dims(1); + state_initializer.add_float_data(0.0f); + graph.AddInitializedTensor(state_initializer); + + graph.SetInputs({&iter_num_in, &cond_in, &state_in}); + graph.SetOutputs({&cond_out, &state_out}); + + auto status = graph.Resolve(); + EXPECT_TRUE(status.IsOK()) << status.ErrorMessage(); + + return graph.ToGraphProto(); + }; + + // Build the main graph: M, cond, state -> Loop -> final_state + onnxruntime::Model model("Loop with malformed body", false, ModelMetaData(), + PathString(), IOnnxRuntimeOpSchemaRegistryList(), + {{kOnnxDomain, 13}}, {}, DefaultLoggingManager().DefaultLogger()); + auto& graph = model.MainGraph(); + + TypeProto int64_tensor; + int64_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_INT64); + int64_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + TypeProto bool_tensor; + bool_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_BOOL); + bool_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + TypeProto float_tensor; + float_tensor.mutable_tensor_type()->set_elem_type(TensorProto_DataType_FLOAT); + float_tensor.mutable_tensor_type()->mutable_shape()->add_dim()->set_dim_value(1); + + auto& M = graph.GetOrCreateNodeArg("M", &int64_tensor); + auto& cond = graph.GetOrCreateNodeArg("cond", &bool_tensor); + auto& state = graph.GetOrCreateNodeArg("state", &float_tensor); + auto& final_state = graph.GetOrCreateNodeArg("final_state", &float_tensor); + + auto& loop_node = graph.AddNode("loop", "Loop", "Loop with an initializer-backed body input", + {&M, &cond, &state}, {&final_state}); + loop_node.AddAttribute("body", create_body()); + + graph.SetInputs({&M, &cond, &state}); + graph.SetOutputs({&final_state}); + + Status st = graph.Resolve(); + ASSERT_TRUE(st.IsOK()) << st.ErrorMessage(); + + SessionOptions so; + so.session_logid = "Loop.BodyInputAlsoInitializerIsRejected"; + InferenceSession session_object{so, GetEnvironment()}; + std::string serialized_model; + ASSERT_TRUE(model.ToProto().SerializeToString(&serialized_model)); + std::stringstream model_istream(serialized_model); + ASSERT_STATUS_OK(session_object.Load(model_istream)); + + // Must be rejected cleanly during initialization (no crash / no out-of-bounds read). + st = session_object.Initialize(); + ASSERT_FALSE(st.IsOK()); + EXPECT_EQ(st.Code(), common::StatusCode::INVALID_GRAPH) << st.ErrorMessage(); + EXPECT_THAT(st.ErrorMessage(), + ::testing::HasSubstr("Subgraph input count does not match the number of inputs provided by the " + "parent node 'loop'")); + EXPECT_THAT(st.ErrorMessage(), + ::testing::HasSubstr("The subgraph declares a graph input that is also one of its initializers")); +} + // Test a combination of things: // Subgraph input for loop state var has no type and is not used in the Loop subgraph (used in nested If subgraph) // Loop subgraph calls an If where the loop state var is an implicit input so it has no shape due to a loop state From 8b64945eb93f7a6166a82a79ffd77bce2e93143f Mon Sep 17 00:00:00 2001 From: shiyi Date: Mon, 21 Sep 2026 10:29:29 +0800 Subject: [PATCH 06/11] Harden shape inference for contrib ops (#32607) ### Summary Strengthens attribute/input validation in three contrib-op shape-inference functions in bert_defs.cc so malformed or malicious models are rejected at `Graph::Resolve` instead of triggering out-of-bounds reads or signed-integer overflow during shape inference. ### Changes `CausalConvWithState`: validate the ndim attribute is in [1, 3] and cross-check tensor ranks against it (weight == ndim+2, channels-first input == ndim+2, channels-last input >= 3) before the spatial-dim loop indexes input.dim(2+i). Previously only a rank >= 2 guard existed, so ndim=2/3 with a low-rank input read past the shape's dimensions. `GatedDeltaNet`: require the head counts/sizes to be positive and add step-by-step overflow guards before computing the state_update capsule width, preventing signed int64 overflow (UB) in state_update_capacity * (num_heads_v + num_heads_k*head_size_qk + num_heads_v*head_size_v). `GroupQueryAttention` / `SparseAttention`: check the parsed `total_sequence_length` initializer is non-empty before indexing data[0]. Backport notes for rel-1.28.3: - GatedDeltaNet does not exist on this branch, so its overflow guards and RejectsStateUpdateWidthOverflow test are dropped. - CausalConvWithState on this branch predates channels_last/state_window/ dilation. Only the ndim range check and the weight/input rank == ndim + 2 checks are backported; ChannelsLastInputRankBelowThreeIsRejected is dropped. - GroupQueryAttention on this branch has no sliding_window_cache, so the call site only gains the total_sequence_length_index argument. (cherry picked from commit bca9d15b639cce1818ba5641c9bd6d7897af85c9) --- .../core/graph/contrib_ops/bert_defs.cc | 35 ++++++++++---- .../causal_conv_with_state_op_test.cc | 46 +++++++++++++++++++ .../group_query_attention_op_test.cc | 30 +++++++++++- .../contrib_ops/sparse_attention_op_test.cc | 36 +++++++++++++++ 4 files changed, 136 insertions(+), 11 deletions(-) diff --git a/onnxruntime/core/graph/contrib_ops/bert_defs.cc b/onnxruntime/core/graph/contrib_ops/bert_defs.cc index 9fdcebf57081d..598daaa58ba00 100644 --- a/onnxruntime/core/graph/contrib_ops/bert_defs.cc +++ b/onnxruntime/core/graph/contrib_ops/bert_defs.cc @@ -234,7 +234,8 @@ void MultiHeadAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& c void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx, int past_key_index = -1, int use_max_past_present_buffer = -1, - int output_qk_index = -1) { + int output_qk_index = -1, + int total_sequence_length_index = -1) { // Type inference for outputs ONNX_NAMESPACE::propagateElemTypeFromInputToOutput(ctx, 0, 0); // output @@ -296,9 +297,13 @@ void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceConte if (ctx.getNumOutputs() >= 3) { // has present output int64_t total_sequence_length_value = 0; - const auto* total_sequence_length_data = ctx.getInputData(6); + const auto* total_sequence_length_data = + total_sequence_length_index >= 0 ? ctx.getInputData(total_sequence_length_index) : nullptr; if (total_sequence_length_data != nullptr) { const auto& data = ParseData(total_sequence_length_data); + if (data.size() != 1) { + fail_shape_inference("total_sequence_length input must contain a single element"); + } total_sequence_length_value = static_cast(data[0]); } @@ -447,13 +452,17 @@ void BaseGroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceConte void GroupQueryAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx, int past_key_index, int qk_output_index) { // TODO(aciddelgado): propagate output shapes depending if kv-share buffer is on or not constexpr int use_max_past_present_buffer = -1; - BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index); + constexpr int total_sequence_length_index = 6; + BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index, + total_sequence_length_index); } void SparseAttentionTypeAndShapeInference(ONNX_NAMESPACE::InferenceContext& ctx, int past_key_index) { constexpr int use_max_past_present_buffer = 1; constexpr int qk_output_index = -1; - BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index); + constexpr int total_sequence_length_index = 7; + BaseGroupQueryAttentionTypeAndShapeInference(ctx, past_key_index, use_max_past_present_buffer, qk_output_index, + total_sequence_length_index); } constexpr const char* Attention_ver1_doc = R"DOC( @@ -2316,6 +2325,11 @@ ONNX_MS_OPERATOR_SET_SCHEMA( propagateElemTypeFromInputToOutput(ctx, 0, 0); propagateElemTypeFromInputToOutput(ctx, 0, 1); + const int64_t ndim = getAttribute(ctx, "ndim", 1); + if (ndim < 1 || ndim > 3) { + fail_shape_inference("CausalConvWithState: ndim must be 1, 2, or 3, got ", ndim); + } + // Output 0: same shape as input (batch_size, channels, ...) propagateShapeFromInputToOutput(ctx, 0, 0); @@ -2326,13 +2340,16 @@ ONNX_MS_OPERATOR_SET_SCHEMA( if (hasInputShape(ctx, 0) && hasInputShape(ctx, 1)) { auto& input_shape = getInputShape(ctx, 0); auto& weight_shape = getInputShape(ctx, 1); - if (input_shape.dim_size() < 2) { - fail_shape_inference("CausalConvWithState: input must have rank >= 2"); + // Both are channels-first with rank ndim + 2: weight is (channels, 1, k_1, ..., k_ndim) and + // input is (batch_size, channels, d_1, ..., d_ndim). + if (weight_shape.dim_size() != ndim + 2) { + fail_shape_inference("CausalConvWithState: weight must have rank ndim + 2 (", + ndim + 2, "), got rank ", weight_shape.dim_size()); } - if (weight_shape.dim_size() < 2) { - fail_shape_inference("CausalConvWithState: weight must have rank >= 2"); + if (input_shape.dim_size() != ndim + 2) { + fail_shape_inference("CausalConvWithState: input must have rank ndim + 2 (", + ndim + 2, "), got rank ", input_shape.dim_size()); } - int64_t ndim = getAttribute(ctx, "ndim", 1); TensorShapeProto state_shape; *state_shape.add_dim() = input_shape.dim(0); // batch_size *state_shape.add_dim() = input_shape.dim(1); // channels diff --git a/onnxruntime/test/contrib_ops/causal_conv_with_state_op_test.cc b/onnxruntime/test/contrib_ops/causal_conv_with_state_op_test.cc index 2a7837dd1ce73..d15c10acc1ed7 100644 --- a/onnxruntime/test/contrib_ops/causal_conv_with_state_op_test.cc +++ b/onnxruntime/test/contrib_ops/causal_conv_with_state_op_test.cc @@ -654,5 +654,51 @@ TEST(CausalConvWithStateTest, LargerDimensions) { batch_size, channels, input_length, kernel_size, "silu"); } +// These tests exercise shape inference which uses fail_shape_inference (throws InferenceError). +// In no-exception builds, fail_shape_inference calls abort(), so these tests must be skipped. +#ifndef ORT_NO_EXCEPTIONS +// ndim outside [1, 3] range is rejected. +TEST(CausalConvWithStateTest, NdimOutOfRangeIsRejected) { + OpTester test("CausalConvWithState", 1, onnxruntime::kMSDomain); + test.AddAttribute("activation", "none"); + test.AddAttribute("ndim", 4); + test.AddInput("input", {1, 1, 2}, {1.0f, 2.0f}); + test.AddInput("weight", {1, 1, 2}, {0.5f, 0.25f}); + test.AddOptionalInputEdge(); // bias + test.AddOptionalInputEdge(); // past_state + test.AddOutput("output", {1, 1, 2}, {0.0f, 0.0f}); + test.AddOutput("present_state", {1, 1, 1}, {0.0f}); + test.Run(OpTester::ExpectResult::kExpectFailure, "ndim must be 1, 2, or 3"); +} + +// With ndim=2 the channels-first input must have rank ndim + 2 == 4 +TEST(CausalConvWithStateTest, InputRankMismatchIsRejected) { + OpTester test("CausalConvWithState", 1, onnxruntime::kMSDomain); + test.AddAttribute("activation", "none"); + test.AddAttribute("ndim", 2); + test.AddInput("input", {1, 1, 2}, {1.0f, 2.0f}); + test.AddInput("weight", {1, 1, 2, 2}, {0.5f, 0.25f, 0.5f, 0.25f}); + test.AddOptionalInputEdge(); // bias + test.AddOptionalInputEdge(); // past_state + test.AddOutput("output", {1, 1, 2}, {0.0f, 0.0f}); + test.AddOutput("present_state", {1, 1, 1}, {0.0f}); + test.Run(OpTester::ExpectResult::kExpectFailure, "input must have rank ndim + 2"); +} + +// The weight is always channels-first with rank ndim + 2. +TEST(CausalConvWithStateTest, WeightRankMismatchIsRejected) { + OpTester test("CausalConvWithState", 1, onnxruntime::kMSDomain); + test.AddAttribute("activation", "none"); + test.AddAttribute("ndim", 1); + test.AddInput("input", {1, 1, 2}, {1.0f, 2.0f}); + test.AddInput("weight", {1, 1}, {0.5f}); + test.AddOptionalInputEdge(); // bias + test.AddOptionalInputEdge(); // past_state + test.AddOutput("output", {1, 1, 2}, {0.0f, 0.0f}); + test.AddOutput("present_state", {1, 1, 1}, {0.0f}); + test.Run(OpTester::ExpectResult::kExpectFailure, "weight must have rank ndim + 2"); +} +#endif // !ORT_NO_EXCEPTIONS + } // namespace test } // namespace onnxruntime diff --git a/onnxruntime/test/contrib_ops/group_query_attention_op_test.cc b/onnxruntime/test/contrib_ops/group_query_attention_op_test.cc index 64c3f206b8a93..5aa1d069f79cc 100644 --- a/onnxruntime/test/contrib_ops/group_query_attention_op_test.cc +++ b/onnxruntime/test/contrib_ops/group_query_attention_op_test.cc @@ -60,7 +60,10 @@ static void RunGQASeqlensKTest( const std::string& expected_message, bool provide_past = false, int past_seq_len = 0, - const std::optional>& seqlens_k_shape = std::nullopt) { + const std::optional>& seqlens_k_shape = std::nullopt, + const std::optional>& total_seq_len_shape = std::nullopt, + const std::optional>& total_seq_len_data = std::nullopt, + bool total_seq_len_is_initializer = false) { constexpr int num_heads = 1; constexpr int kv_num_heads = 1; constexpr int head_size = 8; @@ -94,7 +97,9 @@ static void RunGQASeqlensKTest( ? *seqlens_k_shape : std::vector{batch_size}; tester.AddInput("seqlens_k", shape, seqlens_k_data); - tester.AddInput("total_sequence_length", {1}, {total_seq_len}); + const std::vector ts_shape = total_seq_len_shape.value_or(std::vector{1}); + const std::vector ts_data = total_seq_len_data.value_or(std::vector{total_seq_len}); + tester.AddInput("total_sequence_length", ts_shape, ts_data, total_seq_len_is_initializer); tester.AddOptionalInputEdge(); // cos_cache tester.AddOptionalInputEdge(); // sin_cache @@ -564,6 +569,27 @@ TEST(GroupQueryAttentionTest, SeqlensKScalarRejected) { /*seqlens_k_shape=*/std::vector{}); } +// This test exercises shape inference which uses fail_shape_inference (throws InferenceError). +// In no-exception builds, fail_shape_inference calls abort(), so this test must be skipped. +#ifndef ORT_NO_EXCEPTIONS +// total_sequence_length constant must have a single element. +TEST(GroupQueryAttentionTest, EmptyTotalSequenceLengthInitializerRejected) { + RunGQASeqlensKTest( + /*seqlens_k_data=*/{0}, + /*total_seq_len=*/1, + /*batch_size=*/1, + /*sequence_length=*/1, + OpTester::ExpectResult::kExpectFailure, + "total_sequence_length input must contain a single element", + /*provide_past=*/false, + /*past_seq_len=*/0, + /*seqlens_k_shape=*/std::nullopt, + /*total_seq_len_shape=*/std::vector{0}, + /*total_seq_len_data=*/std::vector{}, + /*total_seq_len_is_initializer=*/true); +} +#endif // !ORT_NO_EXCEPTIONS + // Helper to compare two output vectors (non-zero check + element-wise tolerance). static void ExpectOutputsMatch(const std::vector& a, const std::vector& b, float tolerance, const char* label) { diff --git a/onnxruntime/test/contrib_ops/sparse_attention_op_test.cc b/onnxruntime/test/contrib_ops/sparse_attention_op_test.cc index d7953442d738e..ca27c86d0d027 100644 --- a/onnxruntime/test/contrib_ops/sparse_attention_op_test.cc +++ b/onnxruntime/test/contrib_ops/sparse_attention_op_test.cc @@ -258,6 +258,42 @@ TEST(SparseAttentionTest, RejectsZeroDimBlockRowIndices) { {}, nullptr, &execution_providers); } +// This test exercises shape inference which uses fail_shape_inference (throws InferenceError). +// In no-exception builds, fail_shape_inference calls abort(), so this test must be skipped. +#ifndef ORT_NO_EXCEPTIONS +TEST(SparseAttentionTest, RejectsEmptyTotalSequenceLengthInitializer) { + OpTester test("SparseAttention", 1, onnxruntime::kMSDomain); + test.AddAttribute("num_heads", 2); + test.AddAttribute("kv_num_heads", 2); + test.AddAttribute("sparse_block_size", 1); + test.AddAttribute("scale", 1.0f); + test.AddAttribute("do_rotary", 0); + test.AddAttribute("rotary_interleaved", 0); + + test.AddInput("query", {1, 1, 16}, std::vector(16, 0.0f)); + test.AddInput("key", {1, 1, 16}, std::vector(16, 0.0f)); + test.AddInput("value", {1, 1, 16}, std::vector(16, 0.0f)); + test.AddInput("past_key", {1, 2, 4, 8}, std::vector(64, 0.0f)); + test.AddInput("past_value", {1, 2, 4, 8}, std::vector(64, 0.0f)); + test.AddInput("block_row_indices", {1, 5}, {0, 1, 2, 3, 4}); + test.AddInput("block_col_indices", {1, 1}, std::vector{0}, /*is_initializer=*/true); + // Empty total_sequence_length initializer at input 7 must be rejected by shape inference. + test.AddInput("total_sequence_length", {0}, std::vector{}, /*is_initializer=*/true); + test.AddInput("key_total_sequence_lengths", {1}, {4}); + test.AddOptionalInputEdge(); + test.AddOptionalInputEdge(); + + test.AddOutput("output", {1, 1, 16}, std::vector(16, 0.0f)); + test.AddOutput("present_key", {1, 2, 4, 8}, std::vector(64, 0.0f)); + test.AddOutput("present_value", {1, 2, 4, 8}, std::vector(64, 0.0f)); + + std::vector> execution_providers; + execution_providers.push_back(DefaultCpuExecutionProvider()); + test.Run(OpTester::ExpectResult::kExpectFailure, + "total_sequence_length input must contain a single element", {}, nullptr, &execution_providers); +} +#endif // !ORT_NO_EXCEPTIONS + // Helper for CSR value-validation tests. // Uses: num_heads=2, kv_num_heads=2, sparse_block_size=16, head_size=8. // block_row_indices shape: (1, max_blocks+1), block_col_indices shape: (1, col_count). From b98a7dcaf72cd7c60d52fea9167ff589aff4c66b Mon Sep 17 00:00:00 2001 From: Bryan B Date: Tue, 22 Sep 2026 15:53:47 -0700 Subject: [PATCH 07/11] Fix out-of-bounds read in ConvTransposeWithDynamicPads (#32679) ### Description Add an explicit rank-consistency check in `convTransposeWithDynamicPadsShapeInference()` so that a derived `kernel_shape` (built from the weight tensor's rank when the `kernel_shape` attribute is absent) must have the same length as `n_input_dims` (derived from the input tensor's rank) before it is used, exactly like the existing check on the explicit-attribute path. ### Motivation and Context `n_input_dims` is computed from input `X`'s rank. When `kernel_shape` is not given as an attribute, it is instead derived from weight `W`'s rank. If `W`'s rank disagrees with `X`'s rank, the resulting `kernel_shape` (and `effective_kernel_shape`, sized identically) is shorter or longer than `n_input_dims`. Both directions are unsafe: - **Shorter** (e.g. X rank 5, W rank 3): the output-shape loop iterates `n_input_dims` times over `effective_kernel_shape`/`pads`, reading past the end of the shorter vector. - **Longer** (e.g. X rank 3, W rank 5): the dilation loop iterates `kernel_shape.size()` times over `dilations`, which is always sized to `n_input_dims`, reading past the end of `dilations`. This runs during `Graph::Resolve()` (model load), before any kernel `Compute()` validation is ever reached. ### Testing Added `ConvTransposeWithDynamicPads_MismatchedInputWeightRank` (X rank 5, W rank 3, no `kernel_shape` attribute), asserting the model now fails cleanly at the existing kernel-level rank check (`"X num_dims does not match W num_dims."`) instead of hitting the shape-inference OOB read during load. (cherry picked from commit 2ecddca37183e39694d02f2dae222bed9931c9af) --- .../core/graph/contrib_ops/contrib_defs.cc | 6 +++++ .../conv_transpose_with_dynamic_pads_test.cc | 25 +++++++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/onnxruntime/core/graph/contrib_ops/contrib_defs.cc b/onnxruntime/core/graph/contrib_ops/contrib_defs.cc index 716b3d3f9a061..e26c97bb5b399 100644 --- a/onnxruntime/core/graph/contrib_ops/contrib_defs.cc +++ b/onnxruntime/core/graph/contrib_ops/contrib_defs.cc @@ -108,6 +108,12 @@ void convTransposeWithDynamicPadsShapeInference(InferenceContext& ctx) { } kernel_shape.push_back(second_input_shape.dim(i).dim_value()); } + // A longer kernel_shape (W rank > X rank) overruns `dilations` in the loop right below; + // a shorter one (W rank < X rank) leaves `effective_kernel_shape` too short for the + // output-shape loop further below. + if (kernel_shape.size() != n_input_dims) { + return; + } } std::vector effective_kernel_shape = kernel_shape; diff --git a/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc b/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc index 345a143d5b87c..9845809643886 100644 --- a/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc +++ b/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc @@ -70,6 +70,31 @@ TEST(ContribOpTest, ConvTransposeWithDynamicPads_InvalidInputRank2) { } #endif // !ORT_NO_EXCEPTIONS +// Test that a weight rank shorter than the input rank is rejected. 'Pads' must be an +// initializer so shape inference reaches the output-shape loop that reads past the end of +// the too-short kernel_shape/effective_kernel_shape before the fix. +TEST(ContribOpTest, ConvTransposeWithDynamicPads_ShorterWeightRank) { + OpTester test("ConvTransposeWithDynamicPads", 1, onnxruntime::kMSDomain); + test.AddInput("X", {1, 1, 2, 2, 2}, std::vector(8, 0.0f)); + test.AddInput("W", {1, 1, 3}, std::vector(3, 0.0f)); + test.AddInput("Pads", {6}, std::vector(6, 0), /*is_initializer=*/true); + test.AddOutput("Y", {}, std::vector{0.0f}); + test.Run(OpTester::ExpectResult::kExpectFailure, "num_dims does not match", + {kTensorrtExecutionProvider, kQnnExecutionProvider, kDmlExecutionProvider}); +} + +// Test that a weight rank longer than the input rank is rejected. This exercises the +// dilation-scaling loop, which reads past the end of `dilations` before the fix. +TEST(ContribOpTest, ConvTransposeWithDynamicPads_LongerWeightRank) { + OpTester test("ConvTransposeWithDynamicPads", 1, onnxruntime::kMSDomain); + test.AddInput("X", {1, 1, 2}, std::vector(2, 0.0f)); + test.AddInput("W", {1, 1, 3, 3, 3}, std::vector(27, 0.0f)); + test.AddInput("Pads", {2}, std::vector(2, 0)); + test.AddOutput("Y", {}, std::vector{0.0f}); + test.Run(OpTester::ExpectResult::kExpectFailure, "num_dims does not match", + {kTensorrtExecutionProvider, kQnnExecutionProvider, kDmlExecutionProvider}); +} + // Test that incorrectly sized dynamic pads are rejected. // This runs through kernel validation (not shape inference) so it works in no-exception builds. TEST(ContribOpTest, ConvTransposeWithDynamicPads_InvalidPadsSize) { From cea689103a5f2ac08e87111b5bc0d6681588d46e Mon Sep 17 00:00:00 2001 From: Bryan B Date: Tue, 22 Sep 2026 21:33:10 -0700 Subject: [PATCH 08/11] Fix OOB read in InferenceContextImpl::getInputData (#32681) ### Description `InferenceContextImpl::getInputData` (graph.cc) indexed `node_.InputDefs()` directly by the schema input index during shape inference, with no bounds check. Add the same bounds check already used by `DataPropagationContextImpl::getInputData` in the same file, so `getInputData` returns `nullptr` (treated as "input not available") instead of indexing out of bounds. ### Motivation and Context ONNX schemas can declare trailing inputs as optional (e.g. `ConvTransposeWithDynamicPads`'s `Pads`), so a node can validly omit them entirely, shrinking `Node::InputDefs()` below the schema's full input count. A `TypeAndShapeInferenceFunction` that queries such an omitted optional input's data (e.g. via `ctx.getInputData(index)`) on such a node reads past the end of the vector, causing an out-of-bounds read. (cherry picked from commit e5f272c92661855542cbdeb7593d61087fc8b04a) --- onnxruntime/core/graph/graph.cc | 6 +++++ .../conv_transpose_with_dynamic_pads_test.cc | 22 +++++++++++++++++++ 2 files changed, 28 insertions(+) diff --git a/onnxruntime/core/graph/graph.cc b/onnxruntime/core/graph/graph.cc index f936ee6a681a2..1f1bc3633670e 100644 --- a/onnxruntime/core/graph/graph.cc +++ b/onnxruntime/core/graph/graph.cc @@ -2752,6 +2752,12 @@ class InferenceContextImpl : public ONNX_NAMESPACE::InferenceContext { } const TensorProto* getInputData(size_t index) const override { + // A schema-optional input that's omitted (not even an empty placeholder) shrinks InputDefs(), + // so callers can pass an index the node doesn't actually have. + if (index >= getNumInputs()) { + return nullptr; + } + auto def = node_.InputDefs()[index]; if (!def) return nullptr; diff --git a/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc b/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc index 9845809643886..77474927f7870 100644 --- a/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc +++ b/onnxruntime/test/contrib_ops/conv_transpose_with_dynamic_pads_test.cc @@ -2,7 +2,11 @@ // Licensed under the MIT License. #include "gtest/gtest.h" +#include "core/graph/model.h" #include "test/providers/provider_test_utils.h" +#include "test/test_environment.h" +#include "test/unittest_util/graph_transform_test_builder.h" +#include "test/util/include/asserts.h" #include "default_providers.h" namespace onnxruntime { @@ -172,5 +176,23 @@ TEST(ContribOpTest, ConvTransposeWithDynamicPads_NegativePads_Dml) { } #endif // USE_DML +// 'Pads' is schema-optional; a node may validly supply only X and W. Shape inference must not +// index past the end of InputDefs() when reading the omitted Pads input. +TEST(ContribOpTest, ConvTransposeWithDynamicPads_PadsOmitted_ShapeInferenceOnly) { + std::unordered_map domain_to_version{{kOnnxDomain, 17}, {kMSDomain, 1}}; + Model model("conv_transpose_with_dynamic_pads_pads_omitted", /*is_onnx_domain_only=*/false, ModelMetaData(), + PathString(), IOnnxRuntimeOpSchemaRegistryList(), domain_to_version, {}, + DefaultLoggingManager().DefaultLogger()); + + ModelTestBuilder builder(model.MainGraph()); + NodeArg* x = builder.MakeInput(std::vector{1, 1, 3, 3}); + NodeArg* w = builder.MakeInput(std::vector{1, 1, 3, 3}); + NodeArg* y = builder.MakeOutput(); + builder.AddNode("ConvTransposeWithDynamicPads", {x, w}, {y}, kMSDomain); + builder.SetGraphOutputs(); + + ASSERT_STATUS_OK(model.MainGraph().Resolve()); +} + } // namespace test } // namespace onnxruntime From 63d8bd3e5654dbea5a25b01d366bd6a8a35b356d Mon Sep 17 00:00:00 2001 From: Copilot <198982749+Copilot@users.noreply.github.com> Date: Wed, 9 Sep 2026 06:19:58 +0000 Subject: [PATCH 09/11] Install uuid-dev in the Linux minimal build_full_ort job (#32491) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Description `build_full_ort` in `.github/workflows/linux_minimal_build.yml` now installs `uuid-dev` before building: ```yaml # This job builds with --use_coreml. The coremltools modelpackage sources include , # which is provided by uuid-dev on Ubuntu. - name: Install libuuid development files run: sudo apt-get update -y && sudo apt-get install -y uuid-dev ``` - Scoped to `build_full_ort` only — it is the sole job in this workflow that builds with `--use_coreml` directly on the runner; the rest build inside the CI docker image or don't enable CoreML. - Placed before `Build Full ORT and Prepare Test Files`, so the header/library are present both when CMake configures and when `ModelPackage.cpp` compiles. - No CMake change: `cmake/onnxruntime_providers_coreml.cmake` already does `find_path`/`find_library` for libuuid on Linux (with an actionable `FATAL_ERROR`) and links it into `onnxruntime_providers_coreml`. ### Motivation and Context `Linux CPU Minimal Build E2E` / `build_full_ort` fails in [run 34290674202, job 102276280773](https://github.com/microsoft/onnxruntime/actions/runs/34290674202): ``` _deps/coremltools-src/modelpackage/src/ModelPackage.cpp:35:10: fatal error: uuid/uuid.h: No such file or directory ``` The job enables the CoreML EP, and the fetched coremltools modelpackage sources include ``, supplied on Ubuntu by `uuid-dev`. Neither the workflow nor the reusable `setup-build-tools` action (which only provisions cmake/ccache/vcpkg) installs apt packages, so the header is only available when vcpkg's `coreml-ep` feature happens to provide it; without it Ninja stops while compiling `onnxruntime_providers_coreml`. --------- Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com> Co-authored-by: tianleiwu <30328909+tianleiwu@users.noreply.github.com> (cherry picked from commit f26e546fe25b04e737ec460f05a7d57f4938f23c) --- .github/workflows/linux_minimal_build.yml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/workflows/linux_minimal_build.yml b/.github/workflows/linux_minimal_build.yml index d2d31cb5d77e1..c5350110f9d99 100644 --- a/.github/workflows/linux_minimal_build.yml +++ b/.github/workflows/linux_minimal_build.yml @@ -40,6 +40,11 @@ jobs: with: node-version: 20 + # This job builds with --use_coreml. The coremltools modelpackage sources include , + # which is provided by uuid-dev on Ubuntu. + - name: Install libuuid development files + run: sudo apt-get update -y && sudo apt-get install -y uuid-dev + - name: Setup CCache uses: actions/cache@v5 with: From a0dea2f7403152540c0a9b58ddd43e4e10f2de51 Mon Sep 17 00:00:00 2001 From: adrastogi Date: Fri, 25 Sep 2026 12:23:39 -0700 Subject: [PATCH 10/11] Update Android minimal size baseline for rel-1.28.3 ### Description Raise the Android minimal baseline threshold by 3 KiB, from 1,440,768 to 1,443,840 bytes. The security fixes cherry-picked for 1.28.3 add validation to code that is compiled into the minimal build (session state finalization, execution frame, Loop/Scan subgraph handling). The deterministic section total grows from 1,440,538 bytes (rel-1.28.3 base, run 36072293072) to 1,443,098 bytes (+2,560 bytes), exceeding the old threshold by 2,330 bytes. The new threshold leaves 742 bytes of headroom. This is a release-branch-only change; main tracks its own baseline (see #32772). Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .github/workflows/android.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/android.yml b/.github/workflows/android.yml index 66141c58e7a01..3f0383ae69efa 100644 --- a/.github/workflows/android.yml +++ b/.github/workflows/android.yml @@ -78,8 +78,8 @@ jobs: run: | set -e -x BINARY_SIZE_THRESHOLD_ARGS="" - echo "Binary size threshold in bytes: 1440768" - BINARY_SIZE_THRESHOLD_ARGS="--threshold_size_in_bytes 1440768" + echo "Binary size threshold in bytes: 1443840" + BINARY_SIZE_THRESHOLD_ARGS="--threshold_size_in_bytes 1443840" # Ensure ANDROID_NDK_HOME is available and get its real path if [ -z "$ANDROID_NDK_HOME" ]; then From a5e59c532b67ded2cc637c414d660acfc745f5a8 Mon Sep 17 00:00:00 2001 From: Xiaofei Han Date: Thu, 17 Sep 2026 15:40:27 +0800 Subject: [PATCH 11/11] [Plugin EP] Preserve streams in single-tensor data transfers (#32666) ### Description Forward the caller's stream through the Plugin EP single-tensor `CopyTensorImpl` path. This matches the existing batched-copy behavior while preserving a null stream for synchronous copies. Adds focused callback regression coverage for asynchronous, synchronous, and batched copies. ### Motivation and Context The single-tensor Plugin EP data-transfer path discarded the supplied stream, preventing asynchronous copies from receiving the execution stream. Fixes #32643 Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> (cherry picked from commit 5899c7c954b419d64176f7461c484fd3530296c2) --- .../core/framework/plugin_data_transfer.cc | 4 +- .../framework/data_transfer_manager_test.cc | 42 +++++++++++++++++++ 2 files changed, 44 insertions(+), 2 deletions(-) diff --git a/onnxruntime/core/framework/plugin_data_transfer.cc b/onnxruntime/core/framework/plugin_data_transfer.cc index d6b1680176815..8a6d37a4c30ce 100644 --- a/onnxruntime/core/framework/plugin_data_transfer.cc +++ b/onnxruntime/core/framework/plugin_data_transfer.cc @@ -51,14 +51,14 @@ Status DataTransfer::CopyTensors(const std::vector& src_dst_pairs) c } // optimized version for a single copy. see comments above in CopyTensors regarding the OrtValue usage and const_cast -Status DataTransfer::CopyTensorImpl(const Tensor& src_tensor, Tensor& dst_tensor, onnxruntime::Stream* /*stream*/) const { +Status DataTransfer::CopyTensorImpl(const Tensor& src_tensor, Tensor& dst_tensor, onnxruntime::Stream* stream) const { OrtValue src, dst; Tensor* src_tensor_ptr = const_cast(&src_tensor); src.Init(static_cast(src_tensor_ptr), ml_tensor_type, no_op_deleter); dst.Init(static_cast(&dst_tensor), ml_tensor_type, no_op_deleter); const OrtValue* src_ptr = &src; OrtValue* dst_ptr = &dst; - OrtSyncStream* stream_ptr = nullptr; // static_cast(stream); + OrtSyncStream* stream_ptr = reinterpret_cast(stream); auto* status = impl_.CopyTensors(&impl_, &src_ptr, &dst_ptr, &stream_ptr, 1); return ToStatusAndRelease(status); diff --git a/onnxruntime/test/framework/data_transfer_manager_test.cc b/onnxruntime/test/framework/data_transfer_manager_test.cc index 9b7e705faf4b7..90da354cbc449 100644 --- a/onnxruntime/test/framework/data_transfer_manager_test.cc +++ b/onnxruntime/test/framework/data_transfer_manager_test.cc @@ -7,12 +7,54 @@ #include "core/common/inlined_containers.h" #include "core/framework/data_transfer_manager.h" #include "core/framework/ort_value.h" +#include "core/framework/plugin_data_transfer.h" +#include "core/framework/stream_handles.h" #include "test/unittest_util/framework_test_utils.h" #include "test/util/include/asserts.h" namespace onnxruntime { namespace test { +TEST(DataTransferManagerTest, PluginCopiesForwardStreams) { + struct TestDataTransfer final : OrtDataTransferImpl { + TestDataTransfer() : OrtDataTransferImpl{} { + ort_version_supported = ORT_API_VERSION; + Release = [](OrtDataTransferImpl*) noexcept {}; + CopyTensors = [](OrtDataTransferImpl* impl, const OrtValue**, OrtValue**, + OrtSyncStream** streams, size_t num_tensors) noexcept -> OrtStatus* { + auto& self = *static_cast(impl); + self.copied_tensors += num_tensors; + self.last_stream = streams != nullptr && num_tensors > 0 ? streams[0] : nullptr; + return nullptr; + }; + } + + ORT_DISALLOW_COPY_ASSIGNMENT_AND_MOVE(TestDataTransfer); + + size_t copied_tensors = 0; + OrtSyncStream* last_stream = nullptr; + } impl; + + plugin_ep::DataTransfer data_transfer{impl}; + auto allocator = TestCPUExecutionProvider()->CreatePreferredAllocators()[0]; + Tensor source{DataTypeImpl::GetType(), TensorShape{4}, allocator}; + Tensor destination{DataTypeImpl::GetType(), TensorShape{4}, allocator}; + OrtDevice device; + Stream stream{nullptr, device}; + + ASSERT_STATUS_OK(data_transfer.CopyTensorAsync(source, destination, stream)); + EXPECT_EQ(impl.copied_tensors, 1U); + EXPECT_EQ(impl.last_stream, reinterpret_cast(&stream)); + + ASSERT_STATUS_OK(data_transfer.CopyTensor(source, destination)); + EXPECT_EQ(impl.copied_tensors, 2U); + EXPECT_EQ(impl.last_stream, nullptr); + + ASSERT_STATUS_OK(data_transfer.CopyTensors({{source, destination, &stream}})); + EXPECT_EQ(impl.copied_tensors, 3U); + EXPECT_EQ(impl.last_stream, reinterpret_cast(&stream)); +} + // DataTransferManager::CopyTensors should validate sizes match before calling the IDataTransfer implementation TEST(DataTransferManagerTest, BatchedTensorCopyBadSize) { auto allocator = TestCPUExecutionProvider()->CreatePreferredAllocators()[0];