Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 13 additions & 1 deletion docs/troubleshooting/device-error-codes.md
Original file line number Diff line number Diff line change
Expand Up @@ -108,7 +108,19 @@ layer to go looking in, which is what these columns are for.
| 101 | ASYNC_COMPLETION_INVALID | kernel (async) |
| 102 | ASYNC_WAIT_OVERFLOW | kernel (async) |
| 103 | ASYNC_REGISTRATION_FAILED | runtime-internal |
| 104 | READY_QUEUE_OVERFLOW | runtime-internal / config |
| 104 | READY_QUEUE_OVERFLOW | host bind / runtime-internal / config |

### READY_QUEUE_OVERFLOW origins

Code 104 can be reported before or after device execution starts:

- **Host bind:** one ready queue has more than 32768 reachable tasks. This is
the supported single-queue population ceiling; larger graphs are rejected
during bind and never reach the device. Split the graph or reduce the number
of tasks routed to that queue.
- **Device scheduler:** a queue push found no free slot after bind accepted the
graph. Use the reported queue and occupancy details to investigate a runtime
accounting error or an incompatible configuration.

### SCHEDULER_TIMEOUT sub-classes

Expand Down
94 changes: 94 additions & 0 deletions src/a2a3/runtime/host_build_graph/host/ready_queue_sizing.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,94 @@
/*
* Copyright (c) PyPTO Contributors.
* This program is free software, you can redistribute it and/or modify it under the terms and conditions of
* CANN Open Software License Agreement Version 2.0 (the "License").
* Please refer to the License for details. You may not use this file except in compliance with the License.
* THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
* INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
* See LICENSE in the root of the software repository for the full text of the License.
* -----------------------------------------------------------------------------------------------------------
*/

#include "ready_queue_sizing.h"

#include "../common/runtime_status.h"

namespace {

constexpr uint64_t READY_QUEUE_POPULATION_OVER_LIMIT = READY_QUEUE_CAPACITY_LIMIT + 1;

void add_population(uint64_t *population, uint64_t count) {
if (*population >= READY_QUEUE_POPULATION_OVER_LIMIT || count >= READY_QUEUE_POPULATION_OVER_LIMIT ||
count > READY_QUEUE_POPULATION_OVER_LIMIT - *population) {
*population = READY_QUEUE_POPULATION_OVER_LIMIT;
return;
}
*population += count;
}

uint64_t capacity_for_population(uint64_t population) {
uint64_t capacity = 2;
while (capacity < population)
capacity <<= 1;
return capacity;
}

} // namespace

void ReadyQueuePopulations::add_task(ActiveMask active_mask, TaskAttrs task_attrs, TaskKind task_kind, uint64_t count) {
if (task_kind == TaskKind::GRAPH) {
add_population(&graph_ready, count);
add_population(&graph_prepare, count);
return;
}

const ResourceShape shape = active_mask.to_shape();
if (shape == ResourceShape::DUMMY) {
add_population(&dummy, count);
return;
}

if (task_attrs.has_predicate()) add_population(&dummy, count);
uint64_t *population = task_attrs.requires_sync_start() ? ready_sync : ready;
add_population(&population[static_cast<int32_t>(shape)], count);
}

void ReadyQueuePopulations::add(const ReadyQueuePopulations &other) {
for (int i = 0; i < NUM_RESOURCE_SHAPES; ++i) {
add_population(&ready[i], other.ready[i]);
add_population(&ready_sync[i], other.ready_sync[i]);
}
add_population(&dummy, other.dummy);
add_population(&graph_ready, other.graph_ready);
add_population(&graph_prepare, other.graph_prepare);
}

bool ReadyQueuePopulations::derive_capacities(ReadyQueueCapacities *capacities) const {
if (capacities == nullptr) return false;
for (int i = 0; i < NUM_RESOURCE_SHAPES; ++i) {
if (ready[i] > READY_QUEUE_CAPACITY_LIMIT || ready_sync[i] > READY_QUEUE_CAPACITY_LIMIT) return false;
}
if (dummy > READY_QUEUE_CAPACITY_LIMIT || graph_ready > READY_QUEUE_CAPACITY_LIMIT ||
graph_prepare > READY_QUEUE_CAPACITY_LIMIT) {
return false;
}

*capacities = ReadyQueueCapacities{};
for (int i = 0; i < NUM_RESOURCE_SHAPES; ++i) {
capacities->ready[i] = capacity_for_population(ready[i]);
capacities->ready_sync[i] = capacity_for_population(ready_sync[i]);
}
capacities->dummy = capacity_for_population(dummy);
capacities->graph_ready = capacity_for_population(graph_ready);
capacities->graph_prepare = capacity_for_population(graph_prepare);
return true;
}

int32_t derive_ready_queue_capacities(
const ReadyQueuePopulations &populations, SharedMemoryHeader &sm_header, ReadyQueueCapacities *capacities
) {
if (populations.derive_capacities(capacities)) return 0;

sm_header.sched_error_code.store(SIMPLER_ERROR_READY_QUEUE_OVERFLOW, std::memory_order_release);
return runtime_status_from_error_codes(SIMPLER_ERROR_NONE, SIMPLER_ERROR_READY_QUEUE_OVERFLOW);
}
64 changes: 55 additions & 9 deletions src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,7 @@
#include "../runtime/graph_host_state.h"
#include "../runtime/host_phase_trace.h"
#include "../runtime/orchestrator.h"
#include "../runtime/ready_queue_sizing.h"
#include "graph_recorder_pool.h"
#include "../runtime/runtime_core.h"
#include "../runtime/shared_memory.h"
Expand Down Expand Up @@ -373,7 +374,10 @@ struct DefinitionUploads {
// claimed, so this pass writes the headers, copies in whatever did not fit, and
// issues a single H2D of the used prefix. The device initial classify then replaces
// each task's graph_context with an execution constructed in its own heap.
bool bind_graph_definitions(const HostApi *api, GraphHostState &graph_state, DefinitionUploads *uploads) {
bool bind_graph_definitions(
const HostApi *api, GraphHostState &graph_state, DefinitionUploads *uploads,
ReadyQueuePopulations *ready_queue_populations
) {
*uploads = DefinitionUploads{};
const size_t count = graph_host_upload_count(graph_state);
GraphHostDefinitionList definitions = graph_host_definitions(graph_state);
Expand All @@ -384,6 +388,8 @@ bool bind_graph_definitions(const HostApi *api, GraphHostState &graph_state, Def
size_t object_offset; // of the object's header, from the block base
size_t image_bytes; // the Definition image alone
const std::byte *copy; // the image to copy in, or nullptr when built in place
ReadyQueuePopulations ready_queue_populations;
bool populations_ready{false};
};
std::unordered_map<uint64_t, PackedDefinition> packed;
// Objects the recorders built already occupy the arena's used prefix at the
Expand All @@ -393,12 +399,12 @@ bool bind_graph_definitions(const HostApi *api, GraphHostState &graph_state, Def
for (const GraphHostDefinition &entry : definitions.entries) {
if (entry.bytes < sizeof(GraphDefinition)) continue;
if (entry.spill == nullptr) {
packed.emplace(entry.full_key, PackedDefinition{entry.object_offset, entry.bytes, nullptr});
packed.emplace(entry.full_key, PackedDefinition{entry.object_offset, entry.bytes, nullptr, {}, false});
continue;
}
const size_t object_offset = block_bytes;
block_bytes += align_up(sizeof(GraphDefinitionHeader) + entry.bytes);
packed.emplace(entry.full_key, PackedDefinition{object_offset, entry.bytes, entry.spill});
packed.emplace(entry.full_key, PackedDefinition{object_offset, entry.bytes, entry.spill, {}, false});
uploads->spilled++;
}

Expand Down Expand Up @@ -490,6 +496,22 @@ bool bind_graph_definitions(const HostApi *api, GraphHostState &graph_state, Def
LOG_ERROR("host-orch: Graph runtime storage address is misaligned");
return false;
}
PackedDefinition &packed_definition = object_it->second;
if (!packed_definition.populations_ready) {
const GraphNodeDefinition *nodes =
graph_definition_array<GraphNodeDefinition>(*definition, definition->off_nodes, definition->task_count);
if (nodes == nullptr) {
LOG_ERROR("host-orch: invalid Graph Definition node array");
return false;
}
for (uint32_t i = 0; i < definition->task_count; ++i) {
packed_definition.ready_queue_populations.add_task(
ActiveMask(nodes[i].active_mask), TaskAttrs(nodes[i].task_attrs), TaskKind::GRAPH_NODE
);
}
packed_definition.populations_ready = true;
}
ready_queue_populations->add(packed_definition.ready_queue_populations);
upload->outer_slot->graph_context = reinterpret_cast<GraphDefinition *>(
reinterpret_cast<uintptr_t>(block) + object_it->second.object_offset + sizeof(GraphDefinitionHeader)
);
Expand Down Expand Up @@ -666,12 +688,26 @@ int32_t run_host_orchestration(
// After the span closes: the reduction walks a few hundred records and emits
// five markers, which must not be charged to the bind it measures.

// total_tasks sizes the bounded per-segment H2D copies below; a value outside
// [0, task_capacity] would make those copies read/write out of bounds.
if (total_tasks < 0 || static_cast<uint64_t>(total_tasks) > task_capacity) {
LOG_ERROR("host-orch: total_tasks %d out of range [0, %" PRIu64 "]", total_tasks, task_capacity);
return PTO_RUNTIME_ERR_INTERNAL;
}

ReadyQueuePopulations ready_queue_populations{};
SharedMemoryTaskHeader &tasks = host_sm_handle.header->tasks;
for (int32_t task_id = 0; task_id < total_tasks; ++task_id) {
const ChipTaskSlotState &slot = tasks.get_slot_state_by_task_id(task_id);
ready_queue_populations.add_task(slot.active_mask, slot.task_attrs, slot.task_kind);
}

// Upload each distinct Definition as its own retained device object and bind
// every outer Graph task to it. Per-invocation data already lives in that
// task's payload regions and is copied with the shared-memory image below.
const int64_t t_graph_ns = bind_phase_begin();
DefinitionUploads definition_uploads{};
if (!bind_graph_definitions(api, *graph_state, &definition_uploads)) {
if (!bind_graph_definitions(api, *graph_state, &definition_uploads, &ready_queue_populations)) {
return PTO_RUNTIME_ERR_INTERNAL;
}
{
Expand All @@ -689,12 +725,22 @@ int32_t run_host_orchestration(
record_bind_phase(HostPhaseKind::BindGraphUpload, t_graph_ns, attrs, definition_uploads.bytes);
}

// total_tasks sizes the bounded per-segment H2D copies below; a value outside
// [0, task_capacity] would make those copies read/write out of bounds.
if (total_tasks < 0 || static_cast<uint64_t>(total_tasks) > task_capacity) {
LOG_ERROR("host-orch: total_tasks %d out of range [0, %" PRIu64 "]", total_tasks, task_capacity);
return PTO_RUNTIME_ERR_INTERNAL;
ReadyQueueCapacities ready_queue_capacities{};
const int32_t ready_queue_status =
derive_ready_queue_capacities(ready_queue_populations, *host_sm_handle.header, &ready_queue_capacities);
if (ready_queue_status != 0) {
LOG_ERROR(
Comment thread
TaoZQY marked this conversation as resolved.
"host-orch: ready queue reachable population exceeds %" PRIu64 " (ready=%" PRIu64 "/%" PRIu64 "/%" PRIu64
", sync=%" PRIu64 "/%" PRIu64 "/%" PRIu64 ", dummy=%" PRIu64 ", graph=%" PRIu64 "/%" PRIu64 ")",
READY_QUEUE_CAPACITY_LIMIT, ready_queue_populations.ready[0], ready_queue_populations.ready[1],
ready_queue_populations.ready[2], ready_queue_populations.ready_sync[0],
ready_queue_populations.ready_sync[1], ready_queue_populations.ready_sync[2], ready_queue_populations.dummy,
ready_queue_populations.graph_ready, ready_queue_populations.graph_prepare
);
LOG_RUNTIME_FAILURE(SIMPLER_ERROR_NONE, SIMPLER_ERROR_READY_QUEUE_OVERFLOW, ready_queue_status);
return ready_queue_status;
}
rt->prebuilt_layout.sched.capacities = ready_queue_capacities;
host_phase_trace_note_submitted(static_cast<uint64_t>(total_tasks));

// The count travels inside the header the restack copies wholesale, which is
Expand Down
32 changes: 32 additions & 0 deletions src/a2a3/runtime/host_build_graph/runtime/ready_queue_sizing.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
/*
* Copyright (c) PyPTO Contributors.
* This program is free software, you can redistribute it and/or modify it under the terms and conditions of
* CANN Open Software License Agreement Version 2.0 (the "License").
* Please refer to the License for details. You may not use this file except in compliance with the License.
* THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
* INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
* See LICENSE in the root of the software repository for the full text of the License.
* -----------------------------------------------------------------------------------------------------------
*/

#pragma once

#include <cstdint>

#include "scheduler/scheduler.h"

struct ReadyQueuePopulations {
uint64_t ready[NUM_RESOURCE_SHAPES]{};
uint64_t ready_sync[NUM_RESOURCE_SHAPES]{};
uint64_t dummy{0};
uint64_t graph_ready{0};
uint64_t graph_prepare{0};

void add_task(ActiveMask active_mask, TaskAttrs task_attrs, TaskKind task_kind, uint64_t count = 1);
void add(const ReadyQueuePopulations &other);
bool derive_capacities(ReadyQueueCapacities *capacities) const;
};

int32_t derive_ready_queue_capacities(
const ReadyQueuePopulations &populations, SharedMemoryHeader &sm_header, ReadyQueueCapacities *capacities
);
10 changes: 3 additions & 7 deletions src/a2a3/runtime/host_build_graph/runtime/runtime_types.h
Original file line number Diff line number Diff line change
Expand Up @@ -128,13 +128,9 @@ inline constexpr uint64_t HEAP_VIRTUAL_CAPACITY = GRAPH_RECORD_VIRTUAL_BASE - HE
// Scope management
#define CHIP_MAX_SCOPE_DEPTH 64 // Maximum nesting depth

// Per-shape ready-queue capacity (power of two). This is a ring buffer that
// bounds peak CONCURRENT occupancy (enqueue_pos - dequeue_pos), not total task
// count: slots recycle, so capacity need only exceed the most tasks ever
// simultaneously ready in any one queue. Overflow on the ready/sync/dummy queues
// latches SIMPLER_ERROR_READY_QUEUE_OVERFLOW (safe-fail), so it must exceed the
// worst-case ready burst with margin.
#define CHIP_READY_QUEUE_SIZE 8192
// Per-queue arena reservation ceiling. Bind configures each queue to the next
// power of two covering the tasks that can reach it and rejects a larger graph.
inline constexpr uint64_t READY_QUEUE_CAPACITY_LIMIT = 32768;

// Cross-thread early-dispatch work queue (power of two)
#define CHIP_EARLY_DISPATCH_QUEUE_SIZE 64
Expand Down
14 changes: 11 additions & 3 deletions src/a2a3/runtime/host_build_graph/runtime/scheduler/scheduler.h
Original file line number Diff line number Diff line change
Expand Up @@ -438,6 +438,14 @@ void ready_queue_init_data_from_layout(ChipReadyQueue *queue, uint64_t capacity)
void ready_queue_wire_arena_pointers(ChipReadyQueue *queue, DeviceArena &arena, size_t slots_off);
void ready_queue_destroy(ChipReadyQueue *queue);

struct ReadyQueueCapacities {
uint64_t ready[NUM_RESOURCE_SHAPES]{};
uint64_t ready_sync[NUM_RESOURCE_SHAPES]{};
uint64_t dummy{0};
uint64_t graph_ready{0};
uint64_t graph_prepare{0};
};

/**
* Statistics returned by mixed-task completion processing
*/
Expand All @@ -461,7 +469,7 @@ struct SchedulerLayout {
size_t off_graph_prepare_queue_slots;
size_t off_early_dispatch_queue_slots[NUM_RESOURCE_SHAPES];
size_t off_early_sync_start_queue_slots;
uint64_t ready_queue_capacity;
ReadyQueueCapacities capacities;
Comment thread
TaoZQY marked this conversation as resolved.
};

/**
Expand Down Expand Up @@ -558,8 +566,8 @@ struct SchedulerState {
}
}
// Every ready / sync / dummy / graph task routes to exactly one queue. A
// false push means that queue's peak concurrent occupancy exceeded
// CHIP_READY_QUEUE_SIZE — a capacity mis-sizing, not a normal condition.
// false push means that queue's peak concurrent occupancy exceeded its
// bind-time capacity — a capacity mis-sizing, not a normal condition.
// Silently dropping the task would stall the run, so latch a named error
// (surfaces as READY_QUEUE_OVERFLOW rather than an anonymous
// forward-progress timeout). The graph_ready push is checked identically
Expand Down
32 changes: 19 additions & 13 deletions src/a2a3/runtime/host_build_graph/runtime/shared/runtime_init.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -79,24 +79,30 @@ void SchedulerState::TaskHeaderView::destroy() { tasks = nullptr; }

SchedulerLayout SchedulerState::reserve_layout(DeviceArena &arena) {
SchedulerLayout layout{};
layout.ready_queue_capacity = CHIP_READY_QUEUE_SIZE;
for (int i = 0; i < NUM_RESOURCE_SHAPES; ++i) {
layout.capacities.ready[i] = READY_QUEUE_CAPACITY_LIMIT;
layout.capacities.ready_sync[i] = READY_QUEUE_CAPACITY_LIMIT;
}
layout.capacities.dummy = READY_QUEUE_CAPACITY_LIMIT;
layout.capacities.graph_ready = READY_QUEUE_CAPACITY_LIMIT;
layout.capacities.graph_prepare = READY_QUEUE_CAPACITY_LIMIT;

// Fixed-capacity early-dispatch queues first, then the CHIP_READY_QUEUE_SIZE
// ones. The big nine are the arena's last reservations so that the bytes bind
// Fixed-capacity early-dispatch queues first, then the configurable queues.
// The big nine are the arena's last reservations so that the bytes bind
// uploads stay one contiguous range no matter how much of them is in use.
for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
layout.off_early_dispatch_queue_slots[i] = ready_queue_reserve_layout(arena, CHIP_EARLY_DISPATCH_QUEUE_SIZE);
}
layout.off_early_sync_start_queue_slots = ready_queue_reserve_layout(arena, CHIP_EARLY_DISPATCH_QUEUE_SIZE);
for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
layout.off_ready_queue_slots[i] = ready_queue_reserve_layout(arena, CHIP_READY_QUEUE_SIZE);
layout.off_ready_queue_slots[i] = ready_queue_reserve_layout(arena, READY_QUEUE_CAPACITY_LIMIT);
}
for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
layout.off_ready_sync_queue_slots[i] = ready_queue_reserve_layout(arena, CHIP_READY_QUEUE_SIZE);
layout.off_ready_sync_queue_slots[i] = ready_queue_reserve_layout(arena, READY_QUEUE_CAPACITY_LIMIT);
}
layout.off_dummy_ready_queue_slots = ready_queue_reserve_layout(arena, CHIP_READY_QUEUE_SIZE);
layout.off_graph_ready_queue_slots = ready_queue_reserve_layout(arena, CHIP_READY_QUEUE_SIZE);
layout.off_graph_prepare_queue_slots = ready_queue_reserve_layout(arena, CHIP_READY_QUEUE_SIZE);
layout.off_dummy_ready_queue_slots = ready_queue_reserve_layout(arena, READY_QUEUE_CAPACITY_LIMIT);
layout.off_graph_ready_queue_slots = ready_queue_reserve_layout(arena, READY_QUEUE_CAPACITY_LIMIT);
layout.off_graph_prepare_queue_slots = ready_queue_reserve_layout(arena, READY_QUEUE_CAPACITY_LIMIT);
// Polling: no dep_pool arena region — producer dependencies are inline ids on
// the payload and readiness is via completion_flags.
return layout;
Expand All @@ -115,14 +121,14 @@ bool SchedulerState::init_data_from_layout(const SchedulerLayout &layout, Device
}

for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
ready_queue_init_data_from_layout(&sched->ready_queues[i], layout.ready_queue_capacity);
ready_queue_init_data_from_layout(&sched->ready_queues[i], layout.capacities.ready[i]);
Comment thread
TaoZQY marked this conversation as resolved.
}
for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
ready_queue_init_data_from_layout(&sched->ready_sync_queues[i], layout.ready_queue_capacity);
ready_queue_init_data_from_layout(&sched->ready_sync_queues[i], layout.capacities.ready_sync[i]);
}
ready_queue_init_data_from_layout(&sched->dummy_ready_queue, layout.ready_queue_capacity);
ready_queue_init_data_from_layout(&sched->graph_ready_queue, layout.ready_queue_capacity);
ready_queue_init_data_from_layout(&sched->graph_prepare_queue, layout.ready_queue_capacity);
ready_queue_init_data_from_layout(&sched->dummy_ready_queue, layout.capacities.dummy);
ready_queue_init_data_from_layout(&sched->graph_ready_queue, layout.capacities.graph_ready);
ready_queue_init_data_from_layout(&sched->graph_prepare_queue, layout.capacities.graph_prepare);
for (int i = 0; i < NUM_RESOURCE_SHAPES; i++) {
ready_queue_init_data_from_layout(&sched->early_dispatch_queues[i], CHIP_EARLY_DISPATCH_QUEUE_SIZE);
}
Expand Down
Loading
Loading