Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
/*
* Copyright (c) PyPTO Contributors.
* This program is free software, you can redistribute it and/or modify it under the terms and conditions of
* CANN Open Software License Agreement Version 2.0 (the "License").
* Please refer to the License for details. You may not use this file except in compliance with the License.
* THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
* INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
* See LICENSE in the root of the software repository for the full text of the License.
* -----------------------------------------------------------------------------------------------------------
*/

/**
* available_aicore_counts: spend the counts rt_available_*_count() reports.
*
* The counts are an AICPU-side runtime query with no host-side equivalent, so
* the run is made self-describing rather than compared against a pinned number:
* `shape` carries the reported cluster / AIV counts, and a MIX cohort of
* exactly `cluster_count` blocks writes its block index into `blocks`. The host
* reads the width out of `shape` and checks that many block slots.
*
* require_sync_start on the cohort is what makes the count falsifiable: the
* cohort needs every block co-resident, so an over-reported cluster count trips
* the sync-start deadlock guard instead of silently passing.
*/

#include <stdint.h>

#include "pto_orchestration_api.h" // NOLINT(build/include_subdir)

#define FUNC_SPMD_MIX_AIC 0
#define FUNC_SPMD_MIX_AIV0 1
#define FUNC_SPMD_MIX_AIV1 2

// PTO2LaunchSpec spells the SPMD block-count setter differently per arch
// (a5: set_core_num, a2a3: set_block_num) for the same field. Bridge it so this
// one fixture compiles on both; keyed off the arch's pto_types.h include guard,
// which pto_orchestration_api.h pulls in transitively.
static inline void set_block_count(L0TaskArgs &args, int16_t n) {
#if defined(SRC_A5_RUNTIME_TENSORMAP_AND_RINGBUFFER_RUNTIME_PTO_TYPES_H_)
args.launch_spec.set_core_num(n);
#else
args.launch_spec.set_block_num(n);
#endif
}

extern "C" {

__attribute__((visibility("default"))) PTO2OrchestrationConfig aicpu_orchestration_config(const L2TaskArgs &orch_args) {
(void)orch_args;
return PTO2OrchestrationConfig{
.expected_arg_count = 2,
};
}

__attribute__((visibility("default"))) void aicpu_orchestration_entry(const L2TaskArgs &orch_args) {
const Tensor &blocks = orch_args.tensor(0).ref();
const Tensor &shape = orch_args.tensor(1).ref();

const int32_t cluster_count = rt_available_cluster_count();
const int32_t aiv_count = rt_available_aiv_count();
LOG_INFO_V0("[available_aicore_counts] clusters=%d aiv=%d", cluster_count, aiv_count);

MixedKernels mk;
mk.aic_kernel_id = FUNC_SPMD_MIX_AIC;
mk.aiv0_kernel_id = FUNC_SPMD_MIX_AIV0;
mk.aiv1_kernel_id = FUNC_SPMD_MIX_AIV1;

L0TaskArgs args;
args.add_inout(blocks);
args.add_scalar(static_cast<int64_t>(0)); // base cache line
set_block_count(args, static_cast<int16_t>(cluster_count));
args.launch_spec.set_require_sync_start(true);
rt_submit_task(mk, args);

// shape carries no producer or consumer task, so set_tensor_data writes it
// straight through. Giving it one would hang host_build_graph: its
// orchestrator runs to completion on the host before the device executes
// anything, so a producer's task_state can never reach COMPLETED and
// wait_for_tensor_ready would spin to PTO2_TENSOR_DATA_TIMEOUT_CYCLES.
uint32_t idx[1] = {0};
set_tensor_data<int32_t>(shape, 1, idx, cluster_count);
idx[0] = 1;
set_tensor_data<int32_t>(shape, 1, idx, aiv_count);
}

} // extern "C"
Original file line number Diff line number Diff line change
@@ -0,0 +1,141 @@
#!/usr/bin/env python3
# Copyright (c) PyPTO Contributors.
# This program is free software, you can redistribute it and/or modify it under the terms and conditions of
# CANN Open Software License Agreement Version 2.0 (the "License").
# Please refer to the License for details. You may not use this file except in compliance with the License.
# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED,
# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE.
# See LICENSE in the root of the software repository for the full text of the License.
# -----------------------------------------------------------------------------------------------------------
"""available_aicore_counts: the runtime's core counts are real and spendable.

The orchestration reports what `rt_available_*_count()` gave it in `shape` and
spends it: a MIX cohort of exactly `cluster_count` blocks, each writing its
block index into `blocks`.

Two cases, because neither alone is sufficient:

- **Pinned** fixes `block_dim`, so the host independently knows the answer and
asserts the exact value. This is the only case that can catch an
**under**-reported count — the Auto case derives its expectation from the
reported number, so a count that is too small is self-consistent there.
- **Auto** takes whatever the platform resolves (sim and onboard differ, and
onboard's comes from the driver) and checks the count is in range, that
`aiv == 2 * clusters`, and that every one of those blocks really ran.

An **over**-reported count fails in both: the cohort asks for
`require_sync_start`, so it needs every block co-resident and the deadlock
guard fires on device.

host_build_graph is where the counts are hardest to get right: its
orchestrator runs on the host, inside the bind, so it reads `worker_count`
off `Runtime` rather than the AICPU's handshake result. Publishing the core
geometry any later than that hands host orchestration a zero, which the Pinned
case catches (0 != 4) while leaving the `require_sync_start` guard it feeds
silently disabled.
"""

import torch
from simpler.task_interface import ArgDirection as D

from simpler_setup import SceneTestCase, TaskArgsBuilder, Tensor, scene_test

FLOATS_PER_CACHE_LINE = 16
SLOTS_PER_BLOCK = 3
# Cover the largest cohort the platform can launch (PLATFORM_MAX_BLOCKDIM).
MAX_CLUSTERS = 24
AIV_PER_CLUSTER = 2
TOTAL_CL = MAX_CLUSTERS * SLOTS_PER_BLOCK
# Differs from every platform's auto width (8 on sim, the driver's answer
# onboard), so a count sourced from the ceiling instead of this run fails too.
PINNED_BLOCK_DIM = 4


@scene_test(level=2, runtime="host_build_graph")
class TestAvailableAicoreCounts(SceneTestCase):
"""rt_available_cluster_count() / rt_available_aiv_count() report a spendable width."""

RTOL = 0
ATOL = 0

CALLABLE = {
"orchestration": {
"source": "kernels/orchestration/available_aicore_counts_orch.cpp",
"function_name": "aicpu_orchestration_entry",
"signature": [D.INOUT, D.INOUT],
},
"incores": [
{
"func_id": 0,
"name": "SPMD_MIX_AIC",
"source": "../../tensormap_and_ringbuffer/spmd_multiblock_mix/kernels/aic/kernel_spmd_mix.cpp",
"core_type": "aic",
"signature": [D.INOUT],
},
{
"func_id": 1,
"name": "SPMD_MIX_AIV0",
"source": "../../tensormap_and_ringbuffer/spmd_multiblock_mix/kernels/aiv/kernel_spmd_mix.cpp",
"core_type": "aiv",
"signature": [D.INOUT],
},
{
"func_id": 2,
"name": "SPMD_MIX_AIV1",
"source": "../../tensormap_and_ringbuffer/spmd_multiblock_mix/kernels/aiv/kernel_spmd_mix.cpp",
"core_type": "aiv",
"signature": [D.INOUT],
},
],
}

CASES = [
{
"name": "Pinned",
"platforms": ["a2a3sim", "a2a3"],
"config": {"aicpu_thread_num": 4, "block_dim": PINNED_BLOCK_DIM},
"params": {"expect_clusters": PINNED_BLOCK_DIM},
},
{
"name": "Auto",
"platforms": ["a2a3sim", "a2a3"],
"config": {"aicpu_thread_num": 4},
"params": {},
},
]

def generate_args(self, params):
return TaskArgsBuilder(
Tensor("blocks", torch.zeros(TOTAL_CL * FLOATS_PER_CACHE_LINE, dtype=torch.float32)),
Tensor("shape", torch.zeros(2, dtype=torch.int32)),
)

def compute_golden(self, args, params):
# Both outputs are checked against the reported width in
# compare_outputs; nothing here is host-computable.
pass

def compare_outputs(self, test_args, golden_args, output_names, params):
clusters = int(test_args.shape[0])
aivs = int(test_args.shape[1])
# Only a pinned block_dim gives the host an expectation independent of
# what the runtime reported, so only it can catch an under-count.
expect = params.get("expect_clusters")
if expect is not None:
assert clusters == expect, f"cluster_count {clusters} != pinned block_dim {expect}"
assert 1 <= clusters <= MAX_CLUSTERS, f"cluster_count {clusters} outside [1, {MAX_CLUSTERS}]"
assert aivs == clusters * AIV_PER_CLUSTER, f"aiv_count {aivs} != {clusters} * {AIV_PER_CLUSTER}"

blocks = test_args.blocks.reshape(TOTAL_CL, FLOATS_PER_CACHE_LINE)[:, 0]
expected = torch.zeros(TOTAL_CL, dtype=torch.float32)
for block_idx in range(clusters):
for slot in range(SLOTS_PER_BLOCK):
expected[block_idx * SLOTS_PER_BLOCK + slot] = float(block_idx)
assert torch.equal(blocks, expected), (
f"block slots disagree with the reported cluster_count {clusters}: "
f"got {blocks.tolist()}, expected {expected.tolist()}"
)


if __name__ == "__main__":
SceneTestCase.run_module(__name__)
Original file line number Diff line number Diff line change
Expand Up @@ -72,13 +72,11 @@ __attribute__((visibility("default"))) void aicpu_orchestration_entry(const L2Ta
args.launch_spec.set_require_sync_start(true);
rt_submit_task(mk, args);

// dummy_task stands in as shape's producer so set_tensor_data has a
// producer to wait on.
{
L0TaskArgs shape_args;
shape_args.add_inout(shape);
rt_submit_dummy_task(shape_args);
}
// shape carries no producer or consumer task, so set_tensor_data writes it
// straight through. Giving it one would hang host_build_graph: its
// orchestrator runs to completion on the host before the device executes
// anything, so a producer's task_state can never reach COMPLETED and
// wait_for_tensor_ready would spin to PTO2_TENSOR_DATA_TIMEOUT_CYCLES.
uint32_t idx[1] = {0};
set_tensor_data<int32_t>(shape, 1, idx, cluster_count);
idx[0] = 1;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -9,17 +9,23 @@
# -----------------------------------------------------------------------------------------------------------
"""available_aicore_counts: the runtime's core counts are real and spendable.

The counts have no host-side equivalent — sim and onboard resolve different
values, and onboard's comes from the driver — so there is no constant to pin.
The orchestration instead reports what it saw in `shape` and spends it: a MIX
cohort of exactly `cluster_count` blocks, each writing its block index into
`blocks`. This test reads the width back out of `shape` and checks that many
block slots, so it holds on every platform without knowing any of their numbers.

What would fail: a count of 0 or one exceeding the platform ceiling (checked
here), a count larger than the run really has (the cohort's require_sync_start
needs every block co-resident, so the deadlock guard fires on device), and a
count smaller than it should be (the untouched tail slots stay zero).
The orchestration reports what `rt_available_*_count()` gave it in `shape` and
spends it: a MIX cohort of exactly `cluster_count` blocks, each writing its
block index into `blocks`.

Two cases, because neither alone is sufficient:

- **Pinned** fixes `block_dim`, so the host independently knows the answer and
asserts the exact value. This is the only case that can catch an
**under**-reported count — the Auto case derives its expectation from the
reported number, so a count that is too small is self-consistent there.
- **Auto** takes whatever the platform resolves (sim and onboard differ, and
onboard's comes from the driver) and checks the count is in range, that
`aiv == 2 * clusters`, and that every one of those blocks really ran.

An **over**-reported count fails in both: the cohort asks for
`require_sync_start`, so it needs every block co-resident and the deadlock
guard fires on device.
"""

import torch
Expand All @@ -33,6 +39,9 @@
MAX_CLUSTERS = 24
AIV_PER_CLUSTER = 2
TOTAL_CL = MAX_CLUSTERS * SLOTS_PER_BLOCK
# Differs from every platform's auto width (8 on sim, the driver's answer
# onboard), so a count sourced from the ceiling instead of this run fails too.
PINNED_BLOCK_DIM = 4


@scene_test(level=2, runtime="tensormap_and_ringbuffer")
Expand Down Expand Up @@ -75,7 +84,13 @@ class TestAvailableAicoreCounts(SceneTestCase):

CASES = [
{
"name": "Default",
"name": "Pinned",
"platforms": ["a2a3sim", "a2a3"],
"config": {"aicpu_thread_num": 4, "block_dim": PINNED_BLOCK_DIM},
"params": {"expect_clusters": PINNED_BLOCK_DIM},
},
{
"name": "Auto",
"platforms": ["a2a3sim", "a2a3"],
"config": {"aicpu_thread_num": 4},
"params": {},
Expand All @@ -96,6 +111,11 @@ def compute_golden(self, args, params):
def compare_outputs(self, test_args, golden_args, output_names, params):
clusters = int(test_args.shape[0])
aivs = int(test_args.shape[1])
# Only a pinned block_dim gives the host an expectation independent of
# what the runtime reported, so only it can catch an under-count.
expect = params.get("expect_clusters")
if expect is not None:
assert clusters == expect, f"cluster_count {clusters} != pinned block_dim {expect}"
assert 1 <= clusters <= MAX_CLUSTERS, f"cluster_count {clusters} outside [1, {MAX_CLUSTERS}]"
assert aivs == clusters * AIV_PER_CLUSTER, f"aiv_count {aivs} != {clusters} * {AIV_PER_CLUSTER}"

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -72,13 +72,11 @@ __attribute__((visibility("default"))) void aicpu_orchestration_entry(const L2Ta
args.launch_spec.set_require_sync_start(true);
rt_submit_task(mk, args);

// dummy_task stands in as shape's producer so set_tensor_data has a
// producer to wait on.
{
L0TaskArgs shape_args;
shape_args.add_inout(shape);
rt_submit_dummy_task(shape_args);
}
// shape carries no producer or consumer task, so set_tensor_data writes it
// straight through. Giving it one would hang host_build_graph: its
// orchestrator runs to completion on the host before the device executes
// anything, so a producer's task_state can never reach COMPLETED and
// wait_for_tensor_ready would spin to PTO2_TENSOR_DATA_TIMEOUT_CYCLES.
uint32_t idx[1] = {0};
set_tensor_data<int32_t>(shape, 1, idx, cluster_count);
idx[0] = 1;
Expand Down
Loading
Loading