Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions src/a2a3/platform/include/aicpu/platform_regs.h
Original file line number Diff line number Diff line change
Expand Up @@ -67,10 +67,10 @@ void set_platform_pmu_reg_addrs(uint64_t pmu_regs);
uint64_t get_platform_pmu_reg_addrs();

/**
* Set the ACL device ordinal for the current run. Pushed by the platform layer
* (kernel.cpp) before aicpu_execute() from KernelArgs.device_id; the executor
* reads it to make the staged orchestration SO filename unique per device so
* paired dies sharing the preinstall filesystem never collide.
* Set the ACL device ordinal. Latched once per device by simpler_aicpu_init
* (from InitArgs.device_id) into this resident-SO global; the executor reads it
* to make the staged orchestration SO filename unique per device so paired dies
* sharing the preinstall filesystem never collide.
*/
void set_orch_device_id(int device_id);

Expand Down
19 changes: 10 additions & 9 deletions src/a2a3/platform/include/common/kernel_args.h
Original file line number Diff line number Diff line change
Expand Up @@ -86,9 +86,16 @@ extern "C" {
* - AICore: receives device KernelArgs* via KERNEL_ENTRY
*/
struct KernelArgs {
// Offset-locked front: the front-less launch protocol and the device
// entries require runtime_args @ 0 and regs @ 8 (see static_asserts below).
__may_used_by_aicore__ Runtime *runtime_args{nullptr}; // Task runtime in device memory
uint64_t regs{0}; // Per-core register base address array (platform-specific)
uint64_t ffts_base_addr{0}; // FFTS base address for AICore
// Remaining 64-bit fields. Grouped before the 32-bit tail so the struct
// needs no interior alignment padding — every uint64_t lands on its natural
// 8-byte boundary and the lone trailing uint32_t carries only harmless tail
// padding. Order among these is free (device reads by field name, not
// offset); only runtime_args/regs are offset-locked.
uint64_t ffts_base_addr{0}; // FFTS base address for AICore
uint64_t dump_data_base{0}; // Dump shared memory base address; use explicit flags to detect enablement
// L2 swimlane shared memory base address; use explicit flags to detect enablement
uint64_t l2_swimlane_data_base{0};
Expand All @@ -102,9 +109,6 @@ struct KernelArgs {
// L2SwimlaneAicoreTaskBuffer address. AICore kernel entry indexes by block_idx
// and forwards into platform set/get state. 0 when L2 swimlane is off.
uint64_t l2_swimlane_aicore_rotation_table{0};
uint32_t enable_profiling_flag{0}; // Profiling umbrella bitmask; dump_tensor|l2_swimlane|pmu|dep_gen|scope_stats
uint32_t _pad{0}; // Alignment padding

// Device pointer to the run-wall buffer the platform AICPU entry writes.
// Allocated once and kept resident, reset each run. Onboard AICPU receives
// KernelArgs as a CANN-private copy (see launch_aicpu_kernel), so an
Expand All @@ -118,11 +122,8 @@ struct KernelArgs {
// single-uint64 wall_ns write-through (sim AICPU and host share memory).
// Zero when the buffer was not allocated.
uint64_t device_wall_data_base{0};
// ACL device ordinal. Pushed to the AICPU so the executor can suffix the
// staged orchestration SO name (libdevice_orch_<pid>_<cid>_<device_id>.so):
// paired a2a3 dies share the preinstall filesystem, and a content/pid-only
// name risks a cross-die write/execute collision (see simpler_inner fix).
uint32_t device_id{0};
// 32-bit tail.
uint32_t enable_profiling_flag{0}; // Profiling umbrella bitmask; dump_tensor|l2_swimlane|pmu|dep_gen|scope_stats
};

static_assert(offsetof(KernelArgs, runtime_args) == 0, "KernelArgs::runtime_args offset drift");
Expand Down
23 changes: 3 additions & 20 deletions src/a2a3/platform/onboard/aicpu/kernel.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -35,9 +35,10 @@
// wall = max(end) - min(start). No single-threaded pre-pass is needed to
// seed the start.

// Forward declaration of aicpu_execute (implemented in aicpu_executor.cpp)
// Forward declaration of aicpu_execute (implemented in aicpu_executor.cpp).
// simpler_aicpu_register_callable is NOT declared/forwarded here: it is
// exported directly by the TMARB runtime (host_build_graph does not export it).
extern "C" int aicpu_execute(Runtime *arg);
extern "C" int aicpu_register_callable(const RegisterCallableArgs *arg);

/**
* AICPU kernel main execution entry point.
Expand Down Expand Up @@ -151,21 +152,3 @@ extern "C" __attribute__((visibility("default"))) int simpler_aicpu_init(void *a
LOG_INFO_V0("%s", "simpler_aicpu_init: per-device invariants latched");
return 0;
}

extern "C" __attribute__((visibility("default"))) int simpler_aicpu_register_callable(void *arg) {
if (arg == nullptr) {
LOG_ERROR("%s", "Invalid register_callable kernel arguments: null pointer");
return -1;
}

RegisterCallableArgs *reg_args = reinterpret_cast<RegisterCallableArgs *>(arg);

LOG_INFO_V0("%s", "simpler_aicpu_register_callable: registering callable");
int rc = aicpu_register_callable(reg_args);
if (rc != 0) {
LOG_ERROR("simpler_aicpu_register_callable: registration failed with rc=%d", rc);
return rc;
}
LOG_INFO_V0("%s", "simpler_aicpu_register_callable: registration completed");
return 0;
}
2 changes: 1 addition & 1 deletion src/a2a3/platform/sim/host/device_runner.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -94,7 +94,7 @@ int DeviceRunner::ensure_binaries_loaded() {
};

if (!load_sym("aicpu_execute", reinterpret_cast<void **>(&aicpu_execute_func_))) return -1;
load_optional_sym("aicpu_register_callable", reinterpret_cast<void **>(&aicpu_register_callable_func_));
load_optional_sym("simpler_aicpu_register_callable", reinterpret_cast<void **>(&aicpu_register_callable_func_));
if (!load_sym("set_platform_regs", reinterpret_cast<void **>(&set_platform_regs_func_))) return -1;
load_optional_sym("set_orch_device_id", reinterpret_cast<void **>(&set_orch_device_id_func_));
if (!load_sym("set_platform_dump_base", reinterpret_cast<void **>(&set_platform_dump_base_func_))) return -1;
Expand Down
4 changes: 3 additions & 1 deletion src/a2a3/platform/sim/host/device_runner.h
Original file line number Diff line number Diff line change
Expand Up @@ -55,7 +55,9 @@ class DeviceRunner : public SimDeviceRunnerBase {
// a2a3 sim's dlsym'd function-pointer table. Loaded once via
// ensure_binaries_loaded(), nulled on unload_executor_binaries().
int (*aicpu_execute_func_)(Runtime *){nullptr};
int (*aicpu_register_callable_func_)(const RegisterCallableArgs *){nullptr};
// The runtime exports simpler_aicpu_register_callable(void*) directly (TMARB
// only; hbg does not export it). Optional dlsym: null on the hbg SO.
int (*aicpu_register_callable_func_)(void *){nullptr};
void (*aicore_execute_func_)(Runtime *, int, CoreType, uint32_t, uint64_t, uint32_t, uint64_t){nullptr};
void (*set_platform_regs_func_)(uint64_t){nullptr};
void (*set_orch_device_id_func_)(int){nullptr};
Expand Down
11 changes: 4 additions & 7 deletions src/a2a3/runtime/host_build_graph/aicpu/aicpu_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,6 @@
#include "aicpu/pmu_collector_aicpu.h"
#include "aicpu/tensor_dump_aicpu.h"
#include "callable.h"
#include "common/kernel_args.h"
#include "common/memory_barrier.h"
#include "common/l2_swimlane_profiling.h"
#include "common/platform_config.h"
Expand Down Expand Up @@ -1308,12 +1307,10 @@ void AicpuExecutor::diagnose_stuck_state(

// ===== Public Entry Point =====

extern "C" int aicpu_register_callable(const RegisterCallableArgs *args) {
// host_build_graph resolves orchestration on the host during prepare.
// There is no AICPU orch_so_table_ state to register.
(void)args;
return 0;
}
// host_build_graph resolves orchestration on the host during prepare, so it has
// no device-side registration: it deliberately does NOT export
// simpler_aicpu_register_callable (only the TMARB runtime does). The host's
// register launch is gated on the device-orch path and never targets hbg.

/**
* aicpu_execute - Main AICPU kernel execution entry point
Expand Down
11 changes: 11 additions & 0 deletions src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -504,6 +504,17 @@ int validate_runtime_impl(Runtime *runtime) {
return rc;
}

// host_build_graph resolves orchestration on the host, so it exports no AICPU
// entries beyond the base {simpler_aicpu_exec, simpler_aicpu_init} — in
// particular it does not export simpler_aicpu_register_callable. Reporting an
// empty extra-symbol set keeps the common AICPU loader from looking for it.
const char *const *runtime_extra_aicpu_symbols(size_t *count) {
if (count != nullptr) {
*count = 0;
}
return nullptr;
}

#ifdef __cplusplus
} /* extern "C" */
#endif
17 changes: 11 additions & 6 deletions src/a2a3/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -820,23 +820,28 @@ void AicpuExecutor::deinit(Runtime *runtime) {

// ===== Public Entry Point =====

extern "C" int32_t aicpu_register_callable(const RegisterCallableArgs *args) {
if (args == nullptr) {
LOG_ERROR("%s", "aicpu_register_callable: null RegisterCallableArgs pointer");
// Device orchestration SO registration entry. Exported directly by the runtime
// (not via a platform forwarding shell): registration is a TMARB-only ability,
// so the symbol lives where the capability does. host_build_graph does not
// export it at all (host-side orchestration has nothing to register).
extern "C" __attribute__((visibility("default"))) int simpler_aicpu_register_callable(void *arg) {
if (arg == nullptr) {
LOG_ERROR("%s", "simpler_aicpu_register_callable: null RegisterCallableArgs pointer");
return -1;
}
// `args` is the launch-arg payload CANN copies into the AICPU arg space
const RegisterCallableArgs *args = reinterpret_cast<const RegisterCallableArgs *>(arg);
// `arg` is the launch-arg payload CANN copies into the AICPU arg space
// (same coherent channel exec reads KernelArgs fields from) — no HBM deref,
// so unlike the old prewarm path there is no Runtime to cache-invalidate.
int32_t rc = g_aicpu_executor.load_orch_so(
args->active_callable_id, args->dev_orch_so_addr, args->dev_orch_so_size, args->device_orch_func_name,
args->device_orch_config_name, /*thread_idx=*/0
);
if (rc != 0) {
LOG_ERROR("aicpu_register_callable: SO load failed with rc=%d", rc);
LOG_ERROR("simpler_aicpu_register_callable: SO load failed with rc=%d", rc);
return rc;
}
LOG_INFO_V0("aicpu_register_callable: completed for callable_id=%d", args->active_callable_id);
LOG_INFO_V0("simpler_aicpu_register_callable: completed for callable_id=%d", args->active_callable_id);
return 0;
}

Expand Down
12 changes: 12 additions & 0 deletions src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -743,3 +743,15 @@ extern "C" int validate_runtime_impl(Runtime *runtime) {

return rc;
}

// Extra AICPU entry symbols this runtime exports beyond the base
// {simpler_aicpu_exec, simpler_aicpu_init}. TMARB resolves orchestration on the
// device, so it exports simpler_aicpu_register_callable; the common AICPU loader
// queries this so it carries no runtime-specific symbol knowledge.
extern "C" const char *const *runtime_extra_aicpu_symbols(size_t *count) {
static const char *const kExtra[] = {"simpler_aicpu_register_callable"};
if (count != nullptr) {
*count = sizeof(kExtra) / sizeof(kExtra[0]);
}
return kExtra;
}
8 changes: 4 additions & 4 deletions src/a5/platform/include/aicpu/platform_regs.h
Original file line number Diff line number Diff line change
Expand Up @@ -58,10 +58,10 @@ void set_platform_regs(uint64_t regs);
uint64_t get_platform_regs();

/**
* Set the ACL device ordinal for the current run. Pushed by the platform layer
* (kernel.cpp) before aicpu_execute() from KernelArgs.device_id; the executor
* reads it to make the staged orchestration SO filename unique per device so
* paired dies sharing the preinstall filesystem never collide.
* Set the ACL device ordinal. Latched once per device by simpler_aicpu_init
* (from InitArgs.device_id) into this resident-SO global; the executor reads it
* to make the staged orchestration SO filename unique per device so paired dies
* sharing the preinstall filesystem never collide.
*/
void set_orch_device_id(int device_id);

Expand Down
10 changes: 7 additions & 3 deletions src/a5/platform/include/common/kernel_args.h
Original file line number Diff line number Diff line change
Expand Up @@ -75,8 +75,13 @@ extern "C" {
* - AICore: receives device KernelArgs* via KERNEL_ENTRY
*/
struct KernelArgs {
// Offset-locked front: the front-less launch protocol and the device
// entries require runtime_args @ 0 and regs @ 8 (see static_asserts below).
__may_used_by_aicore__ Runtime *runtime_args{nullptr}; // Task runtime in device memory
uint64_t regs{0}; // Per-core register base address array (platform-specific)
// Remaining 64-bit fields grouped before the 32-bit tail so the struct needs
// no interior alignment padding. Order among these is free (device reads by
// field name, not offset); only runtime_args/regs are offset-locked.
uint64_t dump_data_base{0}; // Dump shared memory base address; use explicit flags to detect enablement
// L2 swimlane shared memory base address; use explicit flags to detect enablement
uint64_t l2_swimlane_data_base{0};
Expand All @@ -91,14 +96,13 @@ struct KernelArgs {
uint64_t scope_stats_data_base{0}; // ScopeStatsBuffer device pointer; 0 when scope_stats is off.
// a5 has no halHostRegister — host keeps a separate shadow and
// refreshes it via rtMemcpy DEVICE_TO_HOST at dump time.
uint32_t enable_profiling_flag{0}; // Profiling umbrella bitmask; dump_tensor|l2_swimlane|pmu|dep_gen|scope_stats
uint32_t _pad{0}; // Alignment padding

// Device pointer to an 8-byte buffer that the platform AICPU entry writes
// the run-wall (ns) into. Allocated once at simpler_init, kept resident.
// See the a2a3 kernel_args.h for the full design rationale (CANN's
// AICPU args copy makes inline fields write-only).
uint64_t device_wall_data_base{0};
// 32-bit tail (two adjacent uint32_t — no interior padding).
uint32_t enable_profiling_flag{0}; // Profiling umbrella bitmask; dump_tensor|l2_swimlane|pmu|dep_gen|scope_stats
// Opaque always-false guard read by the AICore SIMT meta anchor (AIV
// KERNEL_ENTRY). The host never sets it non-zero; its only purpose is to be
// a runtime-valued condition the compiler cannot constant-fold, so the
Expand Down
23 changes: 3 additions & 20 deletions src/a5/platform/onboard/aicpu/kernel.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -35,9 +35,10 @@
// wall = max(end) - min(start). No single-threaded pre-pass is needed to
// seed the start.

// Forward declaration of aicpu_execute (implemented in aicpu_executor.cpp)
// Forward declaration of aicpu_execute (implemented in aicpu_executor.cpp).
// simpler_aicpu_register_callable is NOT declared/forwarded here: it is
// exported directly by the TMARB runtime (host_build_graph does not export it).
extern "C" int aicpu_execute(Runtime *arg);
extern "C" int aicpu_register_callable(const RegisterCallableArgs *arg);

/**
* AICPU kernel main execution entry point.
Expand Down Expand Up @@ -162,21 +163,3 @@ extern "C" __attribute__((visibility("default"))) int simpler_aicpu_init(void *a
LOG_INFO_V0("%s", "simpler_aicpu_init: per-device invariants latched");
return 0;
}

extern "C" __attribute__((visibility("default"))) int simpler_aicpu_register_callable(void *arg) {
if (arg == nullptr) {
LOG_ERROR("%s", "Invalid register_callable kernel arguments: null pointer");
return -1;
}

RegisterCallableArgs *reg_args = reinterpret_cast<RegisterCallableArgs *>(arg);

LOG_INFO_V0("%s", "simpler_aicpu_register_callable: registering callable");
int rc = aicpu_register_callable(reg_args);
if (rc != 0) {
LOG_ERROR("simpler_aicpu_register_callable: registration failed with rc=%d", rc);
return rc;
}
LOG_INFO_V0("%s", "simpler_aicpu_register_callable: registration completed");
return 0;
}
2 changes: 1 addition & 1 deletion src/a5/platform/sim/host/device_runner.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -105,7 +105,7 @@ int DeviceRunner::ensure_binaries_loaded() {
};

if (!load_sym("aicpu_execute", reinterpret_cast<void **>(&aicpu_execute_func_))) return -1;
load_optional_sym("aicpu_register_callable", reinterpret_cast<void **>(&aicpu_register_callable_func_));
load_optional_sym("simpler_aicpu_register_callable", reinterpret_cast<void **>(&aicpu_register_callable_func_));
if (!load_sym("set_platform_regs", reinterpret_cast<void **>(&set_platform_regs_func_))) return -1;
load_optional_sym("set_orch_device_id", reinterpret_cast<void **>(&set_orch_device_id_func_));
if (!load_sym("set_platform_dump_base", reinterpret_cast<void **>(&set_platform_dump_base_func_))) return -1;
Expand Down
4 changes: 3 additions & 1 deletion src/a5/platform/sim/host/device_runner.h
Original file line number Diff line number Diff line change
Expand Up @@ -57,7 +57,9 @@ class DeviceRunner : public SimDeviceRunnerBase {
// a5 sim's dlsym'd function-pointer table. Loaded once via
// ensure_binaries_loaded(), nulled on unload_executor_binaries().
int (*aicpu_execute_func_)(Runtime *){nullptr};
int (*aicpu_register_callable_func_)(const RegisterCallableArgs *){nullptr};
// The runtime exports simpler_aicpu_register_callable(void*) directly (TMARB
// only; hbg does not export it). Optional dlsym: null on the hbg SO.
int (*aicpu_register_callable_func_)(void *){nullptr};
void (*aicore_execute_func_)(Runtime *, int, CoreType, uint32_t, uint64_t, uint32_t, uint64_t, uint64_t){nullptr};
void (*set_platform_regs_func_)(uint64_t){nullptr};
void (*set_orch_device_id_func_)(int){nullptr};
Expand Down
11 changes: 4 additions & 7 deletions src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,6 @@
#include "aicpu/tensor_dump_aicpu.h"
#include "aicpu/platform_regs.h"
#include "callable.h"
#include "common/kernel_args.h"
#include "common/memory_barrier.h"
#include "common/l2_swimlane_profiling.h"
#include "common/platform_config.h"
Expand Down Expand Up @@ -1303,12 +1302,10 @@ void AicpuExecutor::diagnose_stuck_state(

// ===== Public Entry Point =====

extern "C" int aicpu_register_callable(const RegisterCallableArgs *args) {
// host_build_graph resolves orchestration on the host during prepare.
// There is no AICPU orch_so_table_ state to register.
(void)args;
return 0;
}
// host_build_graph resolves orchestration on the host during prepare, so it has
// no device-side registration: it deliberately does NOT export
// simpler_aicpu_register_callable (only the TMARB runtime does). The host's
// register launch is gated on the device-orch path and never targets hbg.

/**
* aicpu_execute - Main AICPU kernel execution entry point
Expand Down
11 changes: 11 additions & 0 deletions src/a5/runtime/host_build_graph/host/runtime_maker.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -504,6 +504,17 @@ int validate_runtime_impl(Runtime *runtime) {
return rc;
}

// host_build_graph resolves orchestration on the host, so it exports no AICPU
// entries beyond the base {simpler_aicpu_exec, simpler_aicpu_init} — in
// particular it does not export simpler_aicpu_register_callable. Reporting an
// empty extra-symbol set keeps the common AICPU loader from looking for it.
const char *const *runtime_extra_aicpu_symbols(size_t *count) {
if (count != nullptr) {
*count = 0;
}
return nullptr;
}

#ifdef __cplusplus
} /* extern "C" */
#endif
17 changes: 11 additions & 6 deletions src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -815,23 +815,28 @@ void AicpuExecutor::deinit(Runtime *runtime) {

// ===== Public Entry Point =====

extern "C" int32_t aicpu_register_callable(const RegisterCallableArgs *args) {
if (args == nullptr) {
LOG_ERROR("%s", "aicpu_register_callable: null RegisterCallableArgs pointer");
// Device orchestration SO registration entry. Exported directly by the runtime
// (not via a platform forwarding shell): registration is a TMARB-only ability,
// so the symbol lives where the capability does. host_build_graph does not
// export it at all (host-side orchestration has nothing to register).
extern "C" __attribute__((visibility("default"))) int simpler_aicpu_register_callable(void *arg) {
if (arg == nullptr) {
LOG_ERROR("%s", "simpler_aicpu_register_callable: null RegisterCallableArgs pointer");
return -1;
}
// `args` is the launch-arg payload CANN copies into the AICPU arg space
const RegisterCallableArgs *args = reinterpret_cast<const RegisterCallableArgs *>(arg);
// `arg` is the launch-arg payload CANN copies into the AICPU arg space
// (same coherent channel exec reads KernelArgs fields from) — no HBM deref,
// so unlike the old prewarm path there is no Runtime to cache-invalidate.
int32_t rc = g_aicpu_executor.load_orch_so(
args->active_callable_id, args->dev_orch_so_addr, args->dev_orch_so_size, args->device_orch_func_name,
args->device_orch_config_name, /*thread_idx=*/0
);
if (rc != 0) {
LOG_ERROR("aicpu_register_callable: SO load failed with rc=%d", rc);
LOG_ERROR("simpler_aicpu_register_callable: SO load failed with rc=%d", rc);
return rc;
}
LOG_INFO_V0("aicpu_register_callable: completed for callable_id=%d", args->active_callable_id);
LOG_INFO_V0("simpler_aicpu_register_callable: completed for callable_id=%d", args->active_callable_id);
return 0;
}

Expand Down
Loading
Loading