From b58965ac38faac72a2e06461903485a3f845bdc0 Mon Sep 17 00:00:00 2001 From: Chao Wang <26245345+ChaoWao@users.noreply.github.com> Date: Sat, 30 May 2026 15:31:58 +0800 Subject: [PATCH] Refactor: drop meaningless `_prepared_` prefix from internal callable methods MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Internal DeviceRunner methods carried a `_prepared_` infix that historically distinguished "register-via-the-prepare_callable-path" from a now-removed eager-register path. With only one path remaining, the prefix is dead naming and creates an asymmetry with the public C API (`prepare_callable` / `unregister_callable` / `run_prepared`). Renames (pure mechanical, no behavior change): - `DeviceRunner::register_prepared_callable[_host_orch]` → `register_callable[_host_orch]` - `DeviceRunner::has_prepared_callable` → `has_callable` - `DeviceRunner::unregister_prepared_callable` → `unregister_callable` - `DeviceRunner::bind_prepared_callable_to_runtime` → `bind_callable_to_runtime` - `prepared_callables_` → `callables_` - `PreparedCallableState` → `CallableState` - `BindPreparedCallableResult` → `BindCallableResult` - `PreparedCallableArtifacts` → `CallableArtifacts` - `bind_prepared_to_runtime_impl` → `bind_callable_to_runtime_impl` (runtime_maker) Public Python / c_api entry points (`prepare_callable`, `run_prepared`, `aicpu_dlopen_count`, `host_dlopen_count`) and the public `unregister_callable` free function are unchanged — those names have real semantic meaning ("prepare the callable", "run something previously prepared"). Both arches built clean (onboard + sim, both runtimes). a2a3 local smoke (dummy_task, prepared_callable suite) 7/7 passed in 11s. --- python/simpler/worker.py | 4 +- .../platform/onboard/host/device_runner.cpp | 79 +++++++++---------- .../platform/onboard/host/device_runner.h | 38 ++++----- .../onboard/host/pto_runtime_c_api.cpp | 28 +++---- src/a2a3/platform/sim/host/device_runner.cpp | 75 ++++++++---------- src/a2a3/platform/sim/host/device_runner.h | 20 ++--- .../platform/sim/host/pto_runtime_c_api.cpp | 22 +++--- .../host_build_graph/host/runtime_maker.cpp | 20 +++-- .../host_build_graph/runtime/runtime.h | 2 +- .../aicpu/aicpu_executor.cpp | 2 +- .../host/runtime_maker.cpp | 13 ++- .../runtime/runtime.h | 4 +- .../platform/onboard/host/device_runner.cpp | 77 ++++++++---------- src/a5/platform/onboard/host/device_runner.h | 16 ++-- .../onboard/host/pto_runtime_c_api.cpp | 24 +++--- src/a5/platform/sim/host/device_runner.cpp | 75 ++++++++---------- src/a5/platform/sim/host/device_runner.h | 14 ++-- .../platform/sim/host/pto_runtime_c_api.cpp | 20 +++-- .../host_build_graph/host/runtime_maker.cpp | 20 +++-- .../aicpu/aicpu_executor.cpp | 2 +- .../host/runtime_maker.cpp | 13 ++- .../runtime/runtime.h | 4 +- src/common/task_interface/callable_protocol.h | 4 +- .../task_interface/prepare_callable_common.h | 14 ++-- .../test_prepared_callable.py | 2 +- tests/ut/py/test_worker/test_host_worker.py | 2 +- 26 files changed, 276 insertions(+), 318 deletions(-) diff --git a/python/simpler/worker.py b/python/simpler/worker.py index 64274d35d8..ab283dcbb6 100644 --- a/python/simpler/worker.py +++ b/python/simpler/worker.py @@ -619,7 +619,7 @@ def _run_chip_main_loop( # noqa: PLR0912 -- TASK_READY + 6 control sub-commands # when a prior _CTRL_UNREGISTER failed before reaching # prepared.discard, while the parent still popped its # registry under best-effort semantics. Without this, - # register_prepared_callable would fail-fast on a slot the + # register_callable would fail-fast on a slot the # user was told is reusable. The `cid in prepared` gate # keeps the happy path at zero added cost. if int(cid) in prepared: @@ -1062,7 +1062,7 @@ def _allocate_cid(self) -> int: # The AICPU side keeps a fixed-size orch_so_table_ keyed by cid; # raise here so the failure surfaces at register-time with a # protocol-aware message, not later from - # DeviceRunner::register_prepared_callable with a generic + # DeviceRunner::register_callable with a generic # "out of range" log. raise RuntimeError( "Worker.register: cid space exhausted " diff --git a/src/a2a3/platform/onboard/host/device_runner.cpp b/src/a2a3/platform/onboard/host/device_runner.cpp index dae47e5c26..f8e92fa976 100644 --- a/src/a2a3/platform/onboard/host/device_runner.cpp +++ b/src/a2a3/platform/onboard/host/device_runner.cpp @@ -607,8 +607,8 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { LOG_ERROR("prepare_orch_so: no active callable_id; prepared-callable flow required"); return -1; } - auto it = prepared_callables_.find(cid); - if (it == prepared_callables_.end()) { + auto it = callables_.find(cid); + if (it == callables_.end()) { LOG_ERROR("prepare_orch_so: callable_id=%d not registered", cid); return -1; } @@ -636,7 +636,7 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { return 0; } -int DeviceRunner::register_prepared_callable( +int DeviceRunner::register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ) { @@ -645,36 +645,34 @@ int DeviceRunner::register_prepared_callable( // callable_id; rejecting an out-of-range id here keeps the host and // AICPU sides in sync and avoids an OOB access at run time. if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { - LOG_ERROR( - "register_prepared_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS - ); + LOG_ERROR("register_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS); return -1; } if (orch_so_data == nullptr || orch_so_size == 0) { - LOG_ERROR("register_prepared_callable: empty orch SO for callable_id=%d", callable_id); + LOG_ERROR("register_callable: empty orch SO for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable: callable_id=%d already registered", callable_id); return -1; } const uint64_t hash = simpler::common::utils::elf_build_id_64(orch_so_data, orch_so_size); // Hash dedup: share device buffer across callable_ids that carry the same - // SO bytes. Refcount drops in unregister_prepared_callable; we only free + // SO bytes. Refcount drops in unregister_callable; we only free // when the count hits zero. auto buf_it = orch_so_dedup_.find(hash); uint64_t dev_addr = 0; if (buf_it == orch_so_dedup_.end()) { void *buf = mem_alloc_.alloc(orch_so_size); if (buf == nullptr) { - LOG_ERROR("register_prepared_callable: alloc %zu bytes failed", orch_so_size); + LOG_ERROR("register_callable: alloc %zu bytes failed", orch_so_size); return -1; } int rc = rtMemcpy(buf, orch_so_size, orch_so_data, orch_so_size, RT_MEMCPY_HOST_TO_DEVICE); if (rc != 0) { - LOG_ERROR("register_prepared_callable: rtMemcpy failed: %d", rc); + LOG_ERROR("register_callable: rtMemcpy failed: %d", rc); mem_alloc_.free(buf); return rc; } @@ -684,16 +682,14 @@ int DeviceRunner::register_prepared_callable( entry.refcount = 1; orch_so_dedup_.emplace(hash, entry); dev_addr = reinterpret_cast(buf); - LOG_INFO_V0("register_prepared_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); + LOG_INFO_V0("register_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); } else { buf_it->second.refcount++; dev_addr = reinterpret_cast(buf_it->second.dev_addr); - LOG_INFO_V0( - "register_prepared_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount - ); + LOG_INFO_V0("register_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount); } - PreparedCallableState state; + CallableState state; state.hash = hash; state.dev_orch_so_addr = dev_addr; state.dev_orch_so_size = orch_so_size; @@ -701,48 +697,47 @@ int DeviceRunner::register_prepared_callable( state.config_name = (config_name != nullptr) ? config_name : ""; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); return 0; } -int DeviceRunner::register_prepared_callable_host_orch( +int DeviceRunner::register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ) { if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { LOG_ERROR( - "register_prepared_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, - MAX_REGISTERED_CALLABLE_IDS + "register_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS ); return -1; } if (host_dlopen_handle == nullptr || host_orch_func_ptr == nullptr) { - LOG_ERROR("register_prepared_callable_host_orch: null handle/fn for callable_id=%d", callable_id); + LOG_ERROR("register_callable_host_orch: null handle/fn for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable_host_orch: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable_host_orch: callable_id=%d already registered", callable_id); return -1; } - PreparedCallableState state; + CallableState state; state.host_dlopen_handle = host_dlopen_handle; state.host_orch_func_ptr = host_orch_func_ptr; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); ++host_dlopen_total_; - LOG_INFO_V0("register_prepared_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); + LOG_INFO_V0("register_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); return 0; } -int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { +int DeviceRunner::unregister_callable(int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { return 0; } - PreparedCallableState state = std::move(it->second); - prepared_callables_.erase(it); + CallableState state = std::move(it->second); + callables_.erase(it); aicpu_seen_callable_ids_.erase(callable_id); if (state.host_dlopen_handle != nullptr) { @@ -761,14 +756,12 @@ int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { return 0; } -bool DeviceRunner::has_prepared_callable(int32_t callable_id) const { - return prepared_callables_.count(callable_id) != 0; -} +bool DeviceRunner::has_callable(int32_t callable_id) const { return callables_.count(callable_id) != 0; } -BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { - LOG_ERROR("bind_prepared_callable_to_runtime: callable_id=%d not registered", callable_id); +BindCallableResult DeviceRunner::bind_callable_to_runtime(Runtime &runtime, int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { + LOG_ERROR("bind_callable_to_runtime: callable_id=%d not registered", callable_id); return {-1, nullptr, nullptr, 0}; } const auto &state = it->second; @@ -779,7 +772,7 @@ BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runti // free kernel binaries — but prepared kernels must survive across runs. for (const auto &kv : state.kernel_addrs) { if (kv.first < 0 || kv.first >= RUNTIME_MAX_FUNC_ID) { - LOG_ERROR("bind_prepared_callable_to_runtime: func_id=%d out of range", kv.first); + LOG_ERROR("bind_callable_to_runtime: func_id=%d out of range", kv.first); return {-1, nullptr, nullptr, 0}; } runtime.replay_function_bin_addr(kv.first, kv.second); @@ -790,7 +783,7 @@ BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runti // with the authoritative first_sighting answer right before launch. runtime.set_active_callable_id(callable_id, /*is_new=*/false); // hbg path: host_orch_func_ptr travels back to the c_api caller, which - // hands it to bind_prepared_to_runtime_impl. trb path: stays null and + // hands it to bind_callable_to_runtime_impl. trb path: stays null and // the device-side orch SO is resolved from the symbol names above. return { 0, state.host_orch_func_ptr, state.signature.empty() ? nullptr : state.signature.data(), @@ -861,12 +854,12 @@ int DeviceRunner::finalize() { // each callable_id, so without this loop the host process leaks one // dlopen handle per (re)created Worker — observable in long-running // pytest sessions. - for (auto &kv : prepared_callables_) { + for (auto &kv : callables_) { if (kv.second.host_dlopen_handle != nullptr) { dlclose(kv.second.host_dlopen_handle); } } - prepared_callables_.clear(); + callables_.clear(); aicpu_seen_callable_ids_.clear(); aicpu_dlopen_total_ = 0; diff --git a/src/a2a3/platform/onboard/host/device_runner.h b/src/a2a3/platform/onboard/host/device_runner.h index cc968dc1a9..c2b3609125 100644 --- a/src/a2a3/platform/onboard/host/device_runner.h +++ b/src/a2a3/platform/onboard/host/device_runner.h @@ -260,22 +260,22 @@ class DeviceRunner : public DeviceRunnerBase { * them onto a fresh Runtime without re-uploading. * @return 0 on success, negative on failure. */ - int register_prepared_callable( + int register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ); /** - * Host-orchestration variant of register_prepared_callable: stores a + * Host-orchestration variant of register_callable: stores a * dlopen handle + entry-symbol pointer that runtime_maker resolved on the * host (host_build_graph variant). Mutually exclusive with the trb-shaped - * `register_prepared_callable` overload — exactly one is invoked for a + * `register_callable` overload — exactly one is invoked for a * given callable_id, picked by the C ABI based on which staging fields the * runtime carries after prepare_callable_impl. dlopen handle is owned by * DeviceRunner from this call onward and dlclose'd by - * unregister_prepared_callable. Increments `host_dlopen_count_`. + * unregister_callable. Increments `host_dlopen_count_`. */ - int register_prepared_callable_host_orch( + int register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ); @@ -287,17 +287,17 @@ class DeviceRunner : public DeviceRunnerBase { * callables and only released by finalize(). * * @param callable_id Id previously passed to one of the - * register_prepared_callable* overloads. + * register_callable* overloads. * @return 0 on success or if the id was not registered. */ - int unregister_prepared_callable(int32_t callable_id); + int unregister_callable(int32_t callable_id); /** * True iff `callable_id` has prepared state staged via - * register_prepared_callable. Lets the c_api layer reject `run_prepared` + * register_callable. Lets the c_api layer reject `run_prepared` * calls without a matching `prepare_callable`. */ - bool has_prepared_callable(int32_t callable_id) const; + bool has_callable(int32_t callable_id) const; /** * Replay the prepared state for `callable_id` onto a freshly-constructed @@ -306,7 +306,7 @@ class DeviceRunner : public DeviceRunnerBase { * subsequent `run` dispatches via the AICPU per-cid table. The kernel * addresses are written directly into func_id_to_addr_ (bypassing * registered_kernel_func_ids_) so validate_runtime_impl will not free them - * — they survive until unregister_prepared_callable / finalize(). + * — they survive until unregister_callable / finalize(). * * Marks the cid as seen so the upcoming prepare_orch_so resolves * `register_new_callable_id_` correctly (true exactly on first sighting @@ -318,10 +318,10 @@ class DeviceRunner : public DeviceRunnerBase { * Replay a previously-registered callable's state onto a fresh Runtime * for a per-run binding. Writes back kernel addrs, orch entry-symbol * names, and active_callable_id; returns the hbg `host_orch_func_ptr` - * (or nullptr on trb / on error) inside a `BindPreparedCallableResult` + * (or nullptr on trb / on error) inside a `BindCallableResult` * so the caller can destructure with structured bindings. */ - BindPreparedCallableResult bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id); + BindCallableResult bind_callable_to_runtime(Runtime &runtime, int32_t callable_id); /** * Number of distinct callable_ids the AICPU has been asked to dlopen for. @@ -335,7 +335,7 @@ class DeviceRunner : public DeviceRunnerBase { /** * Number of host-side dlopen() invocations triggered by - * `register_prepared_callable_host_orch`. Mirrors `aicpu_dlopen_count` but + * `register_callable_host_orch`. Mirrors `aicpu_dlopen_count` but * counts the host_build_graph variant's host-side dlopens; it never * decrements (re-prepare after unregister still counts). Tests assert * `host_dlopen_count == distinct_registered_cids` to verify the prepared @@ -372,14 +372,14 @@ class DeviceRunner : public DeviceRunnerBase { // Per-callable_id prepared state. // - // `prepared_callables_` maps the caller-stable callable_id to the orch + // `callables_` maps the caller-stable callable_id to the orch // SO slice + symbol names needed to launch it. `orch_so_dedup_` shares // device buffers across callable_ids whose orch SO bytes have the same // ELF Build-ID hash (refcounted; freed when the count hits zero). // `aicpu_seen_callable_ids_` tracks which ids have already been delivered // to the AICPU at least once so prepare_orch_so can set // register_new_callable_id_ correctly on first sighting. - struct PreparedCallableState { + struct CallableState { // trb path (AICPU dlopens orch SO from device buffer) uint64_t hash{0}; uint64_t dev_orch_so_addr{0}; @@ -398,7 +398,7 @@ class DeviceRunner : public DeviceRunnerBase { size_t capacity{0}; int refcount{0}; }; - std::unordered_map prepared_callables_; + std::unordered_map callables_; std::unordered_map orch_so_dedup_; std::unordered_set aicpu_seen_callable_ids_; // Monotonic count of AICPU dlopens triggered (incremented on each @@ -407,7 +407,7 @@ class DeviceRunner : public DeviceRunnerBase { // re-prepared. Exposed via aicpu_dlopen_count() for tests. size_t aicpu_dlopen_total_{0}; // Monotonic count of host-side dlopens triggered (incremented on every - // register_prepared_callable_host_orch call; never decremented). Same + // register_callable_host_orch call; never decremented). Same // re-prepare semantics as aicpu_dlopen_total_, but for hbg variants. size_t host_dlopen_total_{0}; // ACL lifecycle (process-wide). aclInit must run exactly once; ensure_acl_ready @@ -431,8 +431,8 @@ class DeviceRunner : public DeviceRunnerBase { /** * Stamp `runtime.{dev_orch_so_addr_, dev_orch_so_size_}` from the - * PreparedCallableState for `runtime.get_active_callable_id()`. The orch - * SO bytes were already H2D'd at `register_prepared_callable` time and + * CallableState for `runtime.get_active_callable_id()`. The orch + * SO bytes were already H2D'd at `register_callable` time and * are shared via `orch_so_dedup_` across cids; this method only refreshes * the device-SO metadata onto the per-run Runtime and bumps the AICPU * first-sighting counter when the cid is new since registration. diff --git a/src/a2a3/platform/onboard/host/pto_runtime_c_api.cpp b/src/a2a3/platform/onboard/host/pto_runtime_c_api.cpp index c23d9cb7b6..47d5f93ad4 100644 --- a/src/a2a3/platform/onboard/host/pto_runtime_c_api.cpp +++ b/src/a2a3/platform/onboard/host/pto_runtime_c_api.cpp @@ -44,10 +44,8 @@ extern "C" { /* =========================================================================== * Runtime Implementation Functions (defined in runtime_maker.cpp) * =========================================================================== */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -); -int bind_prepared_to_runtime_impl( +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out); +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ); @@ -313,7 +311,7 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c int rc = runner->attach_current_thread(runner->device_id()); if (rc != 0) return rc; - PreparedCallableArtifacts artifacts; + CallableArtifacts artifacts; rc = prepare_callable_impl( reinterpret_cast(callable), upload_chip_callable_buffer_wrapper, &artifacts ); @@ -322,8 +320,8 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c } // Re-pack ChildKernelAddr -> std::pair to match the existing - // register_prepared_callable* signature. The named struct only crosses - // the runtime-maker / device-runner interface; PreparedCallableState + // register_callable* signature. The named struct only crosses + // the runtime-maker / device-runner interface; CallableState // stores the historical pair shape. std::vector> kernel_addrs; kernel_addrs.reserve(artifacts.kernel_addrs.size()); @@ -334,12 +332,12 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c // hbg's prepare_callable_impl populates host_dlopen_handle; trb's // leaves it null and fills orch_so_data + func_name/config_name. if (artifacts.host_dlopen_handle != nullptr) { - return runner->register_prepared_callable_host_orch( + return runner->register_callable_host_orch( callable_id, artifacts.host_dlopen_handle, artifacts.host_orch_func_ptr, std::move(kernel_addrs), std::move(artifacts.signature) ); } - return runner->register_prepared_callable( + return runner->register_callable( callable_id, artifacts.orch_so_data, artifacts.orch_so_size, artifacts.func_name.c_str(), artifacts.config_name.c_str(), std::move(kernel_addrs), std::move(artifacts.signature) ); @@ -360,7 +358,7 @@ int run_prepared( if (ctx == NULL || runtime == NULL) return -1; DeviceRunner *runner = static_cast(ctx); - if (!runner->has_prepared_callable(callable_id)) { + if (!runner->has_callable(callable_id)) { LOG_ERROR("run_prepared: callable_id=%d not prepared", callable_id); return -1; } @@ -390,18 +388,18 @@ int run_prepared( // Restore kernel addrs + orch symbol names + active_callable_id; the // returned host_orch_func_ptr is non-null only on the hbg path and is - // handed straight into bind_prepared_to_runtime_impl below. signature + // handed straight into bind_callable_to_runtime_impl below. signature // is the cached ChipCallable signature_[]; it's plumbed end-to-end for // per-tensor direction decisions in runtime_maker but is currently - // unconsumed on both runtimes — see bind_prepared_to_runtime_impl. - auto bind_result = runner->bind_prepared_callable_to_runtime(*r, callable_id); + // unconsumed on both runtimes — see bind_callable_to_runtime_impl. + auto bind_result = runner->bind_callable_to_runtime(*r, callable_id); if (bind_result.rc != 0) { r->~Runtime(); return bind_result.rc; } // Per-run binding (tensor args, GM heap, SM alloc) - rc = bind_prepared_to_runtime_impl( + rc = bind_callable_to_runtime_impl( r, reinterpret_cast(args), bind_result.host_orch_func_ptr, bind_result.signature, bind_result.sig_count ); @@ -443,7 +441,7 @@ int run_prepared( int unregister_callable(DeviceContextHandle ctx, int32_t callable_id) { if (ctx == NULL) return -1; try { - return static_cast(ctx)->unregister_prepared_callable(callable_id); + return static_cast(ctx)->unregister_callable(callable_id); } catch (...) { return -1; } diff --git a/src/a2a3/platform/sim/host/device_runner.cpp b/src/a2a3/platform/sim/host/device_runner.cpp index 5c4f3eac8f..3de9e359b2 100644 --- a/src/a2a3/platform/sim/host/device_runner.cpp +++ b/src/a2a3/platform/sim/host/device_runner.cpp @@ -872,15 +872,15 @@ void DeviceRunner::unload_executor_binaries() { int DeviceRunner::prepare_orch_so(Runtime &runtime) { // Prepared-callable flow only — bytes were staged at - // register_prepared_callable time; here we only stamp metadata onto + // register_callable time; here we only stamp metadata onto // the runtime and resolve `register_new_callable_id_` from first sighting. const int32_t cid = runtime.get_active_callable_id(); if (cid < 0) { LOG_ERROR("prepare_orch_so: no active callable_id; prepared-callable flow required"); return -1; } - auto it = prepared_callables_.find(cid); - if (it == prepared_callables_.end()) { + auto it = callables_.find(cid); + if (it == callables_.end()) { LOG_ERROR("prepare_orch_so: callable_id=%d not registered", cid); return -1; } @@ -905,7 +905,7 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { return 0; } -int DeviceRunner::register_prepared_callable( +int DeviceRunner::register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ) { @@ -914,17 +914,15 @@ int DeviceRunner::register_prepared_callable( // callable_id; rejecting an out-of-range id here keeps the host and // AICPU sides in sync and avoids an OOB access at run time. if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { - LOG_ERROR( - "register_prepared_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS - ); + LOG_ERROR("register_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS); return -1; } if (orch_so_data == nullptr || orch_so_size == 0) { - LOG_ERROR("register_prepared_callable: empty orch SO for callable_id=%d", callable_id); + LOG_ERROR("register_callable: empty orch SO for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable: callable_id=%d already registered", callable_id); return -1; } @@ -935,7 +933,7 @@ int DeviceRunner::register_prepared_callable( if (buf_it == orch_so_dedup_.end()) { void *buf = mem_alloc_.alloc(orch_so_size); if (buf == nullptr) { - LOG_ERROR("register_prepared_callable: alloc %zu bytes failed", orch_so_size); + LOG_ERROR("register_callable: alloc %zu bytes failed", orch_so_size); return -1; } // Sim shares an address space with the simulated AICPU thread, so a @@ -947,16 +945,14 @@ int DeviceRunner::register_prepared_callable( entry.refcount = 1; orch_so_dedup_.emplace(hash, entry); dev_addr = reinterpret_cast(buf); - LOG_INFO_V0("register_prepared_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); + LOG_INFO_V0("register_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); } else { buf_it->second.refcount++; dev_addr = reinterpret_cast(buf_it->second.dev_addr); - LOG_INFO_V0( - "register_prepared_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount - ); + LOG_INFO_V0("register_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount); } - PreparedCallableState state; + CallableState state; state.hash = hash; state.dev_orch_so_addr = dev_addr; state.dev_orch_so_size = orch_so_size; @@ -964,48 +960,47 @@ int DeviceRunner::register_prepared_callable( state.config_name = (config_name != nullptr) ? config_name : ""; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); return 0; } -int DeviceRunner::register_prepared_callable_host_orch( +int DeviceRunner::register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ) { if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { LOG_ERROR( - "register_prepared_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, - MAX_REGISTERED_CALLABLE_IDS + "register_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS ); return -1; } if (host_dlopen_handle == nullptr || host_orch_func_ptr == nullptr) { - LOG_ERROR("register_prepared_callable_host_orch: null handle/fn for callable_id=%d", callable_id); + LOG_ERROR("register_callable_host_orch: null handle/fn for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable_host_orch: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable_host_orch: callable_id=%d already registered", callable_id); return -1; } - PreparedCallableState state; + CallableState state; state.host_dlopen_handle = host_dlopen_handle; state.host_orch_func_ptr = host_orch_func_ptr; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); ++host_dlopen_total_; - LOG_INFO_V0("register_prepared_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); + LOG_INFO_V0("register_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); return 0; } -int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { +int DeviceRunner::unregister_callable(int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { return 0; } - PreparedCallableState state = std::move(it->second); - prepared_callables_.erase(it); + CallableState state = std::move(it->second); + callables_.erase(it); aicpu_seen_callable_ids_.erase(callable_id); if (state.host_dlopen_handle != nullptr) { @@ -1024,20 +1019,18 @@ int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { return 0; } -bool DeviceRunner::has_prepared_callable(int32_t callable_id) const { - return prepared_callables_.count(callable_id) != 0; -} +bool DeviceRunner::has_callable(int32_t callable_id) const { return callables_.count(callable_id) != 0; } -BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { - LOG_ERROR("bind_prepared_callable_to_runtime: callable_id=%d not registered", callable_id); +BindCallableResult DeviceRunner::bind_callable_to_runtime(Runtime &runtime, int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { + LOG_ERROR("bind_callable_to_runtime: callable_id=%d not registered", callable_id); return {-1, nullptr, nullptr, 0}; } const auto &state = it->second; for (const auto &kv : state.kernel_addrs) { if (kv.first < 0 || kv.first >= RUNTIME_MAX_FUNC_ID) { - LOG_ERROR("bind_prepared_callable_to_runtime: func_id=%d out of range", kv.first); + LOG_ERROR("bind_callable_to_runtime: func_id=%d out of range", kv.first); return {-1, nullptr, nullptr, 0}; } runtime.replay_function_bin_addr(kv.first, kv.second); @@ -1087,12 +1080,12 @@ int DeviceRunner::finalize() { // each callable_id, so without this loop the host process leaks one // dlopen handle per (re)created Worker — observable in long-running // pytest sessions. - for (auto &kv : prepared_callables_) { + for (auto &kv : callables_) { if (kv.second.host_dlopen_handle != nullptr) { dlclose(kv.second.host_dlopen_handle); } } - prepared_callables_.clear(); + callables_.clear(); aicpu_seen_callable_ids_.clear(); aicpu_dlopen_total_ = 0; diff --git a/src/a2a3/platform/sim/host/device_runner.h b/src/a2a3/platform/sim/host/device_runner.h index d5134dcce1..d0b29a765c 100644 --- a/src/a2a3/platform/sim/host/device_runner.h +++ b/src/a2a3/platform/sim/host/device_runner.h @@ -257,20 +257,20 @@ class DeviceRunner { */ uint64_t upload_chip_callable_buffer(const ChipCallable *callable); - int register_prepared_callable( + int register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ); - // Host-orchestration sibling of register_prepared_callable; see + // Host-orchestration sibling of register_callable; see // src/a2a3/platform/onboard/host/device_runner.h for the contract. Sim // shares the host-only dlopen path verbatim (no AICPU side effects). - int register_prepared_callable_host_orch( + int register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ); - int unregister_prepared_callable(int32_t callable_id); - bool has_prepared_callable(int32_t callable_id) const; - BindPreparedCallableResult bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id); + int unregister_callable(int32_t callable_id); + bool has_callable(int32_t callable_id) const; + BindCallableResult bind_callable_to_runtime(Runtime &runtime, int32_t callable_id); size_t aicpu_dlopen_count() const { return aicpu_dlopen_total_; } size_t host_dlopen_count() const { return host_dlopen_total_; } @@ -341,7 +341,7 @@ class DeviceRunner { std::unordered_map chip_callable_buffers_; // Per-callable_id prepared state. Mirrors onboard. - struct PreparedCallableState { + struct CallableState { // trb path uint64_t hash{0}; uint64_t dev_orch_so_addr{0}; @@ -360,7 +360,7 @@ class DeviceRunner { size_t capacity{0}; int refcount{0}; }; - std::unordered_map prepared_callables_; + std::unordered_map callables_; std::unordered_map orch_so_dedup_; std::unordered_set aicpu_seen_callable_ids_; size_t aicpu_dlopen_total_{0}; @@ -412,9 +412,9 @@ class DeviceRunner { /** * Stamp `runtime.{dev_orch_so_addr_, dev_orch_so_size_}` from the - * PreparedCallableState for `runtime.get_active_callable_id()`. Identical + * CallableState for `runtime.get_active_callable_id()`. Identical * contract to the onboard version: bytes were staged at - * `register_prepared_callable` time, so this is metadata-only — no copy. + * `register_callable` time, so this is metadata-only — no copy. */ int prepare_orch_so(Runtime &runtime); diff --git a/src/a2a3/platform/sim/host/pto_runtime_c_api.cpp b/src/a2a3/platform/sim/host/pto_runtime_c_api.cpp index 9b5d1b4523..16de55d6c8 100644 --- a/src/a2a3/platform/sim/host/pto_runtime_c_api.cpp +++ b/src/a2a3/platform/sim/host/pto_runtime_c_api.cpp @@ -39,10 +39,8 @@ extern "C" { /* =========================================================================== * Runtime Implementation Functions (defined in runtime_maker.cpp) * =========================================================================== */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -); -int bind_prepared_to_runtime_impl( +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out); +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ); @@ -274,7 +272,7 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c pthread_setspecific(g_runner_key, ctx); try { - PreparedCallableArtifacts artifacts; + CallableArtifacts artifacts; int rc = prepare_callable_impl( reinterpret_cast(callable), upload_chip_callable_buffer_wrapper, &artifacts ); @@ -284,7 +282,7 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c } // Re-pack ChildKernelAddr -> std::pair to match the existing - // register_prepared_callable* signature. + // register_callable* signature. std::vector> kernel_addrs; kernel_addrs.reserve(artifacts.kernel_addrs.size()); for (const ChildKernelAddr &c : artifacts.kernel_addrs) { @@ -292,12 +290,12 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c } if (artifacts.host_dlopen_handle != nullptr) { - rc = runner->register_prepared_callable_host_orch( + rc = runner->register_callable_host_orch( callable_id, artifacts.host_dlopen_handle, artifacts.host_orch_func_ptr, std::move(kernel_addrs), std::move(artifacts.signature) ); } else { - rc = runner->register_prepared_callable( + rc = runner->register_callable( callable_id, artifacts.orch_so_data, artifacts.orch_so_size, artifacts.func_name.c_str(), artifacts.config_name.c_str(), std::move(kernel_addrs), std::move(artifacts.signature) ); @@ -322,7 +320,7 @@ int run_prepared( if (ctx == NULL || runtime == NULL) return -1; DeviceRunner *runner = static_cast(ctx); - if (!runner->has_prepared_callable(callable_id)) { + if (!runner->has_callable(callable_id)) { LOG_ERROR("run_prepared: callable_id=%d not prepared", callable_id); return -1; } @@ -344,7 +342,7 @@ int run_prepared( r->host_api.acquire_pooled_runtime_arena = acquire_pooled_runtime_arena_wrapper; r->host_api.upload_chip_callable_buffer = upload_chip_callable_buffer_wrapper; - auto bind_result = runner->bind_prepared_callable_to_runtime(*r, callable_id); + auto bind_result = runner->bind_callable_to_runtime(*r, callable_id); int rc = bind_result.rc; if (rc != 0) { r->~Runtime(); @@ -352,7 +350,7 @@ int run_prepared( return rc; } - rc = bind_prepared_to_runtime_impl( + rc = bind_callable_to_runtime_impl( r, reinterpret_cast(args), bind_result.host_orch_func_ptr, bind_result.signature, bind_result.sig_count ); @@ -398,7 +396,7 @@ int run_prepared( int unregister_callable(DeviceContextHandle ctx, int32_t callable_id) { if (ctx == NULL) return -1; try { - return static_cast(ctx)->unregister_prepared_callable(callable_id); + return static_cast(ctx)->unregister_callable(callable_id); } catch (...) { return -1; } diff --git a/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp b/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp index ad2e099edd..3aa0e6d10e 100644 --- a/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp +++ b/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp @@ -280,14 +280,12 @@ extern "C" { * Stage the per-callable resources for the host_build_graph variant: upload * kernel binaries and dlopen the orchestration SO on the host. The dlopen * handle and resolved entry-symbol pointer are returned via - * PreparedCallableArtifacts so the platform layer can hoist them into its - * PreparedCallableState. Splitting this out of init_runtime_impl is what + * CallableArtifacts so the platform layer can hoist them into its + * CallableState. Splitting this out of init_runtime_impl is what * the hbg prepare_callable / run_prepared path rests on — the dlopen runs * once per cid instead of every run. */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -) { +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out) { if (callable == nullptr) { LOG_ERROR("Callable pointer is null"); return -1; @@ -296,7 +294,7 @@ int prepare_callable_impl( LOG_ERROR("upload_fn or out is null"); return -1; } - *out = PreparedCallableArtifacts{}; + *out = CallableArtifacts{}; out->signature.assign(callable->signature_, callable->signature_ + callable->sig_count()); LOG_INFO_V0("Registering %d kernel(s) in prepare_callable_impl", callable->child_count()); @@ -322,7 +320,7 @@ int prepare_callable_impl( // Load orchestration SO from binary data via temp file. Held open across // the lifetime of the prepared callable; closed by - // DeviceRunner::unregister_prepared_callable. + // DeviceRunner::unregister_callable. std::string fd_path; if (!create_temp_so_file(orch_so_binary, orch_so_size, &fd_path)) { LOG_ERROR("Failed to create temp SO file"); @@ -356,10 +354,10 @@ int prepare_callable_impl( * Per-run binding for hbg: invoke the previously-resolved orchestration entry * point against the supplied args, then upload tensor info / allocation * storage. The c_api caller passes `host_orch_func_ptr` straight through from - * DeviceRunner::bind_prepared_callable_to_runtime (which read it from - * PreparedCallableState for this run's callable_id). + * DeviceRunner::bind_callable_to_runtime (which read it from + * CallableState for this run's callable_id). */ -int bind_prepared_to_runtime_impl( +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ) { @@ -373,7 +371,7 @@ int bind_prepared_to_runtime_impl( } OrchestrationFunc orch_func = reinterpret_cast(host_orch_func_ptr); if (orch_func == nullptr) { - LOG_ERROR("bind_prepared_to_runtime_impl: host orch_func pointer is null"); + LOG_ERROR("bind_callable_to_runtime_impl: host orch_func pointer is null"); return -1; } diff --git a/src/a2a3/runtime/host_build_graph/runtime/runtime.h b/src/a2a3/runtime/host_build_graph/runtime/runtime.h index ccdc05ce06..346c3b9fd3 100644 --- a/src/a2a3/runtime/host_build_graph/runtime/runtime.h +++ b/src/a2a3/runtime/host_build_graph/runtime/runtime.h @@ -423,7 +423,7 @@ class Runtime { /** * Replay a previously-uploaded kernel address onto a fresh Runtime * without recording it in registered_kernel_func_ids_. Used by - * DeviceRunner::bind_prepared_callable_to_runtime when restoring kernels + * DeviceRunner::bind_callable_to_runtime when restoring kernels * across run_prepared invocations: the prepared callable owns the * kernel binaries' device memory until unregister, so * validate_runtime_impl must NOT free them. diff --git a/src/a2a3/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp b/src/a2a3/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp index f468ead395..7a49663611 100644 --- a/src/a2a3/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp +++ b/src/a2a3/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp @@ -97,7 +97,7 @@ static PTO2Runtime *rt{nullptr}; // that callable_id, kept warm across runs). // MAX_REGISTERED_CALLABLE_IDS is the protocol hard cap on callable_id values // (mailbox uint32 callable_id, register() returns small ints) and is shared -// with the host bounds check in DeviceRunner::register_prepared_callable — +// with the host bounds check in DeviceRunner::register_callable — // see src/common/task_interface/callable_protocol.h. struct OrchSoEntry { diff --git a/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp b/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp index e40aa5ae72..6638d20a61 100644 --- a/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp +++ b/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp @@ -96,7 +96,7 @@ static int32_t pto2_read_runtime_status(Runtime *runtime, PTO2SharedMemoryHeader /** * Stage the per-callable resources (kernel binaries + orchestration SO) into - * the supplied runtime so a subsequent bind_prepared_to_runtime_impl can use + * the supplied runtime so a subsequent bind_callable_to_runtime_impl can use * them. This is the cacheable half of init_runtime_impl: nothing here depends * on per-run argument values, so the prepare_callable / run_prepared split * lets us run this once per callable_id and amortize across runs. @@ -105,9 +105,8 @@ static int32_t pto2_read_runtime_status(Runtime *runtime, PTO2SharedMemoryHeader * @param callable ChipCallable carrying the orch SO + child kernel binaries * @return 0 on success, -1 on failure */ -extern "C" int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -) { +extern "C" int +prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out) { if (callable == nullptr) { LOG_ERROR("Callable pointer is null"); return -1; @@ -116,7 +115,7 @@ extern "C" int prepare_callable_impl( LOG_ERROR("upload_fn or out is null"); return -1; } - *out = PreparedCallableArtifacts{}; + *out = CallableArtifacts{}; out->signature.assign(callable->signature_, callable->signature_ + callable->sig_count()); LOG_INFO_V0("Registering %d kernel(s) in prepare_callable_impl", callable->child_count()); @@ -161,7 +160,7 @@ extern "C" int prepare_callable_impl( * @param orch_args Separated tensor/scalar arguments for this run * @return 0 on success, -1 on failure */ -extern "C" int bind_prepared_to_runtime_impl( +extern "C" int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection * /*signature*/, int /*sig_count*/ ) { @@ -177,7 +176,7 @@ extern "C" int bind_prepared_to_runtime_impl( // function pointer to invoke. The c_api signature accepts one for // symmetry with hbg; assert the trb-side invariant here. if (host_orch_func_ptr != nullptr) { - LOG_ERROR("bind_prepared_to_runtime_impl: trb does not accept a host_orch_func_ptr"); + LOG_ERROR("bind_callable_to_runtime_impl: trb does not accept a host_orch_func_ptr"); return -1; } diff --git a/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/runtime.h b/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/runtime.h index 8e1bb1567e..a829fecd06 100644 --- a/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/runtime.h +++ b/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/runtime.h @@ -261,7 +261,7 @@ class Runtime { void set_orch_args(const ChipStorageTaskArgs &args); // Prebuilt-arena fast path (trb only). Set by host's - // bind_prepared_to_runtime_impl; consumed by AICPU at boot to attach a + // bind_callable_to_runtime_impl; consumed by AICPU at boot to attach a // DeviceArena to `prebuilt_arena_base_` and pick up the PTO2Runtime at // `prebuilt_arena_base_ + prebuilt_runtime_offset_`. Both stay zero on // first construction (Runtime() ctor zeros them) so a non-prebuilt boot @@ -291,7 +291,7 @@ class Runtime { /** * Replay a previously-uploaded kernel address onto a fresh Runtime * without recording it in registered_kernel_func_ids_. Used by - * DeviceRunner::bind_prepared_callable_to_runtime so prepared kernel + * DeviceRunner::bind_callable_to_runtime so prepared kernel * binaries are not freed by validate_runtime_impl across runs. */ void replay_function_bin_addr(int func_id, uint64_t addr); diff --git a/src/a5/platform/onboard/host/device_runner.cpp b/src/a5/platform/onboard/host/device_runner.cpp index 36b0370fe1..7644ddd5b2 100644 --- a/src/a5/platform/onboard/host/device_runner.cpp +++ b/src/a5/platform/onboard/host/device_runner.cpp @@ -425,8 +425,8 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { LOG_ERROR("prepare_orch_so: no active callable_id; prepared-callable flow required"); return -1; } - auto it = prepared_callables_.find(cid); - if (it == prepared_callables_.end()) { + auto it = callables_.find(cid); + if (it == callables_.end()) { LOG_ERROR("prepare_orch_so: callable_id=%d not registered", cid); return -1; } @@ -453,7 +453,7 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { return 0; } -int DeviceRunner::register_prepared_callable( +int DeviceRunner::register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ) { @@ -462,36 +462,34 @@ int DeviceRunner::register_prepared_callable( // it by callable_id; rejecting an out-of-range id here keeps host and AICPU // in sync and avoids an OOB access at run time. if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { - LOG_ERROR( - "register_prepared_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS - ); + LOG_ERROR("register_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS); return -1; } if (orch_so_data == nullptr || orch_so_size == 0) { - LOG_ERROR("register_prepared_callable: empty orch SO for callable_id=%d", callable_id); + LOG_ERROR("register_callable: empty orch SO for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable: callable_id=%d already registered", callable_id); return -1; } const uint64_t hash = simpler::common::utils::elf_build_id_64(orch_so_data, orch_so_size); // Hash dedup: share device buffer across callable_ids that carry the same - // SO bytes. Refcount drops in unregister_prepared_callable; we only free + // SO bytes. Refcount drops in unregister_callable; we only free // when the count hits zero. auto buf_it = orch_so_dedup_.find(hash); uint64_t dev_addr = 0; if (buf_it == orch_so_dedup_.end()) { void *buf = mem_alloc_.alloc(orch_so_size); if (buf == nullptr) { - LOG_ERROR("register_prepared_callable: alloc %zu bytes failed", orch_so_size); + LOG_ERROR("register_callable: alloc %zu bytes failed", orch_so_size); return -1; } int rc = rtMemcpy(buf, orch_so_size, orch_so_data, orch_so_size, RT_MEMCPY_HOST_TO_DEVICE); if (rc != 0) { - LOG_ERROR("register_prepared_callable: rtMemcpy failed: %d", rc); + LOG_ERROR("register_callable: rtMemcpy failed: %d", rc); mem_alloc_.free(buf); return rc; } @@ -501,16 +499,14 @@ int DeviceRunner::register_prepared_callable( entry.refcount = 1; orch_so_dedup_.emplace(hash, entry); dev_addr = reinterpret_cast(buf); - LOG_INFO_V0("register_prepared_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); + LOG_INFO_V0("register_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); } else { buf_it->second.refcount++; dev_addr = reinterpret_cast(buf_it->second.dev_addr); - LOG_INFO_V0( - "register_prepared_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount - ); + LOG_INFO_V0("register_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount); } - PreparedCallableState state; + CallableState state; state.hash = hash; state.dev_orch_so_addr = dev_addr; state.dev_orch_so_size = orch_so_size; @@ -518,48 +514,47 @@ int DeviceRunner::register_prepared_callable( state.config_name = (config_name != nullptr) ? config_name : ""; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); return 0; } -int DeviceRunner::register_prepared_callable_host_orch( +int DeviceRunner::register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ) { if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { LOG_ERROR( - "register_prepared_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, - MAX_REGISTERED_CALLABLE_IDS + "register_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS ); return -1; } if (host_dlopen_handle == nullptr || host_orch_func_ptr == nullptr) { - LOG_ERROR("register_prepared_callable_host_orch: null handle/fn for callable_id=%d", callable_id); + LOG_ERROR("register_callable_host_orch: null handle/fn for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable_host_orch: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable_host_orch: callable_id=%d already registered", callable_id); return -1; } - PreparedCallableState state; + CallableState state; state.host_dlopen_handle = host_dlopen_handle; state.host_orch_func_ptr = host_orch_func_ptr; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); ++host_dlopen_total_; - LOG_INFO_V0("register_prepared_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); + LOG_INFO_V0("register_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); return 0; } -int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { +int DeviceRunner::unregister_callable(int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { return 0; } - PreparedCallableState state = std::move(it->second); - prepared_callables_.erase(it); + CallableState state = std::move(it->second); + callables_.erase(it); aicpu_seen_callable_ids_.erase(callable_id); if (state.host_dlopen_handle != nullptr) { @@ -578,14 +573,12 @@ int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { return 0; } -bool DeviceRunner::has_prepared_callable(int32_t callable_id) const { - return prepared_callables_.count(callable_id) != 0; -} +bool DeviceRunner::has_callable(int32_t callable_id) const { return callables_.count(callable_id) != 0; } -BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { - LOG_ERROR("bind_prepared_callable_to_runtime: callable_id=%d not registered", callable_id); +BindCallableResult DeviceRunner::bind_callable_to_runtime(Runtime &runtime, int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { + LOG_ERROR("bind_callable_to_runtime: callable_id=%d not registered", callable_id); return {-1, nullptr, nullptr, 0}; } const auto &state = it->second; @@ -597,7 +590,7 @@ BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runti // be freed by finalize(). for (const auto &kv : state.kernel_addrs) { if (kv.first < 0 || kv.first >= RUNTIME_MAX_FUNC_ID) { - LOG_ERROR("bind_prepared_callable_to_runtime: func_id=%d out of range", kv.first); + LOG_ERROR("bind_callable_to_runtime: func_id=%d out of range", kv.first); return {-1, nullptr, nullptr, 0}; } runtime.replay_function_bin_addr(kv.first, kv.second); @@ -677,12 +670,12 @@ int DeviceRunner::finalize() { // each callable_id, so without this loop the host process leaks one // dlopen handle per (re)created Worker — observable in long-running // pytest sessions. - for (auto &kv : prepared_callables_) { + for (auto &kv : callables_) { if (kv.second.host_dlopen_handle != nullptr) { dlclose(kv.second.host_dlopen_handle); } } - prepared_callables_.clear(); + callables_.clear(); aicpu_seen_callable_ids_.clear(); aicpu_dlopen_total_ = 0; diff --git a/src/a5/platform/onboard/host/device_runner.h b/src/a5/platform/onboard/host/device_runner.h index 666f1671ee..cc88eb8100 100644 --- a/src/a5/platform/onboard/host/device_runner.h +++ b/src/a5/platform/onboard/host/device_runner.h @@ -212,7 +212,7 @@ class DeviceRunner : public DeviceRunnerBase { * them onto a fresh Runtime without re-uploading. * @return 0 on success, negative on failure. */ - int register_prepared_callable( + int register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ); @@ -222,7 +222,7 @@ class DeviceRunner : public DeviceRunnerBase { * device_runner.h for full contract. Mutually exclusive with the * trb-shaped overload. */ - int register_prepared_callable_host_orch( + int register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ); @@ -232,16 +232,16 @@ class DeviceRunner : public DeviceRunnerBase { * refcount, free when zero. hbg path: dlclose the host handle. Kernel * binaries are shared and only released by finalize(). */ - int unregister_prepared_callable(int32_t callable_id); + int unregister_callable(int32_t callable_id); /** True iff `callable_id` has prepared state staged. */ - bool has_prepared_callable(int32_t callable_id) const; + bool has_callable(int32_t callable_id) const; /** * Replay the prepared state for `callable_id` onto a freshly-constructed * Runtime. See a2a3 onboard documentation for full contract. */ - BindPreparedCallableResult bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id); + BindCallableResult bind_callable_to_runtime(Runtime &runtime, int32_t callable_id); /** * Number of distinct callable_ids the AICPU has been asked to dlopen for. @@ -251,7 +251,7 @@ class DeviceRunner : public DeviceRunnerBase { /** * Number of host-side dlopens triggered by - * `register_prepared_callable_host_orch` (hbg variant). Mirrors + * `register_callable_host_orch` (hbg variant). Mirrors * `aicpu_dlopen_count` for the host-orchestration path. */ size_t host_dlopen_count() const { return host_dlopen_total_; } @@ -286,7 +286,7 @@ class DeviceRunner : public DeviceRunnerBase { // Per-callable_id prepared state. See a2a3 onboard device_runner.h for // the full design narrative; mirrored here so a5 shares the same // dispatch surface. - struct PreparedCallableState { + struct CallableState { // trb path uint64_t hash{0}; uint64_t dev_orch_so_addr{0}; @@ -305,7 +305,7 @@ class DeviceRunner : public DeviceRunnerBase { size_t capacity{0}; int refcount{0}; }; - std::unordered_map prepared_callables_; + std::unordered_map callables_; std::unordered_map orch_so_dedup_; std::unordered_set aicpu_seen_callable_ids_; // Monotonic AICPU dlopen counter (first-sighting bind only; never decremented). diff --git a/src/a5/platform/onboard/host/pto_runtime_c_api.cpp b/src/a5/platform/onboard/host/pto_runtime_c_api.cpp index 6b743d00cf..450c9fa824 100644 --- a/src/a5/platform/onboard/host/pto_runtime_c_api.cpp +++ b/src/a5/platform/onboard/host/pto_runtime_c_api.cpp @@ -44,10 +44,8 @@ extern "C" { /* =========================================================================== * Runtime Implementation Functions (defined in runtime_maker.cpp) * =========================================================================== */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -); -int bind_prepared_to_runtime_impl( +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out); +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ); @@ -376,7 +374,7 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c int rc = runner->attach_current_thread(runner->device_id()); if (rc != 0) return rc; - PreparedCallableArtifacts artifacts; + CallableArtifacts artifacts; rc = prepare_callable_impl( reinterpret_cast(callable), upload_chip_callable_buffer_wrapper, &artifacts ); @@ -391,12 +389,12 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c } if (artifacts.host_dlopen_handle != nullptr) { - return runner->register_prepared_callable_host_orch( + return runner->register_callable_host_orch( callable_id, artifacts.host_dlopen_handle, artifacts.host_orch_func_ptr, std::move(kernel_addrs), std::move(artifacts.signature) ); } - return runner->register_prepared_callable( + return runner->register_callable( callable_id, artifacts.orch_so_data, artifacts.orch_so_size, artifacts.func_name.c_str(), artifacts.config_name.c_str(), std::move(kernel_addrs), std::move(artifacts.signature) ); @@ -417,7 +415,7 @@ int run_prepared( if (ctx == NULL || runtime == NULL) return -1; DeviceRunner *runner = static_cast(ctx); - if (!runner->has_prepared_callable(callable_id)) { + if (!runner->has_callable(callable_id)) { LOG_ERROR("run_prepared: callable_id=%d not prepared", callable_id); return -1; } @@ -447,18 +445,18 @@ int run_prepared( // Restore kernel addrs + orch symbol names + active_callable_id; the // returned host_orch_func_ptr is non-null only on the hbg path and is - // handed straight into bind_prepared_to_runtime_impl below. signature + // handed straight into bind_callable_to_runtime_impl below. signature // is the cached ChipCallable signature_[]; it's plumbed end-to-end for // per-tensor direction decisions in runtime_maker but is currently - // unconsumed on both runtimes — see bind_prepared_to_runtime_impl. - auto bind_result = runner->bind_prepared_callable_to_runtime(*r, callable_id); + // unconsumed on both runtimes — see bind_callable_to_runtime_impl. + auto bind_result = runner->bind_callable_to_runtime(*r, callable_id); if (bind_result.rc != 0) { r->~Runtime(); return bind_result.rc; } // Per-run binding (tensor args, GM heap, SM alloc) - rc = bind_prepared_to_runtime_impl( + rc = bind_callable_to_runtime_impl( r, reinterpret_cast(args), bind_result.host_orch_func_ptr, bind_result.signature, bind_result.sig_count ); @@ -499,7 +497,7 @@ int run_prepared( int unregister_callable(DeviceContextHandle ctx, int32_t callable_id) { if (ctx == NULL) return -1; try { - return static_cast(ctx)->unregister_prepared_callable(callable_id); + return static_cast(ctx)->unregister_callable(callable_id); } catch (...) { return -1; } diff --git a/src/a5/platform/sim/host/device_runner.cpp b/src/a5/platform/sim/host/device_runner.cpp index 728f20c16a..85b79fbdfc 100644 --- a/src/a5/platform/sim/host/device_runner.cpp +++ b/src/a5/platform/sim/host/device_runner.cpp @@ -765,15 +765,15 @@ void DeviceRunner::unload_executor_binaries() { int DeviceRunner::prepare_orch_so(Runtime &runtime) { // Prepared-callable flow only — bytes were staged at - // register_prepared_callable time; here we only stamp metadata onto + // register_callable time; here we only stamp metadata onto // the runtime and resolve `register_new_callable_id_` from first sighting. const int32_t cid = runtime.get_active_callable_id(); if (cid < 0) { LOG_ERROR("prepare_orch_so: no active callable_id; prepared-callable flow required"); return -1; } - auto it = prepared_callables_.find(cid); - if (it == prepared_callables_.end()) { + auto it = callables_.find(cid); + if (it == callables_.end()) { LOG_ERROR("prepare_orch_so: callable_id=%d not registered", cid); return -1; } @@ -797,22 +797,20 @@ int DeviceRunner::prepare_orch_so(Runtime &runtime) { return 0; } -int DeviceRunner::register_prepared_callable( +int DeviceRunner::register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ) { if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { - LOG_ERROR( - "register_prepared_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS - ); + LOG_ERROR("register_callable: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS); return -1; } if (orch_so_data == nullptr || orch_so_size == 0) { - LOG_ERROR("register_prepared_callable: empty orch SO for callable_id=%d", callable_id); + LOG_ERROR("register_callable: empty orch SO for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable: callable_id=%d already registered", callable_id); return -1; } @@ -823,7 +821,7 @@ int DeviceRunner::register_prepared_callable( if (buf_it == orch_so_dedup_.end()) { void *buf = mem_alloc_.alloc(orch_so_size); if (buf == nullptr) { - LOG_ERROR("register_prepared_callable: alloc %zu bytes failed", orch_so_size); + LOG_ERROR("register_callable: alloc %zu bytes failed", orch_so_size); return -1; } // Sim shares an address space with the simulated AICPU thread, so a @@ -835,16 +833,14 @@ int DeviceRunner::register_prepared_callable( entry.refcount = 1; orch_so_dedup_.emplace(hash, entry); dev_addr = reinterpret_cast(buf); - LOG_INFO_V0("register_prepared_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); + LOG_INFO_V0("register_callable: hash=0x%lx new buffer %zu bytes", hash, orch_so_size); } else { buf_it->second.refcount++; dev_addr = reinterpret_cast(buf_it->second.dev_addr); - LOG_INFO_V0( - "register_prepared_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount - ); + LOG_INFO_V0("register_callable: hash=0x%lx shared buffer (refcount=%d)", hash, buf_it->second.refcount); } - PreparedCallableState state; + CallableState state; state.hash = hash; state.dev_orch_so_addr = dev_addr; state.dev_orch_so_size = orch_so_size; @@ -852,48 +848,47 @@ int DeviceRunner::register_prepared_callable( state.config_name = (config_name != nullptr) ? config_name : ""; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); return 0; } -int DeviceRunner::register_prepared_callable_host_orch( +int DeviceRunner::register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ) { if (callable_id < 0 || callable_id >= MAX_REGISTERED_CALLABLE_IDS) { LOG_ERROR( - "register_prepared_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, - MAX_REGISTERED_CALLABLE_IDS + "register_callable_host_orch: callable_id=%d out of range [0, %d)", callable_id, MAX_REGISTERED_CALLABLE_IDS ); return -1; } if (host_dlopen_handle == nullptr || host_orch_func_ptr == nullptr) { - LOG_ERROR("register_prepared_callable_host_orch: null handle/fn for callable_id=%d", callable_id); + LOG_ERROR("register_callable_host_orch: null handle/fn for callable_id=%d", callable_id); return -1; } - if (prepared_callables_.count(callable_id) != 0) { - LOG_ERROR("register_prepared_callable_host_orch: callable_id=%d already registered", callable_id); + if (callables_.count(callable_id) != 0) { + LOG_ERROR("register_callable_host_orch: callable_id=%d already registered", callable_id); return -1; } - PreparedCallableState state; + CallableState state; state.host_dlopen_handle = host_dlopen_handle; state.host_orch_func_ptr = host_orch_func_ptr; state.kernel_addrs = std::move(kernel_addrs); state.signature = std::move(signature); - prepared_callables_.emplace(callable_id, std::move(state)); + callables_.emplace(callable_id, std::move(state)); ++host_dlopen_total_; - LOG_INFO_V0("register_prepared_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); + LOG_INFO_V0("register_callable_host_orch: cid=%d (host dlopen #%zu)", callable_id, host_dlopen_total_); return 0; } -int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { +int DeviceRunner::unregister_callable(int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { return 0; } - PreparedCallableState state = std::move(it->second); - prepared_callables_.erase(it); + CallableState state = std::move(it->second); + callables_.erase(it); aicpu_seen_callable_ids_.erase(callable_id); if (state.host_dlopen_handle != nullptr) { @@ -912,20 +907,18 @@ int DeviceRunner::unregister_prepared_callable(int32_t callable_id) { return 0; } -bool DeviceRunner::has_prepared_callable(int32_t callable_id) const { - return prepared_callables_.count(callable_id) != 0; -} +bool DeviceRunner::has_callable(int32_t callable_id) const { return callables_.count(callable_id) != 0; } -BindPreparedCallableResult DeviceRunner::bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id) { - auto it = prepared_callables_.find(callable_id); - if (it == prepared_callables_.end()) { - LOG_ERROR("bind_prepared_callable_to_runtime: callable_id=%d not registered", callable_id); +BindCallableResult DeviceRunner::bind_callable_to_runtime(Runtime &runtime, int32_t callable_id) { + auto it = callables_.find(callable_id); + if (it == callables_.end()) { + LOG_ERROR("bind_callable_to_runtime: callable_id=%d not registered", callable_id); return {-1, nullptr, nullptr, 0}; } const auto &state = it->second; for (const auto &kv : state.kernel_addrs) { if (kv.first < 0 || kv.first >= RUNTIME_MAX_FUNC_ID) { - LOG_ERROR("bind_prepared_callable_to_runtime: func_id=%d out of range", kv.first); + LOG_ERROR("bind_callable_to_runtime: func_id=%d out of range", kv.first); return {-1, nullptr, nullptr, 0}; } runtime.replay_function_bin_addr(kv.first, kv.second); @@ -986,12 +979,12 @@ int DeviceRunner::finalize() { // each callable_id, so without this loop the host process leaks one // dlopen handle per (re)created Worker — observable in long-running // pytest sessions. - for (auto &kv : prepared_callables_) { + for (auto &kv : callables_) { if (kv.second.host_dlopen_handle != nullptr) { dlclose(kv.second.host_dlopen_handle); } } - prepared_callables_.clear(); + callables_.clear(); aicpu_seen_callable_ids_.clear(); aicpu_dlopen_total_ = 0; diff --git a/src/a5/platform/sim/host/device_runner.h b/src/a5/platform/sim/host/device_runner.h index 03ffa37804..57caec5e70 100644 --- a/src/a5/platform/sim/host/device_runner.h +++ b/src/a5/platform/sim/host/device_runner.h @@ -248,25 +248,25 @@ class DeviceRunner { * Stage a per-callable_id orchestration SO and its supporting metadata. * See a5 onboard or a2a3 device_runner.h for full contract. */ - int register_prepared_callable( + int register_callable( int32_t callable_id, const void *orch_so_data, size_t orch_so_size, const char *func_name, const char *config_name, std::vector> kernel_addrs, std::vector signature ); /** Host-orchestration sibling for hbg variants. See a2a3 onboard. */ - int register_prepared_callable_host_orch( + int register_callable_host_orch( int32_t callable_id, void *host_dlopen_handle, void *host_orch_func_ptr, std::vector> kernel_addrs, std::vector signature ); /** Drop prepared state for `callable_id`; trb refcounts SO, hbg dlcloses handle. */ - int unregister_prepared_callable(int32_t callable_id); + int unregister_callable(int32_t callable_id); /** True iff `callable_id` has prepared state staged. */ - bool has_prepared_callable(int32_t callable_id) const; + bool has_callable(int32_t callable_id) const; /** Replay prepared state onto a freshly-constructed Runtime. */ - BindPreparedCallableResult bind_prepared_callable_to_runtime(Runtime &runtime, int32_t callable_id); + BindCallableResult bind_callable_to_runtime(Runtime &runtime, int32_t callable_id); /** Monotonic AICPU dlopen counter (first-sighting only; never decremented). */ size_t aicpu_dlopen_count() const { return aicpu_dlopen_total_; } @@ -340,7 +340,7 @@ class DeviceRunner { std::unordered_map chip_callable_buffers_; // Per-callable_id prepared state. Mirrors onboard. - struct PreparedCallableState { + struct CallableState { // trb path uint64_t hash{0}; uint64_t dev_orch_so_addr{0}; @@ -359,7 +359,7 @@ class DeviceRunner { size_t capacity{0}; int refcount{0}; }; - std::unordered_map prepared_callables_; + std::unordered_map callables_; std::unordered_map orch_so_dedup_; std::unordered_set aicpu_seen_callable_ids_; size_t aicpu_dlopen_total_{0}; diff --git a/src/a5/platform/sim/host/pto_runtime_c_api.cpp b/src/a5/platform/sim/host/pto_runtime_c_api.cpp index f82ba95ea2..48a2db70c1 100644 --- a/src/a5/platform/sim/host/pto_runtime_c_api.cpp +++ b/src/a5/platform/sim/host/pto_runtime_c_api.cpp @@ -39,10 +39,8 @@ extern "C" { /* =========================================================================== * Runtime Implementation Functions (defined in runtime_maker.cpp) * =========================================================================== */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -); -int bind_prepared_to_runtime_impl( +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out); +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ); @@ -271,7 +269,7 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c pthread_setspecific(g_runner_key, ctx); try { - PreparedCallableArtifacts artifacts; + CallableArtifacts artifacts; int rc = prepare_callable_impl( reinterpret_cast(callable), upload_chip_callable_buffer_wrapper, &artifacts ); @@ -287,12 +285,12 @@ int prepare_callable(DeviceContextHandle ctx, int32_t callable_id, const void *c } if (artifacts.host_dlopen_handle != nullptr) { - rc = runner->register_prepared_callable_host_orch( + rc = runner->register_callable_host_orch( callable_id, artifacts.host_dlopen_handle, artifacts.host_orch_func_ptr, std::move(kernel_addrs), std::move(artifacts.signature) ); } else { - rc = runner->register_prepared_callable( + rc = runner->register_callable( callable_id, artifacts.orch_so_data, artifacts.orch_so_size, artifacts.func_name.c_str(), artifacts.config_name.c_str(), std::move(kernel_addrs), std::move(artifacts.signature) ); @@ -317,7 +315,7 @@ int run_prepared( if (ctx == NULL || runtime == NULL) return -1; DeviceRunner *runner = static_cast(ctx); - if (!runner->has_prepared_callable(callable_id)) { + if (!runner->has_callable(callable_id)) { LOG_ERROR("run_prepared: callable_id=%d not prepared", callable_id); return -1; } @@ -339,7 +337,7 @@ int run_prepared( r->host_api.acquire_pooled_runtime_arena = acquire_pooled_runtime_arena_wrapper; r->host_api.upload_chip_callable_buffer = upload_chip_callable_buffer_wrapper; - auto bind_result = runner->bind_prepared_callable_to_runtime(*r, callable_id); + auto bind_result = runner->bind_callable_to_runtime(*r, callable_id); int rc = bind_result.rc; if (rc != 0) { r->~Runtime(); @@ -347,7 +345,7 @@ int run_prepared( return rc; } - rc = bind_prepared_to_runtime_impl( + rc = bind_callable_to_runtime_impl( r, reinterpret_cast(args), bind_result.host_orch_func_ptr, bind_result.signature, bind_result.sig_count ); @@ -392,7 +390,7 @@ int run_prepared( int unregister_callable(DeviceContextHandle ctx, int32_t callable_id) { if (ctx == NULL) return -1; try { - return static_cast(ctx)->unregister_prepared_callable(callable_id); + return static_cast(ctx)->unregister_callable(callable_id); } catch (...) { return -1; } diff --git a/src/a5/runtime/host_build_graph/host/runtime_maker.cpp b/src/a5/runtime/host_build_graph/host/runtime_maker.cpp index ad2e099edd..3aa0e6d10e 100644 --- a/src/a5/runtime/host_build_graph/host/runtime_maker.cpp +++ b/src/a5/runtime/host_build_graph/host/runtime_maker.cpp @@ -280,14 +280,12 @@ extern "C" { * Stage the per-callable resources for the host_build_graph variant: upload * kernel binaries and dlopen the orchestration SO on the host. The dlopen * handle and resolved entry-symbol pointer are returned via - * PreparedCallableArtifacts so the platform layer can hoist them into its - * PreparedCallableState. Splitting this out of init_runtime_impl is what + * CallableArtifacts so the platform layer can hoist them into its + * CallableState. Splitting this out of init_runtime_impl is what * the hbg prepare_callable / run_prepared path rests on — the dlopen runs * once per cid instead of every run. */ -int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -) { +int prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out) { if (callable == nullptr) { LOG_ERROR("Callable pointer is null"); return -1; @@ -296,7 +294,7 @@ int prepare_callable_impl( LOG_ERROR("upload_fn or out is null"); return -1; } - *out = PreparedCallableArtifacts{}; + *out = CallableArtifacts{}; out->signature.assign(callable->signature_, callable->signature_ + callable->sig_count()); LOG_INFO_V0("Registering %d kernel(s) in prepare_callable_impl", callable->child_count()); @@ -322,7 +320,7 @@ int prepare_callable_impl( // Load orchestration SO from binary data via temp file. Held open across // the lifetime of the prepared callable; closed by - // DeviceRunner::unregister_prepared_callable. + // DeviceRunner::unregister_callable. std::string fd_path; if (!create_temp_so_file(orch_so_binary, orch_so_size, &fd_path)) { LOG_ERROR("Failed to create temp SO file"); @@ -356,10 +354,10 @@ int prepare_callable_impl( * Per-run binding for hbg: invoke the previously-resolved orchestration entry * point against the supplied args, then upload tensor info / allocation * storage. The c_api caller passes `host_orch_func_ptr` straight through from - * DeviceRunner::bind_prepared_callable_to_runtime (which read it from - * PreparedCallableState for this run's callable_id). + * DeviceRunner::bind_callable_to_runtime (which read it from + * CallableState for this run's callable_id). */ -int bind_prepared_to_runtime_impl( +int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection *signature, int sig_count ) { @@ -373,7 +371,7 @@ int bind_prepared_to_runtime_impl( } OrchestrationFunc orch_func = reinterpret_cast(host_orch_func_ptr); if (orch_func == nullptr) { - LOG_ERROR("bind_prepared_to_runtime_impl: host orch_func pointer is null"); + LOG_ERROR("bind_callable_to_runtime_impl: host orch_func pointer is null"); return -1; } diff --git a/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp b/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp index 6d0510a77a..92dd7db82b 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp @@ -96,7 +96,7 @@ static PTO2Runtime *rt{nullptr}; // that callable_id, kept warm across runs). // MAX_REGISTERED_CALLABLE_IDS is the protocol hard cap on callable_id values // (mailbox uint32 callable_id, register() returns small ints) and is shared -// with the host bounds check in DeviceRunner::register_prepared_callable — +// with the host bounds check in DeviceRunner::register_callable — // see src/common/task_interface/callable_protocol.h. struct OrchSoEntry { diff --git a/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp b/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp index 037d3ab04c..2bac4c9781 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp @@ -96,7 +96,7 @@ static int32_t read_runtime_status(Runtime *runtime, PTO2SharedMemoryHeader *hos /** * Stage the per-callable resources (kernel binaries + orchestration SO) into - * the supplied runtime so a subsequent bind_prepared_to_runtime_impl can use + * the supplied runtime so a subsequent bind_callable_to_runtime_impl can use * them. This is the cacheable half of init_runtime_impl: nothing here depends * on per-run argument values, so the prepare_callable / run_prepared split * lets us run this once per callable_id and amortize across runs. @@ -105,9 +105,8 @@ static int32_t read_runtime_status(Runtime *runtime, PTO2SharedMemoryHeader *hos * @param callable ChipCallable carrying the orch SO + child kernel binaries * @return 0 on success, -1 on failure */ -extern "C" int prepare_callable_impl( - const ChipCallable *callable, uint64_t (*upload_fn)(const void *), PreparedCallableArtifacts *out -) { +extern "C" int +prepare_callable_impl(const ChipCallable *callable, uint64_t (*upload_fn)(const void *), CallableArtifacts *out) { if (callable == nullptr) { LOG_ERROR("Callable pointer is null"); return -1; @@ -116,7 +115,7 @@ extern "C" int prepare_callable_impl( LOG_ERROR("upload_fn or out is null"); return -1; } - *out = PreparedCallableArtifacts{}; + *out = CallableArtifacts{}; out->signature.assign(callable->signature_, callable->signature_ + callable->sig_count()); LOG_INFO_V0("Registering %d kernel(s) in prepare_callable_impl", callable->child_count()); @@ -161,7 +160,7 @@ extern "C" int prepare_callable_impl( * @param orch_args Separated tensor/scalar arguments for this run * @return 0 on success, -1 on failure */ -extern "C" int bind_prepared_to_runtime_impl( +extern "C" int bind_callable_to_runtime_impl( Runtime *runtime, const ChipStorageTaskArgs *orch_args, void *host_orch_func_ptr, const ArgDirection * /*signature*/, int /*sig_count*/ ) { @@ -177,7 +176,7 @@ extern "C" int bind_prepared_to_runtime_impl( // function pointer to invoke. The c_api signature accepts one for // symmetry with hbg; assert the trb-side invariant here. if (host_orch_func_ptr != nullptr) { - LOG_ERROR("bind_prepared_to_runtime_impl: trb does not accept a host_orch_func_ptr"); + LOG_ERROR("bind_callable_to_runtime_impl: trb does not accept a host_orch_func_ptr"); return -1; } diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h b/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h index 4a690e8caa..d2c325e384 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h @@ -269,7 +269,7 @@ class Runtime { void set_orch_args(const ChipStorageTaskArgs &args); // Prebuilt-arena fast path (trb only). Set by host's - // bind_prepared_to_runtime_impl; consumed by AICPU at boot to attach a + // bind_callable_to_runtime_impl; consumed by AICPU at boot to attach a // DeviceArena to `prebuilt_arena_base_` and pick up the PTO2Runtime at // `prebuilt_arena_base_ + prebuilt_runtime_offset_`. Both stay zero on // first construction (Runtime() ctor zeros them) so a non-prebuilt boot @@ -299,7 +299,7 @@ class Runtime { /** * Replay a previously-uploaded kernel address onto a fresh Runtime * without recording it in registered_kernel_func_ids_. Used by - * DeviceRunner::bind_prepared_callable_to_runtime so prepared kernel + * DeviceRunner::bind_callable_to_runtime so prepared kernel * binaries are not freed by validate_runtime_impl across runs. */ void replay_function_bin_addr(int func_id, uint64_t addr); diff --git a/src/common/task_interface/callable_protocol.h b/src/common/task_interface/callable_protocol.h index 4e38988040..0d3107a4fc 100644 --- a/src/common/task_interface/callable_protocol.h +++ b/src/common/task_interface/callable_protocol.h @@ -16,7 +16,7 @@ * pulling in /. * * Both sides must agree on these bounds: - * - Host: DeviceRunner::register_prepared_callable rejects out-of-range ids. + * - Host: DeviceRunner::register_callable rejects out-of-range ids. * - AICPU: AicpuExecutor::run guards `orch_so_table_[callable_id]` access. */ @@ -25,7 +25,7 @@ #include // Hard cap on the number of distinct callable_ids that can be registered -// via Worker.register / DeviceRunner::register_prepared_callable. The AICPU +// via Worker.register / DeviceRunner::register_callable. The AICPU // executor reserves a fixed-size `orch_so_table_[MAX_REGISTERED_CALLABLE_IDS]` // keyed by callable_id, so this bound is part of the host↔AICPU protocol. constexpr int32_t MAX_REGISTERED_CALLABLE_IDS = 64; diff --git a/src/common/task_interface/prepare_callable_common.h b/src/common/task_interface/prepare_callable_common.h index d7c1b41cbd..d2a2ddb486 100644 --- a/src/common/task_interface/prepare_callable_common.h +++ b/src/common/task_interface/prepare_callable_common.h @@ -54,7 +54,7 @@ struct ChildKernelAddr { * resolves its entry symbol during prepare_callable_impl and stores the * resulting function pointer in host_orch_func_ptr directly. */ -struct PreparedCallableArtifacts { +struct CallableArtifacts { std::vector kernel_addrs; // Chip-level entry-tensor directions, copied from ChipCallable::signature_[]. // Scalars are also present (ArgDirection::SCALAR) and follow the tensor @@ -70,23 +70,23 @@ struct PreparedCallableArtifacts { }; /** - * Result of DeviceRunner::bind_prepared_callable_to_runtime — what the c_api - * needs to pass on to bind_prepared_to_runtime_impl for a per-run binding. + * Result of DeviceRunner::bind_callable_to_runtime — what the c_api + * needs to pass on to bind_callable_to_runtime_impl for a per-run binding. * * Returning a struct (rather than a `void**` out-parameter) keeps the caller * site idiomatic — destructure with C++17 structured bindings: * * auto [rc, host_orch_func_ptr] = - * runner->bind_prepared_callable_to_runtime(*r, callable_id); + * runner->bind_callable_to_runtime(*r, callable_id); * * `host_orch_func_ptr` is type-erased as `void *` (rather than the concrete * OrchestrationFunc) so this header stays runtime-agnostic; only the hbg path - * sets it. trb leaves it null and bind_prepared_to_runtime_impl asserts so. + * sets it. trb leaves it null and bind_callable_to_runtime_impl asserts so. */ -struct BindPreparedCallableResult { +struct BindCallableResult { int rc{0}; void *host_orch_func_ptr{nullptr}; - // Pointer into PreparedCallableState's cached signature vector — valid + // Pointer into CallableState's cached signature vector — valid // until the callable_id is unregistered. Nullptr + 0 when the callable // had no recorded signature (legacy path). const ArgDirection *signature{nullptr}; diff --git a/tests/st/a2a3/host_build_graph/prepared_callable/test_prepared_callable.py b/tests/st/a2a3/host_build_graph/prepared_callable/test_prepared_callable.py index 39cfd778cc..7b5149054f 100644 --- a/tests/st/a2a3/host_build_graph/prepared_callable/test_prepared_callable.py +++ b/tests/st/a2a3/host_build_graph/prepared_callable/test_prepared_callable.py @@ -142,7 +142,7 @@ def _run_and_validate_l2( # noqa: PLR0913 # ------------------------------------------------------------------ # host_dlopen_count assertions (hbg path). # - # hbg increments host_dlopen_count on every register_prepared_callable_host_orch + # hbg increments host_dlopen_count on every register_callable_host_orch # invocation (i.e. each `prepare_callable` call), independent of how many # times run is invoked afterwards. AICPU never dlopens the orch # SO on this variant, so aicpu_dlopen_count stays at 0. diff --git a/tests/ut/py/test_worker/test_host_worker.py b/tests/ut/py/test_worker/test_host_worker.py index 51db62f19e..47f476dee7 100644 --- a/tests/ut/py/test_worker/test_host_worker.py +++ b/tests/ut/py/test_worker/test_host_worker.py @@ -978,7 +978,7 @@ def test_unregister_middle_cid_reuses_hole(self): def test_register_overflow_raises(self): # The AICPU side reserves a fixed-size orch_so_table_[MAX_REGISTERED_CALLABLE_IDS]; # Worker.register must surface the bound at register-time, not later when - # DeviceRunner::register_prepared_callable rejects the cid. + # DeviceRunner::register_callable rejects the cid. hw = Worker(level=3, num_sub_workers=0) try: for _ in range(MAX_REGISTERED_CALLABLE_IDS):