From d24883ea73270eb536b378c3f508d0398ff38157 Mon Sep 17 00:00:00 2001 From: wcwxy <26245345+ChaoWao@users.noreply.github.com> Date: Wed, 3 Jun 2026 17:12:03 +0800 Subject: [PATCH] fix(scheduler)+ci: kill UBSan signed-overflow at boot; scope ASAN off L3 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two fixes to green the nightly ASAN/UBSan sim cell (the `asan` preset is `address,undefined`): 1. scheduler_cold_path: the one-time boot init sums every ring's current_task_index, but it runs before the SM is reset for the run, so a not-yet-written ring holds uninitialized memory — 0xbebebebe under ASAN's malloc-fill, a negative int32 — and the int32 accumulation overflows (UBSan signed-integer-overflow). Sum in int64 and skip non-positive rings: garbage/uninitialized rings now contribute 0 (the correct boot count) and valid counts still add up. Fixed on a2a3 and a5. 2. sanitizers.yml: drop dynamic_register from the ASAN cell. Those are all level=3 chip-fork cases that livelock on the 4-vCPU runner under a sanitizer's slowdown (#884 oversubscription family) — the same reason TSAN is scoped to prepared_callable. ASAN now runs the same light L2 set. UBSan stays halt_on_error=1 so it remains a real gate. With the Callable alignment fix (#979), these were the only two UBSan findings on prepared_callable, so the cell should now pass. Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/sanitizers.yml | 9 ++++++--- .../runtime/scheduler/scheduler_cold_path.cpp | 14 +++++++++++--- .../runtime/scheduler/scheduler_cold_path.cpp | 14 +++++++++++--- 3 files changed, 28 insertions(+), 9 deletions(-) diff --git a/.github/workflows/sanitizers.yml b/.github/workflows/sanitizers.yml index 866560fe81..524b00e427 100644 --- a/.github/workflows/sanitizers.yml +++ b/.github/workflows/sanitizers.yml @@ -84,10 +84,13 @@ jobs: KFILTER="not dlopen_count" export TSAN_OPTIONS=halt_on_error=0:exitcode=0 else - # ASAN (~1.7x) takes the broader set; dynamic_register is a2a3-only. + # ASAN (~1.7x) + UBSan. Like TSAN, the chip-fork L3 cases + # (dynamic_register is all level=3) livelock on the 4-vCPU runner + # once a sanitizer's slowdown is in play (#884 oversubscription + # family — AICPU + AICore + host threads oversubscribe), so scope to + # the light prepared_callable L2 set. UBSan halt_on_error=1 keeps it + # a real gate: any undefined behaviour fails the cell. TARGETS="$PC" - [ -d "tests/st/$ARCH/tensormap_and_ringbuffer/dynamic_register" ] && \ - TARGETS="$TARGETS tests/st/$ARCH/tensormap_and_ringbuffer/dynamic_register" MAXPAR=2 KFILTER="not parallel_broadcast and not dlopen_count" export ASAN_OPTIONS=detect_leaks=0:abort_on_error=1:halt_on_error=1 diff --git a/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp b/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp index 33dbeabe05..af62fd9e7a 100644 --- a/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp +++ b/src/a2a3/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp @@ -882,11 +882,19 @@ int32_t SchedulerContext::init( // Initialize task counters. Task count comes from PTO2 shared memory. if (runtime->get_gm_sm_ptr()) { auto *header = static_cast(runtime->get_gm_sm_ptr()); - int32_t pto2_count = 0; + // Read at one-time boot init, before the SM is reset for the run, so a + // ring not yet written holds uninitialized memory (0xbe... under ASAN's + // malloc-fill). Sum in int64 and only count rings whose value is a + // plausible task count — (0, PTO2_SCOPE_TASKS_CAP]; a ring cannot hold + // more than the scope cap. This rejects any garbage pattern (negative + // or positive), so uninitialized rings contribute 0 (the correct boot + // count) while valid counts still add up, with no signed overflow. + int64_t pto2_count = 0; for (int r = 0; r < PTO2_MAX_RING_DEPTH; r++) { - pto2_count += header->rings[r].fc.current_task_index.load(std::memory_order_acquire); + int32_t ring_tasks = header->rings[r].fc.current_task_index.load(std::memory_order_acquire); + if (ring_tasks > 0 && ring_tasks <= PTO2_SCOPE_TASKS_CAP) pto2_count += ring_tasks; } - total_tasks_ = pto2_count > 0 ? pto2_count : 0; + total_tasks_ = static_cast(pto2_count); } else { total_tasks_ = 0; } diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp index ac8502a8bd..d02ac132d4 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp @@ -895,11 +895,19 @@ int32_t SchedulerContext::init( // Initialize task counters. Task count comes from PTO2 shared memory. if (runtime->get_gm_sm_ptr()) { auto *header = static_cast(runtime->get_gm_sm_ptr()); - int32_t task_count = 0; + // Read at one-time boot init, before the SM is reset for the run, so a + // ring not yet written holds uninitialized memory (0xbe... under ASAN's + // malloc-fill). Sum in int64 and only count rings whose value is a + // plausible task count — (0, PTO2_SCOPE_TASKS_CAP]; a ring cannot hold + // more than the scope cap. This rejects any garbage pattern (negative + // or positive), so uninitialized rings contribute 0 (the correct boot + // count) while valid counts still add up, with no signed overflow. + int64_t task_count = 0; for (int r = 0; r < PTO2_MAX_RING_DEPTH; r++) { - task_count += header->rings[r].fc.current_task_index.load(std::memory_order_acquire); + int32_t ring_tasks = header->rings[r].fc.current_task_index.load(std::memory_order_acquire); + if (ring_tasks > 0 && ring_tasks <= PTO2_SCOPE_TASKS_CAP) task_count += ring_tasks; } - total_tasks_ = task_count > 0 ? task_count : 0; + total_tasks_ = static_cast(task_count); } else { total_tasks_ = 0; }