From 87ce4653bbea29841b188f7a6b4906d073bf9dac Mon Sep 17 00:00:00 2001 From: qweasdzxcht <37801908+qweasdzxcht@users.noreply.github.com> Date: Wed, 30 Sep 2026 12:52:10 +0800 Subject: [PATCH] Fix: complete A5 AICore retirement across TMR and HBG Retire a claimed group by broadcasting EXIT, polling all ACKs against one deadline, writing IDLE back to each acknowledged core's dispatch register, reading back that same register, and draining the closes before releasing the workers' isolated GM return gates. A silent core stays gated for host recovery. A5 has no software-defined FAST_PATH register to close, so its dispatch-register close is the available window operation. Initialize gates before opening any worker window. In TMR, arbitrate normal and emergency retirement per core so requests racing with core assignment remain pending until the owner publishes the core. In HBG, gate both resident and legacy returns, including startup and failure paths; wait for this run's AICPU EXIT before an autonomous resident ACK so a stale prior-run release cannot authorize an early return. Cover the platform ACK/close/read-back/release ordering and simulated AICore wait, plus HBG's gate reset, normal and failed legacy exit, and concurrent normal/emergency per-core ownership. Focused A5 C++ targets pass 7/7, with the HBG ownership test repeated 100 times. On the DT device, HBG selected scenes pass 3/3 plus explicit legacy 1/1; TMR selected scenes pass 2/2. Same-card TMR main/candidate/main sampling used 50 rounds per arm and case. Effective time for alternating Case1 changed by -145.8 us (-10.79%) against the main midpoint; sliding Dense16 regressed by +691.9 us (+2.76%). The Dense16 cause remains unknown and is not claimed as retirement-tail cost. The final test-only follow-up does not change the measured production code. --- .../2026-09-a5-aicore-retirement-port.md | 227 ++++++++++++ ...ntime-descriptor-input-visibility-chain.md | 44 ++- docs/investigations/README.md | 3 +- src/a5/platform/include/aicpu/platform_regs.h | 52 ++- src/a5/platform/onboard/aicore/inner_kernel.h | 17 + .../platform/shared/aicpu/platform_regs.cpp | 103 +++++- src/a5/platform/sim/aicore/inner_kernel.h | 15 + .../aicore/aicore_executor.cpp | 19 +- .../aicore/aicore_legacy_executor.cpp | 4 +- .../aicpu/aicore_lifecycle.cpp | 61 ++-- .../host_build_graph/aicpu/aicore_lifecycle.h | 4 +- .../host_build_graph/aicpu/aicpu_executor.cpp | 5 +- .../aicpu/aicpu_legacy_executor.cpp | 6 +- .../runtime/scheduler/scheduler_cold_path.cpp | 71 ++-- .../runtime/scheduler/scheduler_context.h | 6 +- .../aicore/aicore_executor.cpp | 5 +- .../aicpu/aicpu_executor.cpp | 4 +- .../docs/RUNTIME_LOGIC.md | 2 +- .../runtime/runtime.h | 29 +- .../runtime/scheduler/scheduler_cold_path.cpp | 205 ++++++++--- .../runtime/scheduler/scheduler_context.h | 29 ++ .../runtime/shared/runtime.cpp | 6 +- src/common/host_build_graph/runtime.h | 12 +- src/common/task_interface/aicore_teardown.h | 4 +- tests/ut/cpp/a5/platform/CMakeLists.txt | 22 ++ .../a5/platform/test_aicore_retirement.cpp | 342 ++++++++++++++++++ .../cpp/a5/platform/test_return_gate_wait.cpp | 236 ++++++++++++ .../runtime/host_build_graph/CMakeLists.txt | 18 + .../test_hbg_retirement_wiring.cpp | 208 +++++++++++ .../tensormap_and_ringbuffer/CMakeLists.txt | 15 + .../test_scheduler_retirement.cpp | 254 +++++++++++++ .../support/stall_dump_level_a5_stubs.cpp | 11 +- 32 files changed, 1836 insertions(+), 203 deletions(-) create mode 100644 docs/investigations/2026-09-a5-aicore-retirement-port.md create mode 100644 tests/ut/cpp/a5/platform/test_aicore_retirement.cpp create mode 100644 tests/ut/cpp/a5/platform/test_return_gate_wait.cpp create mode 100644 tests/ut/cpp/a5/runtime/host_build_graph/test_hbg_retirement_wiring.cpp create mode 100644 tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/test_scheduler_retirement.cpp diff --git a/docs/investigations/2026-09-a5-aicore-retirement-port.md b/docs/investigations/2026-09-a5-aicore-retirement-port.md new file mode 100644 index 0000000000..ce60ecbb54 --- /dev/null +++ b/docs/investigations/2026-09-a5-aicore-retirement-port.md @@ -0,0 +1,227 @@ +# 2026-09 — Porting the a2a3 AICore retirement protocol to a5: what was dropped on the way + +**Verdict: preserve a2a3's per-core claim, emergency batching and post-close +return gate.** PR #2388 (issue #2387) ports these to a5 TMR and covers the +corresponding HBG teardown paths. Separate fast-path +register control has no defined a5 software counterpart. Historical measurements +below describe earlier candidates, not the current per-core gated implementation. + +Supersedes the verdict of the entry proposed in #2300 ("dropped, no net gain"), +which was measured over a shape this one does not use. See *Where the earlier +verdict went wrong* below. + +## Setting + +a5 retired AICores one at a time: write EXIT, block for that core's ACK +against a per-core deadline, write IDLE. a2a3 retires the same hardware as a +claimed group with a read-back close. The question was which of a2a3's +mechanisms apply to a5, and what they cost. + +Measurements: a5 development card, nine graphs by three blocks, paired within +each block, medians of 20 timed calls per process, five rounds of 171 +processes. The segment is the retirement tail — `graph_build` end minus +`max(orch end, sched end)` — derived from the `[STRACE]` markers already +emitted. Parenthesised counts are blocks where the sign held. + +## Protocol + +The mechanisms are the per-core claim, `wmb` after the broadcast, a non-blocking +round-robin sweep against one shared deadline, deferred window close, the +close read-back, the `rmb` that drains it, per-core reporting of cores not +released, fatal-before-completion publication, and a GM return gate released only +after close and its drain. Normal owners retire in parallel. Emergency retirement +collects all ready cores it wins into one broadcast and one deadline. + +## Platform adaptation: fast-path window control + +Not a cost decision. a5 has no fast-path enable anywhere in its software — no +offset, no `RegId` member, no `reg_offset` case, no AICore-side case, no +caller. Writing the dispatch register to idle is what opens a window there, and +there is no separately defined FAST_PATH close operation. This establishes the +software interface, not the absence of an equivalent on silicon; copying a2a3's +register offset without an a5 definition would be unjustified. + +The earlier conclusion that this also made the GM return gate unnecessary was +too strong. The gate independently orders the worker's return after the AICPU's +last IDLE write, read-back and drain. TMR therefore uses isolated per-core gates, +reset before window-open. The worker acknowledges EXIT, then waits on its gate +before returning. Unacknowledged cores remain unreleased for host recovery. +The HBG resident and legacy paths use the same post-close gate rule. The +resident path keeps its cross-thread EXIT broadcast before each partition's +ACK sweep; the legacy path claims cores so normal and emergency retirement +cannot close or release the same worker twice. + +## Historical claim-cost experiments + +The claim is the whole cost. Taken as a2a3 takes it, per core, the port +measures **-12.35 us (0/27)**; everything else together measures +**-0.40 us (5/27)**, i.e. free — the sweep's batched broadcast pays for the +read-back. So three attempts went at the claim itself: + +| Attempt | Result | +| ------- | ------ | +| Move the flags onto each core's own cache line (`CoreExecState` is already `alignas(64)` and the retiring thread has just read `reg_addr` from that line) | **Recovers half.** -12.35 to -7.03. The remaining cost is not false sharing | +| Relax the claim to `__ATOMIC_RELAXED` (sound: the loser only skips the core, and the winner's `reg_addr` was published at handshake) | **Nothing.** -8.56, no better than `acq_rel` | +| Take one claim per owning scheduler thread instead of one per core | **Works.** 8.24 us to 0.98 us | + +The cost is thread scatter, not work, and it is proportional to how many times +the claim is taken — not to where the flags live or how they are ordered. A +probe arm with no claim at all isolates it: 8.24 us (27/27) between that arm +and the same build with a per-core claim, and the claim is the only thing that +inflates the broadcast phase, which contains no claim at all. + +Disjoint ownership proves only that a per-owner claim can prevent duplicate +retirement. It does not prove equivalent emergency progress. The per-owner +implementation called the blocking platform helper once per owner, with a fresh +deadline each time. If the emergency caller alone handles several ready groups +with silent members, later groups receive EXIT only after earlier timeouts; +the wait can approach one budget per group. a2a3 broadcasts its entire claimed +emergency set first and spends one budget on that set. The port restores that +behavior and per-core arbitration, including overlapping target subsets. + +The historical 8.24/0.98 us comparison does not price the current initialization +handoff or return gate, and is not a reason to weaken the fault protocol. + +**Do not re-propose cache-line placement or memory ordering for this claim.** + +## Dropped: two micro-optimizations of the close + +| Attempt | Result | +| ------- | ------ | +| Split the close into a store burst then a read-back burst, so the read-backs pipeline instead of each draining its own store | **No distinguishable difference** (+0.05, 7/12). The predicted ~2 us from the STR/LDR round-trip cost table did not appear; that is not the bottleneck in this segment | +| Close each window on its ACK instead of deferring every close until the group has acknowledged | **No distinguishable difference** (+0.68, 21/27, below the segment's 1 us floor). The measured candidate kept the deferred close; the current A5 port also releases each acknowledged core's return gate only after the close/read-back/drain sequence | + +## Ruled out on architecture, not measurement + +- **A `dsb` instead of the read-back.** The window is Device-nGnRE; the E is + the early write-acknowledgement, which is what the barrier waits for. +- **One read-back covering the group.** nR orders accesses within one + peripheral, minimum 4 KB granule; each core has its own window. +- **Reading back `COND` rather than `DATA_MAIN_BASE`.** Same reason — + `DATA_MAIN_BASE` (0xD0) and `COND` (0x5108) are not in the same granule, so a + COND load does not order against the close. +- **A shared deadline with a blocking per-core wait** (what #2288 proposed). + Exit waiting has two independent properties: every core judged on its own + full budget, and the group bounded by one budget. The per-core wait has the + first, a shared deadline with a blocking wait has the second, and only the + non-blocking sweep has both. + +## Where the earlier verdict went wrong + +PR #2300 proposed recording this as "dropped, no net gain". Two of its findings +hold and are carried forward: the absence of a separately defined A5 FAST_PATH +register operation, and the +diagnostic round's eliminations (`runtime_destroy` unchanged, no extra COND +reads from the sweep). The verdict does not, for three reasons. + +1. **"Only three steps have an a5 counterpart" undercounts.** That was measured + over the wait-and-close sequence alone. The claim, the read-back, its drain, + fatal-before-completion, and per-core reporting also port. The historical + candidate counted eight of ten, rather than three; the current port also + carries the independent GM return gate. +2. **The read-back was treated as tied to fast-path close and dropped with it.** + It is independent. a5's own close is a posted store to a Device-nGnRE window, + and the run publishes completion through a Normal-cacheable release store + that cannot order it. +3. **Its 59% transfer figure is specific to the per-core claim.** The earlier + per-thread candidate measured -1.43 us for the whole port; this does not + measure the current per-core claim, initialization handoff and return gate. + +## Fault injection: where it had to move to + +None of these mechanisms can be validated by a normal benchmark — their value +is entirely on the fault path, where a benchmark can only price them. The +first attempt drove the fault path from the host system tests and established +nothing: an unresponsive core forces a device reset, so the log line naming +the core never reaches the host, anything timed around the retirement is +dominated by the reset instead, and a test that counts retirement log lines +passes vacuously when no lines arrive at all. All three were withdrawn. + +What replaced them is a device-side unit suite driving +`platform_retire_aicore_group` against simulated register blocks, in the style +the a2a3 suite already uses +(`tests/ut/cpp/a5/platform/test_aicore_retirement.cpp`). +Seven cases cover: the broadcast completing before any core is waited on; no +window closing until the group has acknowledged; a silent core left unclosed +while its answering peers are released; the `released[]` contract; one shared +budget for the group rather than one per core; address validation; and the +single-target group inheriting all of it. Three more cover the return +gates: release only after close and only to acknowledged cores, a gate held shut +while a peer's ACK is pending, and an invalid target rejecting the whole batch +before any EXIT is signalled. That is ten platform cases in total. + +`tests/ut/cpp/a5/platform/test_return_gate_wait.cpp` adds three more: they include +the real a5sim `aicore/inner_kernel.h` and run the actual inline +`wait_for_post_close_release` on host threads. One waits on the gate word alone; +the other drives it against the real `platform_retire_aicore_group`, where the +worker stays held until the whole group has acknowledged and the production +retire path publishes the gate. The third covers HBG resident's autonomous +startup failure: with a stale previous-run RELEASE, it waits for the AICPU's +actual EXIT signal before ACK, then for close before return. All publish the +release before joining and +assert non-fatally, so a failed expectation cannot leave the un-timeouted wait +unjoined. + +Each of the original seven was confirmed to discriminate by mutating the source +it tests — closing on acknowledgement, closing an unacknowledged core, retiring +serially with a per-core budget, reporting every core released, dropping the +validation, and bypassing the group from the then-existing single-core entry. Every mutant fails +the case that owns its property, and the unmutated source passes repeatedly. Two +of the seven were rewritten after the first mutation round showed they passed +against a serialized retirement and against that entry closing on +the signal alone. The three gate cases were added with the gate itself and are not +covered by that mutation round. + +## Initialization and retirement ownership + +An emergency can precede a peer's barrier-free assignment. Reading that +peer's tracker as a stable partition can retire an incomplete set, consume +its claim before its cores initialize, or retire the same cores again as +orphans. Clearing trackers between runs does not synchronize this run's +initialization. + +Each core's atomic retirement state has READY and REQUESTED bits. Publishing +READY releases initialization; requesting retirement acquires it. +The operation that observes only the other bit owns retirement. A request +before READY is serviced by the initializer, without requiring it to reach +the dispatch loop or normal shutdown. Barrier-free groups use the handshake's +fixed blocked partition even when tracker assignment fails. Serial startup +publishes assigned groups after its handshake barrier, or gives all cores to +one fallback publisher if assignment cannot complete. Emergency requests every +core before waiting, without reading a peer's partially assigned tracker. A +single retired boolean is insufficient here: skipping an unpublished core loses +a late request, while marking it retired consumes its claim before it is usable. +Pending requests are serviced after publication; no claim primitive guarantees +progress if the initializer itself never reaches publication. + +`test_scheduler_retirement.cpp` in the A5 TMR unit directory drives the +production scheduler cold path with an observable retirement sink. Its twelve +cases cover late and partial initialization, concurrent publication and +retirement, normal/emergency exactly-once ownership, serial failure and success, +reset between generations, and fatal publication observed through completion. +The late and partial initialization cases fail against the original claim. It +also checks one emergency batch/deadline across owners, overlapping subsets, and +return-gate reset between generations, and +`ConcurrentOverlappingSetsClaimEachCoreOnce` has five threads retire arbitrary +overlapping, duplicated core sets from one start line and asserts every core is +retired exactly once — the per-core claim, not a per-caller one. Counts by file: +platform group 10, real sim inline wait 3, scheduler 12. Simulation checks +ordering, not physical MMIO completion. + +## Still open + +The read-back inside the window close has no observable effect on a simulated +register block; its justification remains the memory-attribute argument in +`docs/hardware/mmio-performance.md`. The scheduler tests use DFX disabled and +do not validate PMU finalization or device MMIO behavior on silicon. + +On A5 DT device 0, the gated HBG candidate passed the selected empty, +mixed-chain, vector, and explicit legacy scene tests; the selected TMR +alternating and Dense16 cases also passed golden. The six focused C++ test +targets passed. This is normal-path coverage, not a hardware fault injection. +The official TMR benchmark ran each selected case for 50 rounds in one +main/candidate/main sequence on that device. Against the mean of the two main +arms, Effective changed by -10.79% for alternating Case1 and +2.76% for +Dense16. The latter is an unresolved workload-specific regression: the Sched +window includes AICore execution and dependency waits, so its change cannot +be attributed directly to the post-dispatch retirement tail. diff --git a/docs/investigations/2026-09-runtime-descriptor-input-visibility-chain.md b/docs/investigations/2026-09-runtime-descriptor-input-visibility-chain.md index 645913189c..c070be608b 100644 --- a/docs/investigations/2026-09-runtime-descriptor-input-visibility-chain.md +++ b/docs/investigations/2026-09-runtime-descriptor-input-visibility-chain.md @@ -208,14 +208,24 @@ a5 TRB's `deinit` takes `Runtime * /*runtime*/` unnamed and performs none | ---- | ------------------------------------ | | a2a3 TRB | `dcci(my_hank, SINGLE_CACHE_LINE, CACHELINE_OUT)` (`:276`), then a **read** — bypass-load of `dev.teardown_gates[block_idx].post_close_release` (`:280`) | | a2a3 HBG | same shape: `dcci` (`:301`), then teardown-gate read (`:305`) | -| a5 TRB | `dcci(my_hank, SINGLE_CACHE_LINE, CACHELINE_OUT)` (`:255`); no teardown gate exists on this variant | +| a5 TRB | at baseline `dcci(my_hank, SINGLE_CACHE_LINE, CACHELINE_OUT)` (`:255`), with no gate member on this variant; see the post-baseline note below | | a5 HBG legacy | final `dcci(my_hank, SINGLE_CACHE_LINE, CACHELINE_OUT)` before return, commented "Flush all dirty cache lines to HBM before kernel exit" (`aicore_legacy_executor.cpp`, end of function) | | a5 HBG resident | **no final handshake `dcci`**; the function ends at `write_reg(RegId::COND, AICORE_EXITED_VALUE)` (`aicore_executor.cpp:695`) | +> **Post-baseline change (#2388).** Everything here is stated at commit +> `31da0560e`, before the a2a3 retirement port. a5 TRB has since gained a +> `teardown_gates` member and now matches the a2a3 exit shape: `dcci(..., +> CACHELINE_OUT)` (`aicore_executor.cpp:268`) then a bypass load of +> `dev.teardown_gates[block_idx].post_close_release` (`:269`). Statements below +> that say a5 TRB "has no gate member" describe the baseline, not current code. +> The final #2388 revision also gates a5 HBG legacy and resident returns; +> resident startup/exit watchdog paths wait for the AICPU's actual EXIT before +> ACK so a reused descriptor's stale RELEASE cannot bypass the close. + The a2a3 paths' last descriptor access is a *read* placed after their own -write-back, which is a different exit shape from a5 TRB's write-back-and-return -and from a5 HBG resident's no-final-flush. The resident path's exit must not be -described using the legacy path's flush. +write-back. At this audit's baseline a5 TRB instead wrote back and returned, but +post-baseline (#2388) it takes the a2a3 shape; a5 HBG resident still has no final +flush. The resident path's exit must not be described using the legacy path's flush. **What the primitive is** (`src/common/platform/onboard/aicpu/cache_ops.cpp:20`): @@ -231,11 +241,12 @@ described using the legacy path's flush. exercises any of this. The AICPU **does store** into descriptor lines — `workers[i].task` -(`scheduler_cold_path.cpp`) and, on a2a3, `teardown_gates`. Whether any given line -is *dirty* at the `dc civac` call depends on the mapping's write policy and on -prior maintenance, neither of which is established here; so the clean half of that -call has a *possible* role in publishing AICPU-authored bytes, not a demonstrated -one. Either way, the call is not reducible to "invalidate for the next DMA". +(`scheduler_cold_path.cpp`) and, on a2a3 and post-baseline a5 TRB, +`teardown_gates`. Whether any given line is *dirty* at the `dc civac` call depends +on the mapping's write policy and on prior maintenance, neither of which is +established here; so the clean half of that call has a *possible* role in +publishing AICPU-authored bytes, not a demonstrated one. Either way, the call is +not reducible to "invalidate for the next DMA". > **Withdrawn from the first version:** "the `dc civac` is performing a write-back > of AICPU-authored bytes." Observed stores do not establish a dirty write-back @@ -558,8 +569,12 @@ these: cannot succeed at all, so the question there is recovery, not publication. - **A core released but still spinning when the AICPU tears down** is the interleaving above. -- **a5** has no gate member on TRB and does not use the gates on HBG, so this whole row - is a2a3-only. +- **a5** at this audit's baseline had no gate member on TRB and does not use the + gates on HBG, so this whole row is a2a3-only. Post-baseline (#2388) a5 TRB + carries per-core `teardown_gates`, and the final revision also uses them in + a5 HBG. The same failure and partial-exit questions apply; the ordering and + publication arguments above were written for a2a3 and are not restated here + as validated on a5. **Limit of what is established about the failure path.** `retire_cores` logs each unreleased core and returns `-1` (`scheduler_cold_path.cpp:695-703` TRB); this entry @@ -603,7 +618,12 @@ in exactly this direction. It does **not** add up to "the clean has no consumer" while the clean still spans `sizeof(runtime->dev)`, so the clean covers a device-only region. **This is a post-`31da0560e` change**: at the older baseline `runtime_device_copy_size` returned `sizeof(DeviceRuntimeLaunchDesc)` and the two - extents agreed. a5 TRB has no gate member, so its upload is the whole descriptor. + extents agreed. At baseline a5 TRB had no gate member, so its upload was the whole + descriptor. **Post-baseline (#2388)** a5 TRB gains `teardown_gates`, so its + `runtime_device_initialized_prefix_size` stops at + `offsetof(..., teardown_gates)` while `runtime_device_extent_size` stays + `sizeof(DeviceRuntimeLaunchDesc)` and the same two-extent split applies; + `runtime_device_copy_size` is unchanged. ### Evidence of no reader, versus unresolved reader diff --git a/docs/investigations/README.md b/docs/investigations/README.md index 8b2f3e0b30..cf68092f6e 100644 --- a/docs/investigations/README.md +++ b/docs/investigations/README.md @@ -85,7 +85,7 @@ that ...". Newest first. - [2026-09 — Reordering the scheduler loop's graph control pass ahead of the `sync_start` drain check](2026-09-hbg-graph-release-vs-drain-reorder.md) — **dropped; superseded by releasing a Graph shell's roots on the completion path**: #2256's deadlock has an obvious fix — move the graph control block ahead of the Phase 2 drain check, a ~26-line block move that clears the hang for a bounded and small added drain latency (per retry, at most one thread pays a 4-task materialization slice; activation is once per graph, not once per retry). It was rejected because it leaves the load-bearing invariant — *every path that can free a core another task waits on must sit before the drain check* — held by nothing but the order of two blocks plus a comment, with no compile-time or test signal, and the same livelock returns as a zero-diagnostic hang the next time a control pass lands below the check. The archaeology says why it is likely rather than hypothetical: the check's position is inherited from a loop where the code after it was *only* dispatch, so "skip the rest" was exactly the drain's contract, and #1444 broke that equivalence silently while deleting the sentence that had stated it. Underneath sits an asymmetry — a shell's roots are staged through the early-dispatch publish chain like any candidate, but were released through a scheduler-loop queue, where every other release happens inline on the completion path. The shipped fix removes the asymmetry instead: `push_ready_routed`'s GRAPH branch calls `activate_graph_task` inline, so the scheduler loop stops being a release path at all. Reconsider if the resolution thread's doorbell loop becomes a measured bottleneck — and then with an explicit `sync_start_pending` guard, not a placement convention. -- [2026-09 — The Runtime descriptor's input-visibility chain: what the code expresses and what remains unestablished](2026-09-runtime-descriptor-input-visibility-chain.md) — **host↔device coherence supplied as a project premise; the remaining obligations open, nothing changed in the runtime**: #2254's field inventory noted that three onboard variants invalidate the whole device descriptor at end-of-run and a5 TRB does not, which reads as "a5 TRB is missing one". Tracing host snapshot → slot-owned destination → synchronous H2D → launch → first/last reads at `31da0560e` finds that reading unsupported, **and its opposite unsupported too**. The documented rule (`docs/hardware/cache-coherency.md:130-145`) says a5 host-DMA→AICPU is coherent and a5 must not invalidate — yet **a5 HBG calls it at two sites** and a5 TRB does not, so which one matches the rule is open. The rule's basis is recorded as a **project architectural premise from the owner — A3 host/device is not cache-coherent, A5 is** — which is neither a vendor citation nor a measurement, and which settles only the **inbound** host→AICPU edge: not ordering, not multi-field publication atomicity, not writer ownership, not the outbound `clean`, not on-device producers (AICore, SDMA), and not A2, whom the premise does not name. It authorises no code change. Two further differences are plain facts about location and extent, independent of any hardware question, and are corrected in that file and in `chip-architecture.md` by this change: the rule reads *"before reading host-written Runtime"* while all four call sites sit in `deinit()` at the **end of the previous run**, and its snippet's `sizeof(Runtime)` predates the narrowing to `sizeof(dev)`. Structurally: **there is no single "first device read"** — the two kernels go to distinct streams with **AICore submitted first** (`device_runner.cpp:664-687`), so first reads are given per processor and mode. The earliest *AICPU* read is the platform affinity gate (`kernel.cpp:93` / `:121`), which performs no cache operation and whose staleness would change the thread population rather than a value; on the AICore side only **a5 HBG** consumes host-published descriptor content early (mode word at `aicore_executor.cpp:533-535`, then the host `task` pointer and its pointee before READY), while the other paths' first AICore descriptor access is their own report *write* and their first read is the task pointer after window-open. Per-path **exit** shapes differ and must not be pooled: a2a3 TRB/HBG write back then *read* a teardown gate, a5 TRB writes back and returns, a5 HBG **legacy** has a final handshake `dcci`, and a5 HBG **resident has none** — it ends at `write_reg(COND, AICORE_EXITED_VALUE)`. `cache_invalidate_range` is **`dc civac`** — clean *and* invalidate, line-granular and rounded outward — so it has **two possible roles**, inbound freshness and outbound publication of device-authored fields. An **outbound audit** (pinned to its own later baseline, not the chain baseline above) narrows the second per field: **six** descriptor members are device-written, in **four** writer groups. Three are excluded as the clean's beneficiaries with a named dependency each — the AICore-written trio by its own `dcci(…, CACHELINE_OUT)`, the a5 HBG resident hand-off by its own `dc cvac`, and `workers[i].task` by the **`EXITED`** acknowledgment, since the AICore writes `COND = AICORE_EXITED_VALUE` only after its `task` read and the AICPU polls for `EXITED` before `deinit` — not every `COND` write qualifies, as the initial `AICORE_IDLE_VALUE` report precedes that read. **The a2a3 teardown gate is not excluded by that audit:** the AICPU stores `post_close_release` and returns with **no acknowledgment** that the AICore consumed it, so `store → AICPU clean → AICore gate read` is a permitted interleaving, and `EXITED` orders the store rather than the subsequent GM read. **A second owner premise later closed that row**: on a2a3 an AICPU ordinary GM store is observable by an AICore `ld_dev` bypass load of the same address with **no AICPU-side clean**, so the clean is not the gate's publisher and the interleaving is benign. Same attribution as the host premise — an architectural statement from the project, not a vendor citation, not proven from the public SDK, not measured — and read no wider: it does not reach the host edge, the reverse direction, or a reader taking a different access path, which is why `workers[i].task` stays open. It supplies visibility, not ordering: the gate's two edges are ordered by code, reset then `wmb()` before window-open and the CLOSE readbacks then `rmb()` before RELEASE. **Still no licence to remove the call** — the host D2H read of AICPU-written `task` on A3 is unexamined. Exceptional exits are audited separately (an un-ACKed core gets no RELEASE at all and goes to host recovery, whose behaviour this entry does not trace). On the host direction the only readers after the clean are D2H reads of `workers[]` — `print_handshake_results`, consumed only by `LOG_DEBUG` and since gated so that it copies nothing unless that threshold is enabled, and the fixture-only `simpler_probe_run_retention` — so no *functional host-side* consumer was identified, which does not add up to "no consumer" while the host D2H read of AICPU-written `task` on A3 stays unexamined. Also recorded, and specific to the audit baseline: the uploaded prefix now stops at `offsetof(teardown_gates)` while the clean still spans `sizeof(dev)`, where at the older chain baseline the two agreed. **None of this licenses removing the call** — the entry replaces its earlier "mutually exclusive candidate fixes" with a table of the individual obligations any change must discharge. Kernel mode is **scaffolding at this baseline and its device-read chain is not delivered** — `prepare_once` allocates and uploads, but `simpler_kernel_mode_supported` returns 0 and `simpler_kernel_mode_launch` unconditionally returns `INVALID_STATE` (`c_api_shared.cpp:1561`, `:1700`), so no repeated-run cache contract can be inferred from the helper. The narrower carry-forward finding: a5 HBG resident's single 64-byte `Handshake` line carries words authored by **three** tiers (host mode + context address, AICore `dcci … CACHELINE_OUT`, AICPU `dc cvac`), each publishing with a whole-line operation, and the only thing sequencing them is a marker dependency on `aicore_done` — **which the code expresses as intent but does not prove**: observing the marker is not an observation that the producer's line write-back retired, and no currently executing race is established either. Several first-version claims are explicitly withdrawn in place, including "the writers are serialized"/"keeps it correct", "the affinity gate is the earliest device read", a kernel-mode repeated-run contract, "a fresh allocation would make this moot", "the AICPU does dirty descriptor lines" (stores ≠ proven dirty write-back), CI passing as a frequency bound, and write-through as a sufficient answer to input freshness. Probe ideas are demoted to **open questions with their non-unique interpretations spelled out** — a cross-arch comparison is not a positive control until a stale readable line is independently established, a dirty-sentinel run cannot separate "cache survived" from "old line written back over the host value then read fresh", and page attributes are not classifiable by timing against an MMIO reference. Nothing built, run or measured; sim cannot exercise it either, both cache primitives being no-ops there +- [2026-09 — The Runtime descriptor's input-visibility chain: what the code expresses and what remains unestablished](2026-09-runtime-descriptor-input-visibility-chain.md) — **host↔device coherence supplied as a project premise; the remaining obligations open, nothing changed in the runtime**: #2254's field inventory noted that three onboard variants invalidate the whole device descriptor at end-of-run and a5 TRB does not, which reads as "a5 TRB is missing one". Tracing host snapshot → slot-owned destination → synchronous H2D → launch → first/last reads at `31da0560e` finds that reading unsupported, **and its opposite unsupported too**. The documented rule (`docs/hardware/cache-coherency.md:130-145`) says a5 host-DMA→AICPU is coherent and a5 must not invalidate — yet **a5 HBG calls it at two sites** and a5 TRB does not, so which one matches the rule is open. The rule's basis is recorded as a **project architectural premise from the owner — A3 host/device is not cache-coherent, A5 is** — which is neither a vendor citation nor a measurement, and which settles only the **inbound** host→AICPU edge: not ordering, not multi-field publication atomicity, not writer ownership, not the outbound `clean`, not on-device producers (AICore, SDMA), and not A2, whom the premise does not name. It authorises no code change. Two further differences are plain facts about location and extent, independent of any hardware question, and are corrected in that file and in `chip-architecture.md` by this change: the rule reads *"before reading host-written Runtime"* while all four call sites sit in `deinit()` at the **end of the previous run**, and its snippet's `sizeof(Runtime)` predates the narrowing to `sizeof(dev)`. Structurally: **there is no single "first device read"** — the two kernels go to distinct streams with **AICore submitted first** (`device_runner.cpp:664-687`), so first reads are given per processor and mode. The earliest *AICPU* read is the platform affinity gate (`kernel.cpp:93` / `:121`), which performs no cache operation and whose staleness would change the thread population rather than a value; on the AICore side only **a5 HBG** consumes host-published descriptor content early (mode word at `aicore_executor.cpp:533-535`, then the host `task` pointer and its pointee before READY), while the other paths' first AICore descriptor access is their own report *write* and their first read is the task pointer after window-open. At that audit's frozen baseline, per-path **exit** shapes differed and could not be pooled: a2a3 TRB/HBG wrote back then *read* a teardown gate, a5 TRB wrote back and returned, a5 HBG **legacy** had a final handshake `dcci`, and a5 HBG **resident had none** — it ended at `write_reg(COND, AICORE_EXITED_VALUE)`. #2388 later added a return gate to all three A5 paths. `cache_invalidate_range` is **`dc civac`** — clean *and* invalidate, line-granular and rounded outward — so it has **two possible roles**, inbound freshness and outbound publication of device-authored fields. An **outbound audit** (pinned to its own later baseline, not the chain baseline above) narrows the second per field: **six** descriptor members are device-written, in **four** writer groups. Three are excluded as the clean's beneficiaries with a named dependency each — the AICore-written trio by its own `dcci(…, CACHELINE_OUT)`, the a5 HBG resident hand-off by its own `dc cvac`, and `workers[i].task` by the **`EXITED`** acknowledgment, since the AICore writes `COND = AICORE_EXITED_VALUE` only after its `task` read and the AICPU polls for `EXITED` before `deinit` — not every `COND` write qualifies, as the initial `AICORE_IDLE_VALUE` report precedes that read. **The a2a3 teardown gate is not excluded by that audit:** the AICPU stores `post_close_release` and returns with **no acknowledgment** that the AICore consumed it, so `store → AICPU clean → AICore gate read` is a permitted interleaving, and `EXITED` orders the store rather than the subsequent GM read. **A second owner premise later closed that row**: on a2a3 an AICPU ordinary GM store is observable by an AICore `ld_dev` bypass load of the same address with **no AICPU-side clean**, so the clean is not the gate's publisher and the interleaving is benign. Same attribution as the host premise — an architectural statement from the project, not a vendor citation, not proven from the public SDK, not measured — and read no wider: it does not reach the host edge, the reverse direction, or a reader taking a different access path, which is why `workers[i].task` stays open. It supplies visibility, not ordering: the gate's two edges are ordered by code, reset then `wmb()` before window-open and the CLOSE readbacks then `rmb()` before RELEASE. **Still no licence to remove the call** — the host D2H read of AICPU-written `task` on A3 is unexamined. Exceptional exits are audited separately (an un-ACKed core gets no RELEASE at all and goes to host recovery, whose behaviour this entry does not trace). On the host direction the only readers after the clean are D2H reads of `workers[]` — `print_handshake_results`, consumed only by `LOG_DEBUG` and since gated so that it copies nothing unless that threshold is enabled, and the fixture-only `simpler_probe_run_retention` — so no *functional host-side* consumer was identified, which does not add up to "no consumer" while the host D2H read of AICPU-written `task` on A3 stays unexamined. Also recorded, and specific to the audit baseline: the uploaded prefix now stops at `offsetof(teardown_gates)` while the clean still spans `sizeof(dev)`, where at the older chain baseline the two agreed. **None of this licenses removing the call** — the entry replaces its earlier "mutually exclusive candidate fixes" with a table of the individual obligations any change must discharge. Kernel mode is **scaffolding at this baseline and its device-read chain is not delivered** — `prepare_once` allocates and uploads, but `simpler_kernel_mode_supported` returns 0 and `simpler_kernel_mode_launch` unconditionally returns `INVALID_STATE` (`c_api_shared.cpp:1561`, `:1700`), so no repeated-run cache contract can be inferred from the helper. The narrower carry-forward finding: a5 HBG resident's single 64-byte `Handshake` line carries words authored by **three** tiers (host mode + context address, AICore `dcci … CACHELINE_OUT`, AICPU `dc cvac`), each publishing with a whole-line operation, and the only thing sequencing them is a marker dependency on `aicore_done` — **which the code expresses as intent but does not prove**: observing the marker is not an observation that the producer's line write-back retired, and no currently executing race is established either. Several first-version claims are explicitly withdrawn in place, including "the writers are serialized"/"keeps it correct", "the affinity gate is the earliest device read", a kernel-mode repeated-run contract, "a fresh allocation would make this moot", "the AICPU does dirty descriptor lines" (stores ≠ proven dirty write-back), CI passing as a frequency bound, and write-through as a sufficient answer to input freshness. Probe ideas are demoted to **open questions with their non-unique interpretations spelled out** — a cross-arch comparison is not a positive control until a stale readable line is independently established, a dirty-sentinel run cannot separate "cache survived" from "old line written back over the host value then read fresh", and page attributes are not classifiable by timing against an MMIO reference. Nothing built, run or measured; sim cannot exercise it either, both cache primitives being no-ops there - [2026-09 — Narrowing the AICore FIN-ordering `dsb` to tasks that have outputs](2026-09-fin-ordering-dsb-per-task-cost.md) — **measured & dropped**: #2324's `dsb(DSB_DDR)` sits on two per-task AICore paths and is required for correctness (#2233), which invites gating it on the task having outputs. It costs **~14 ns per task (95% CI ~2–26 ns)**, so up to two barriers per task put a single `dsb` at order **7–14 ns** — cheaper than the branch that would skip it. The prior said otherwise: `2026-07-aicore-swimlane-switch-overhead-and-ack-gate.md`'s ~0.5 µs figure would have made the barrier 3% of `sliding_window_deps`, and it is off by roughly **50×** for this site, because a `dsb` costs what it has to drain and a returning kernel leaves far less outstanding than a record write-back. **The drift control is the whole result.** On A-vs-B alone two of nine cases looked real — qwen3 at +33.9 ±10.5 µs (t = 3.11) and `pa_unroll_manual_scope` C2 at +4.9 ±5.5 — and a repeat of arm A run after B accounts for both: two runs of a byte-identical `b9ba5d90` differ by **−30.4 ±11.2** on qwen3, the same magnitude in the opposite direction, and C2's same-binary delta (+5.6) *exceeds* its cross-arm one. Only `sliding_window_deps Dense16` survives (+14.0 ±12.3 across arms, −0.4 ±16.0 same-binary), and only because `sliding_window_orch.cpp:76-103` submits one AIV task per step in a strict `i → i-1` chain, putting 1001 tasks on the critical path — the other eight cases are near-chains and cannot resolve the effect at all. Two traps each produced a wrong answer first: `build_runtimes` does not track this header (#2331), so without `rm -rf build/cache/a2a3/onboard` both arms run the same binary and report no difference — that is how an earlier attempt measured a fake 18/20-vs-17/20 null on #2233 itself, and the md5 column is the positive control; and the box is shared, at 13/16 devices during A and B against 9/16 during A2, so the same-arm repeat is load-bearing rather than decorative. Carry-forward beyond the one decision: this corpus cannot resolve a per-task cost below roughly 10 ns × critical-path length, so a future per-task micro-cost question should go straight to a long-chain case with a same-binary repeat and should not expect an answer from the production workloads. a2a3 only — the identical a5 change is unmeasured @@ -95,6 +95,7 @@ Newest first. - [2026-09 — Widening the bounded WAIT-reduction bitmap past BL=64](2026-09-wait-reduction-bitmap-window-sizing.md) — **BL=64 shipped (#2009 / issue #1376), BL=128 and BL=256 deferred**, and the larger result: **BL=64 is end-to-end neutral on both production workloads** — but *amended 2026-09-07*, that is a property of those graphs, not of the mechanism: both are near-chains (per-task fanin degree ≤ 1.5 across the whole benchmark corpus, and two of its cases have **zero** redundant edges), so there is almost nothing for reduction to remove. Under a degree-16 graph (`sliding_window_deps`, 15,865 WAIT edges of which BL=64 removes 100%) the same code is **−20.47% `Orch` and −2.59% `Effective`**, and `Orch` *falls* rather than rises because `wire_fanin_task` is charged to that span — ~26 ns saved per removed edge against a per-submit bitmap cost an order of magnitude smaller. Widening only moves one of the two production workloads — Qwen3-14B decode removes 1 of 40 redundant edges at *every* window because 39 of them are cross-ring long edges outside BL=256 too, while DeepSeek-V4 FLASH decode goes 10,065 → 20,214 of 21,698 (2.0x) — against `WaitReachEntry` growing 16 → 40 B per slot (1 → 2.5 MiB) and a 1-word shift-merge becoming 4-word on the AICPU submit path for every task. But the same campaign shows the removals buy no time to begin with: 10 pinned-die rounds put every row inside the ±1.1% run-to-run band, which is the *expected* result because bounded reduction preserves WAIT reachability exactly, so no task's earliest start time moves. Instrumentation says why so little is left to make cheaper — **~90% of the 9,495 cleared edges point at producers already `CHIP_TASK_COMPLETED` at wiring**, which take the `completed_fanin` branch and never allocate a dep-pool entry, so the real saving is 990 of 19,114 entries (−5.2%) and −4.2% peak occupancy, not on the critical path of a ~30 ms AICore-bound step. Consequences worth carrying forward: the simulator's `estimated_dep_pool_entries_removed` is an **edge-count upper bound that overstates the runtime saving ~10x** (its *edge* counts are good — `drop` predicted exactly, total within 6%); and two DeepSeek-V4 aggregations produce wrong answers — keeping decode step 0 reports **−37.8%** from a single 7x launch-skew sample whose spike sits entirely in `Sched`, and the first step's *slow* rank cannot separate the arms at all (rank-sum U=5 against a critical value of 1), so only the fast rank is a clean first-step signal - [2026-09 — Sizing the host-log queue: the claim budget is not the constraint](2026-09-host-log-queue-claim-budget.md) — **measured & dropped**: `kProducerClaimAttempts = 1024` is not the knob it looks like. A producer wins its MPSC slot on the **first attempt** 57–78% of the time and the worst count across every shape tried is **86**, so the bound has ~12× headroom; a bound of 16 would have turned 787 successful writes (0.06%) at 64 threads into drops for a worst-case CPU saving of ~1.6 µs against ~100 µs that never occurs. The constraint on loss is `kQueueCapacity` — the writer's drain rate — and every loss lands in `queue_full` at 4, 16 and 64 threads. Structural, not incidental: a full queue exits on `difference < 0` **before spending any attempt**, so `queue_full` and `claim_exhausted` are nearly mutually exclusive. Records the first wrong conclusion too — a *saturating* benchmark measures the early-exit path and barely visits the claim loop, so the `claim_exhausted == 0` it reports says nothing about the attempt distribution below the bound; pacing the producers is what puts the claim path under test. Also records two properties of this implementation that item 6 does not require and nothing tests: **head-of-line blocking** (a producer preempted between claiming position *P* and publishing it parks the writer on *P*, so later published records cannot drain and *other* producers start dropping — the writer sleeping rather than spinning is tested, the queue not backing up is not) and **fairness** (the losses fall on the slowest producers, the opposite of the useful bias for diagnostics). The 45–94% drop rates quoted come from deliberately pathological unpaced workloads and are not representative; what they establish is that the queue absorbs bursts, not sustained overload +- [2026-09 — Porting the a2a3 AICore retirement protocol to a5: what was dropped on the way](2026-09-a5-aicore-retirement-port.md) — **ported (#2388): one a2a3 mechanism has no a5 counterpart, and the GM return gate is ported anyway.** a5 has no fast-path window control anywhere in its software — no offset, no `RegId`, no caller — so there is no separately defined FAST_PATH close to copy; its dispatch/window close is the only window control. The earlier reading that this also made a GM return gate pointless is withdrawn: the gate independently orders the worker's return after the AICPU's last IDLE write, read-back and drain, so the port keeps isolated per-core gates, reset before window-open and released only to acknowledged cores. The HBG resident and legacy paths now use the same post-close gate. Historical measurements below describe earlier candidates, not the current per-core gated implementation: the claim is the entire cost (per core -12.35 us against -0.40 us for everything else combined, the sweep's broadcast paying for the read-back), and it is thread scatter proportional to how often the claim is taken: private cache lines recover half, relaxed ordering recovers nothing, one claim per owning thread takes 8.24 us to 0.98 us. Records two close micro-optimizations that measured nothing (store-burst then read-back-burst; closing on ACK instead of deferring), and four alternatives ruled out on architecture (a `dsb` for the read-back, one read-back for the group, reading back `COND`, and a shared deadline with a blocking wait). Supersedes the "dropped, no net gain" verdict proposed in #2300, which undercounted the portable mechanisms at three, treated the read-back as tied to fast-path close, and measured a shape using the per-core claim. - [2026-08 — hbg: `rt_set_tensor_data`'s consumer-wait cannot observe device progress](2026-08-hbg-consumer-wait-cannot-observe-device.md) — **fixed**, found while retiring host-orchestration leftovers (#2068). `wait_for_tensor_ready(wait_for_consumers=true)` spun on the **host mirror's** `completed_watermark`, which only the device advances — in its own copy — so the comparison was `-1 < last_consumer_local_id`, unconditionally true, and the wait could only end at the 15 s `TENSOR_DATA_TIMEOUT_MS`. **No subset worked**: a producer completed inline on the host was not an exception, because the seed is the task's own id rather than `-1`, so even a producer with no consumers stalled (an earlier draft of #2068's description claimed that exception — it is wrong). The producer half was different and did have a working case, since it reads `task_state`, which `alloc_tensors` sets host-side. Left open at the time because two readings fit the code and implied opposite fixes — delete the wait as meaningless under host orchestration, or add the D2H it is missing. Resolved by the first: both halves are deleted and scalar access now rejects a tensor with a producer (`owner_task_id` or an overlapping TensorMap entry) with `INVALID_ARGS` at the call. The producer half went the same way rather than keeping its one working case, since a runtime allocation's buffer is uninitialized and has no host view either. `last_consumer_local_id` and `completed_watermark` existed only for that wait and are gone with it, which also takes `update_completed_watermark()`'s per-completion CAS prefix walk out of the scheduler's completion path - [2026-08 — The host-orchestration phase tail is page faults, not the code in the phase](2026-08-host-orch-phase-tail-is-page-faults.md) — root cause of the two shapes on every hbg host swimlane: 447 of the 449 `record_node` (now `record_sub_task`) calls above 10 µs took a minor fault, 19% of calls carry 79% of the phase, and a fault costs 14–33 µs here against 1.7 µs off-tree because the process's own `mmap`/`munmap` holds `mmap_lock` against every faulting thread — three 64 MiB unmaps in an off-tree reproducer recreate the whole distribution. The fault count is deterministic (1063/1065/1168 per two orchestrations) and drops to 29 when glibc keeps freed memory; the cost per fault varies 2.4× between runs of the same binary, which is the measurement noise that hid this for fourteen iterations. Refutes THP (`PR_SET_THP_DISABLE` leaves the count unchanged), preemption, and node shape; records why the tunables are not a fix (`args` regresses, and the probe build's −53% is the probe amplifying its own subject). Amended 2026-08-23: recording the Definition image into the retained upload staging (8 × ~126 KB per dsv4 bind, previously a vector per recording) moved `graph_upload`'s faults 38 → 1 per bind but left `host_orch`'s count unresolvable in both directions, because a freed 126 KB block is reused without re-faulting — size against glibc's mmap and trim thresholds, not byte count, decides what shows up in this tail. Amended 2026-08-25: retaining the 82 MB SM mirror on the runner (one buffer per pipeline slot instead of one per bind) removes an `mmap` + `munmap` of that size per bind — `hblkhd` stops returning to its pre-bind value on 6 of 6 binds, in both arms of two interleaved repetitions — and shows that a retained buffer must be handed over **uninitialized**: the first implementation used `std::vector::resize`, whose value-initialization faulted in all 20132 pages of the window on each rank's cold bind (~20k minflt against ~1100) and left the whole 82 MB resident. `host_orch`'s warm-bind fault count resolves in neither direction (base [1160, 1256] over eight binds, retained [181, 1268]), since the mirror is ~6 THP faults of a ~1200-fault bind; an earlier attribution of a warm-bind rise to glibc's dynamic mmap threshold is retracted there. That amendment also closes the entry's "Where a fix would go" list — items 1–3 shipped as #1981, item 4 as #1988 plus #2013, and item 3's flat-region form as #2015 — and records that none of them reached the ~1100 faults the submitting thread takes per bind, which is what is left. Amended again the same day: the claim in that amendment that pre-sizing the recorder's node and tensor storage would only help a cold bind is **retracted** — a slot-creation counter shows warm dsv4 binds still creating 1336 node slots of 1679, because #1981's retention is per thread while the pool hands bodies out through one shared FIFO. Reserving each node's own buffer to the cap makes it exactly one page and is *worse* than main (minflt 1070 → 2540); packing every body's tensors into one never-grown bump region is what helps (`record_node` warm min 1702/3423 → 1239/1563 µs). **Amended 2026-08-25 (last)**: the *count × price* framing this entry opened with is refuted by two arms pointing opposite ways — glibc keeping freed memory removes 86% of the faults and buys **no** time, while #2015 removes 6% and buys 29–43%. The count is not a lever; only the price is, and the in-tree `mmap_lock` writer that sets it is **`mprotect`**, which glibc uses to open a non-main arena (26 calls per bind in that band; those arenas only grow, `MADV_DONTNEED` there is 0). Every earlier strace here traced `madvise`/`mmap`/`munmap`/`brk` and not `mprotect`, which is why the in-tree source of the exclusion went unfound for three rounds. Consequence: the residual ~1100 resident-page re-faults per bind — survived #1981, #1988, #2013, #2015, mechanism undetermined — showed **no measurable latency change** when the tunable arm removed 86% of them, so they are **not currently established as a performance defect**; userspace tools are exhausted (`mincore` and pagemap both report presence, not writability), so pricing them at all needs `bpftrace` on `handle_mm_fault`. **Corrected 2026-08-26**: they are a **warm-up cost and they end** — at `f40cacf30`, `host_orch`'s per-bind `minflt` runs 989, 983 (cold), then 114, 173, 54, 3, 13, 13, 11, 8, and reaches 0 by the sixth bind on three independent runs. Every per-bind count in this entry came from three-to-six-round runs divided by the bind count, so each averaged two cold binds and three or four still-decaying ones under a steady-state label; the quantity being divided was never per-bind. Nothing "survived" the four changes, and the mechanism left undetermined for three rounds turned out not to need determining. Control plane at that commit: 0.529 ms min / 0.570 median over 8 warm binds (`host_orch` 0.341/0.364). The reusable half of this — that dropping the cold bind does not reach the steady state — is now a trap in [hbg-bind-phases.md](../dfx/hbg-bind-phases.md). Also carries the per-site fault table (two of its top three sites were deleted by #2019 and #2015 within days — keep the method, not the numbers) and the three tooling traps that each produced a wrong conclusion first diff --git a/src/a5/platform/include/aicpu/platform_regs.h b/src/a5/platform/include/aicpu/platform_regs.h index a4da9b939e..c91cbf538b 100644 --- a/src/a5/platform/include/aicpu/platform_regs.h +++ b/src/a5/platform/include/aicpu/platform_regs.h @@ -31,6 +31,7 @@ #include #include +#include "aicore_teardown.h" #include "aicpu/cache_maintenance.h" #include "common/platform_config.h" @@ -120,22 +121,51 @@ void write_reg(uint64_t reg_base_addr, RegId reg, uint64_t value); * Initialize AICore registers after core discovery * * This function performs platform-agnostic register initialization that works - * for both a5 and a5sim, including enabling fast path control and clearing - * dispatch registers. + * for both a5 and a5sim. Writing the dispatch register to idle is what opens + * the core's window: there is no separate window-enable control on a5. * * @param reg_addr Register base address of the AICore */ void platform_init_aicore_regs(uint64_t reg_addr); -/** - * Deinitialize AICore registers before termination - * - * This function sends exit signal and closes fast path control. - * - * @param reg_addr Register base address of the AICore - * @return 0 if the core acknowledged exit, non-zero on timeout - */ -int32_t platform_deinit_aicore_regs(uint64_t reg_addr); +// Signal one core to exit, without waiting for its acknowledgement. The store is +// posted -- the window is Device-nGnRE, see docs/hardware/mmio-performance.md -- +// so a caller signalling several cores owes one wmb() before it starts polling. +void platform_signal_aicore_exit(uint64_t reg_addr); + +// One absolute timeout deadline shared by a group of exiting cores. +uint64_t platform_aicore_exit_deadline(); + +// Quiesce a core whose COND the caller has already observed as EXITED: dispatch +// back to idle, with that posted store read back so it is complete. The readback +// targets DATA_MAIN_BASE, the register just written: nR orders accesses within +// one peripheral, and DATA_MAIN_BASE (0xD0) and COND (0x5108) do not share a +// 4 KB granule, so a COND load would not order against this store. Issues no +// fence of its own -- a caller closing several windows owes one rmb() after the +// last call and before it publishes anything those closes must precede. +void platform_close_aicore_window(uint64_t reg_addr); + +struct AicoreExitTarget { + uint64_t reg_addr; + AicoreTeardownControl *teardown; +}; + +// Retires one exclusively-claimed set of cores: signal every member, collect +// every ACK against one shared deadline, close every acknowledged window, then +// release those workers. +// Callers claim their targets first, so concurrent callers never name the same +// core and the set need not be the whole chip. An unacknowledged core is neither +// closed nor released and stays the host recovery path's responsibility. +// +// `released`, when non-null, receives one flag per target and lets the caller +// name the cores it failed to retire; this layer takes no logging dependency. +// Every path that returns fills those entries first, rejection included, so the +// caller may read them without initializing the buffer; a `count` above +// PLATFORM_MAX_CORES is rejected and only that many are filled. +// Returns 0 when every target was released, -1 on timeout or invalid targets. +int32_t platform_retire_aicore_group( + const AicoreExitTarget *targets, size_t count, uint64_t deadline, bool *released = nullptr +); /** * Variant-specific AICore deinit wait timeout, in ticks of get_sys_cnt_aicpu. diff --git a/src/a5/platform/onboard/aicore/inner_kernel.h b/src/a5/platform/onboard/aicore/inner_kernel.h index f003c38fc3..77486b826b 100644 --- a/src/a5/platform/onboard/aicore/inner_kernel.h +++ b/src/a5/platform/onboard/aicore/inner_kernel.h @@ -21,6 +21,7 @@ #define PLATFORM_A5_AICORE_INNER_KERNEL_H_ #include +#include "aicore_teardown.h" #include "common/platform_config.h" @@ -48,6 +49,14 @@ // OUT_OF_ORDER_FULL_BARRIER - no-op on real hardware (dcci handles full cache coherency) #define OUT_OF_ORDER_FULL_BARRIER() ((void)0) +// EXITED acknowledges quiescence; only the AICPU's post-close gate permits return. +__aicore__ inline void wait_for_post_close_release(__gm__ uint32_t *release) { + while (static_cast(ld_dev(release, 0)) != AICORE_POST_CLOSE_RELEASE) { + SPIN_WAIT_HINT(); + } + dsb(DSB_DDR); +} + /** * Read an AICore register via SPR access * @@ -72,6 +81,14 @@ __aicore__ inline uint64_t read_reg(RegId reg) { } } +// Resident startup can fail before AICPU has reset this run's return gate. +// Only its DMB EXIT, published after that reset, permits the EXITED ACK. +__aicore__ inline void wait_for_aicpu_exit_signal() { + while (static_cast(read_reg(RegId::DATA_MAIN_BASE)) != AICORE_EXIT_SIGNAL) { + SPIN_WAIT_HINT(); + } +} + /** * Read the high 32 bits of DATA_MAIN_BASE. * diff --git a/src/a5/platform/shared/aicpu/platform_regs.cpp b/src/a5/platform/shared/aicpu/platform_regs.cpp index 6a74d2a21c..99e2948984 100644 --- a/src/a5/platform/shared/aicpu/platform_regs.cpp +++ b/src/a5/platform/shared/aicpu/platform_regs.cpp @@ -21,8 +21,8 @@ #include #include "aicpu/device_time.h" #include "aicpu/platform_regs.h" +#include "common/memory_barrier.h" #include "common/platform_config.h" -#include "common/unified_log.h" #include "spin_hint.h" static uint64_t g_platform_regs = 0; @@ -36,26 +36,93 @@ void platform_init_aicore_regs(uint64_t reg_addr) { write_reg(reg_addr, RegId::DATA_MAIN_BASE, AICPU_IDLE_TASK_ID); } -int32_t platform_deinit_aicore_regs(uint64_t reg_addr) { - // Send exit signal to AICore - write_reg(reg_addr, RegId::DATA_MAIN_BASE, AICORE_EXIT_SIGNAL); - - // Wait for AICore to acknowledge exit by writing AICORE_EXITED_VALUE to COND. - // Timeout is variant-specific (sim wider than onboard) — see - // inner_get_deinit_timeout_ticks declaration in platform_regs.h. - const uint64_t deinit_timeout_ticks = inner_get_deinit_timeout_ticks(); - uint64_t t0 = get_sys_cnt_aicpu(); - while (read_reg(reg_addr, RegId::COND) != AICORE_EXITED_VALUE) { - if (get_sys_cnt_aicpu() - t0 > deinit_timeout_ticks) { - LOG_ERROR("Timed out waiting for AICore exit ack at reg_addr=0x%lx", static_cast(reg_addr)); - return -1; - } - SPIN_WAIT_HINT(); - } +void platform_signal_aicore_exit(uint64_t reg_addr) { write_reg(reg_addr, RegId::DATA_MAIN_BASE, AICORE_EXIT_SIGNAL); } + +// Timeout is variant-specific (sim wider than onboard) — see +// inner_get_deinit_timeout_ticks declaration in platform_regs.h. +uint64_t platform_aicore_exit_deadline() { return get_sys_cnt_aicpu() + inner_get_deinit_timeout_ticks(); } +void platform_close_aicore_window(uint64_t reg_addr) { // Initialize task dispatch register to idle state write_reg(reg_addr, RegId::DATA_MAIN_BASE, AICPU_IDLE_TASK_ID); - return 0; + // Complete the posted MMIO close. The store retires into the bus's + // outstanding queue on its early write-ack and is not a device-write + // completion fence on its own; a load to the same register cannot pass it, + // so this readback is what drains it. The drain that pairs with this read is + // the caller's, so several windows share one. + (void)read_reg(reg_addr, RegId::DATA_MAIN_BASE); +} + +int32_t platform_retire_aicore_group(const AicoreExitTarget *targets, size_t count, uint64_t deadline, bool *released) { + // `released` is filled before anything can return, rejection included, so a + // caller may read it without initializing the buffer. An over-large `count` + // says nothing about how big that buffer is, so the fill stops at the one + // size the contract guarantees. + if (released != nullptr) { + const size_t reportable = count < PLATFORM_MAX_CORES ? count : PLATFORM_MAX_CORES; + for (size_t i = 0; i < reportable; ++i) + released[i] = false; + } + if (count > PLATFORM_MAX_CORES || (count != 0 && targets == nullptr)) return -1; + for (size_t i = 0; i < count; ++i) { + if (targets[i].reg_addr == 0 || targets[i].teardown == nullptr) return -1; + } + + // Broadcast to the whole group before waiting on any member, so the cores + // drain concurrently and a dead core's wait does not serialize behind the + // cores ahead of it. + for (size_t i = 0; i < count; ++i) { + platform_signal_aicore_exit(targets[i].reg_addr); + } + wmb(); + + // Round-robin rather than blocking on one core at a time. Blocking spends + // the shared deadline on whichever core happens to come first, and every + // core behind it is then judged on a peer's timeout instead of its own. + // Sweeping non-blockingly gives each core the whole budget: a core is only + // abandoned once the deadline passes with it still silent. + // The sweep below reads an entry before it writes it, so the first `count` + // must start false. Entries past `count` are never read. + bool acknowledged[PLATFORM_MAX_CORES]; + for (size_t i = 0; i < count; ++i) + acknowledged[i] = false; + size_t remaining = count; + while (remaining != 0) { + for (size_t i = 0; i < count; ++i) { + if (acknowledged[i]) continue; + if (read_reg(targets[i].reg_addr, RegId::COND) == AICORE_EXITED_VALUE) { + acknowledged[i] = true; + --remaining; + } + } + if (remaining == 0 || get_sys_cnt_aicpu() > deadline) break; + } + + // No window closes until every ACK above is in, so a core is never quiesced + // while a peer is still being waited on. COND is not re-read here: the + // passes above already established it. + int32_t rc = 0; + for (size_t i = 0; i < count; ++i) { + if (acknowledged[i]) { + platform_close_aicore_window(targets[i].reg_addr); + } else { + rc = -1; + } + } + // One drain covers every readback the close pass issued, and it is what + // orders every store below after the close it belongs to: a dsb blocks every + // later instruction until it completes. + rmb(); + for (size_t i = 0; i < count; ++i) { + if (acknowledged[i]) { + __atomic_store_n(&targets[i].teardown->post_close_release, AICORE_POST_CLOSE_RELEASE, __ATOMIC_RELAXED); + } + } + if (released != nullptr) { + for (size_t i = 0; i < count; ++i) + released[i] = acknowledged[i]; + } + return rc; } uint32_t platform_get_physical_cores_count() { diff --git a/src/a5/platform/sim/aicore/inner_kernel.h b/src/a5/platform/sim/aicore/inner_kernel.h index e6b9b510fc..6d4bd57d11 100644 --- a/src/a5/platform/sim/aicore/inner_kernel.h +++ b/src/a5/platform/sim/aicore/inner_kernel.h @@ -22,6 +22,7 @@ #include #include +#include "aicore_teardown.h" #include #include "aicpu/device_time.h" @@ -113,6 +114,12 @@ typedef int mem_dsb_t; // Equivalent to dmb ish (aarch64) / mfence (x86). #define OUT_OF_ORDER_FULL_BARRIER() __sync_synchronize() +inline void wait_for_post_close_release(uint32_t *release) { + while (__atomic_load_n(release, __ATOMIC_ACQUIRE) != AICORE_POST_CLOSE_RELEASE) { + SPIN_WAIT_HINT(); + } +} + // ============================================================================= // MMIO Load/Store Intrinsics (sim stubs) // ============================================================================= @@ -177,6 +184,14 @@ inline uint64_t read_reg(RegId reg) { return static_cast(__atomic_load_n(ptr, __ATOMIC_ACQUIRE)); } +// Resident startup can fail before AICPU has reset this run's return gate. +// Only its DMB EXIT, published after that reset, permits the EXITED ACK. +inline void wait_for_aicpu_exit_signal() { + while (static_cast(read_reg(RegId::DATA_MAIN_BASE)) != AICORE_EXIT_SIGNAL) { + SPIN_WAIT_HINT(); + } +} + /** * Read the high 32 bits of DATA_MAIN_BASE (early-dispatch doorbell). * The high word lives one 32-bit slot above the dispatch token. diff --git a/src/a5/runtime/host_build_graph/aicore/aicore_executor.cpp b/src/a5/runtime/host_build_graph/aicore/aicore_executor.cpp index c3959d1d0e..2b7ecf57aa 100644 --- a/src/a5/runtime/host_build_graph/aicore/aicore_executor.cpp +++ b/src/a5/runtime/host_build_graph/aicore/aicore_executor.cpp @@ -653,10 +653,13 @@ __aicore__ __attribute__((noinline)) void run_resident_scheduler( context->exit_ack_publish_cycles = stats.exit_ack_publish_cycles; scheduler_publish_cache_line(&context->completion_enqueue_cycles); } - // `platform_deinit_aicore_regs` waits for this acknowledgement, and the run's - // finalizer folds this core's `scheduler_error` once it is observed. Complete - // the preceding GM writes first. Stays outside the profiling block above: the - // tail is optional, this is not. + // A local register/exit watchdog may have ended the loop without AICPU + // signalling EXIT. Its current-run signal is published after the return + // gate reset, so wait for it before making the acknowledgement visible. + // The finalizer also folds this core's `scheduler_error` once observed. + wait_for_aicpu_exit_signal(); + // Complete preceding GM writes before ACK. Stays outside profiling: the + // tail is optional, this ordering is not. OUT_OF_ORDER_STORE_BARRIER(); write_reg(RegId::COND, AICORE_EXITED_VALUE); } @@ -676,6 +679,7 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in if (runtime_mode != SCHEDULER_RUNTIME_MODE_RESIDENT_PENDING && runtime_mode != SCHEDULER_RUNTIME_MODE_RESIDENT_READY) { legacy_aicore_execute(runtime, block_idx, core_type); + wait_for_post_close_release(&runtime->dev.teardown_gates[block_idx].post_close_release); return; } const bool chip_swimlane_enabled = SIMPLER_GET_DFX_FLAG(profiling_flag, SIMPLER_DFX_FLAG_CHIP_SWIMLANE); @@ -750,11 +754,17 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in SPIN_WAIT_HINT(); } if (startup_signal == AICORE_EXIT_SIGNAL) { + // A local watchdog or scheduler error can request exit before AICPU + // has reset this run's return gate. Wait for its actual DMB EXIT, + // which AICPU publishes only after that reset. Do this before ACK: + // after ACK the AICPU may close the window back to IDLE. + wait_for_aicpu_exit_signal(); // The AICPU reads this acknowledgement as the point this core's writes // have landed, and the timeout above publishes its error through // `pending_run_control`. Complete the preceding GM writes first. OUT_OF_ORDER_STORE_BARRIER(); write_reg(RegId::COND, AICORE_EXITED_VALUE); + wait_for_post_close_release(&runtime->dev.teardown_gates[block_idx].post_close_release); return; } @@ -780,4 +790,5 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in context, run_control, scheduler_state_base, profiling_level, aicore_entry_cycles, handshake_publish_cycles ); } + wait_for_post_close_release(&runtime->dev.teardown_gates[block_idx].post_close_release); } diff --git a/src/a5/runtime/host_build_graph/aicore/aicore_legacy_executor.cpp b/src/a5/runtime/host_build_graph/aicore/aicore_legacy_executor.cpp index d7a5c49cfa..951b7d6002 100644 --- a/src/a5/runtime/host_build_graph/aicore/aicore_legacy_executor.cpp +++ b/src/a5/runtime/host_build_graph/aicore/aicore_legacy_executor.cpp @@ -98,7 +98,7 @@ legacy_aicore_execute(__gm__ Runtime *runtime, int block_idx, CoreType core_type // Phase 2: Wait for the AICPU to open our register window. A kernel launch // resets DATA_MAIN_BASE to 0 (verified on a2a3 silicon); the AICPU writes - // DATA_MAIN_BASE = AICPU_IDLE_TASK_ID (non-zero) as it opens FAST_PATH, so a + // DATA_MAIN_BASE = AICPU_IDLE_TASK_ID (non-zero) to open the window, so a // non-zero read means the window is open and reads/writes are valid. The // AICPU runs assign_cores_to_threads (µs) between opening the window and the // first dispatch, so this IDLE is observed long before any task_id lands — @@ -108,7 +108,7 @@ legacy_aicore_execute(__gm__ Runtime *runtime, int block_idx, CoreType core_type while (read_reg(RegId::DATA_MAIN_BASE) == 0) { SPIN_WAIT_HINT(); } - // Report initial idle status via register (FAST_PATH is now open). + // Report initial idle status via register (the window is now open). write_reg(RegId::COND, AICORE_IDLE_VALUE); // The AICPU writes task after observing our report (so our CACHELINE_OUT flush diff --git a/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.cpp b/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.cpp index 6194242fa2..8cf1ac40f7 100644 --- a/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.cpp +++ b/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.cpp @@ -68,6 +68,10 @@ int32_t AicoreLifecycle::pre_handshake_init(Runtime *runtime, int32_t aicpu_thre std::memset(physical_core_ids_, 0, sizeof(physical_core_ids_)); std::memset(thread_handshake_timing_, 0, sizeof(thread_handshake_timing_)); core_count_ = runtime->dev.worker_count; + // The descriptor's gate tail is device-owned and is not copied from the host. + // Clear prior-run releases before any partition can open a register window. + std::memset(runtime->get_teardown_gates(), 0, sizeof(AicoreTeardownControl) * core_count_); + wmb(); aicpu_thread_num_ = aicpu_thread_num; regs_base_ = regs_base; lifecycle_traces_ = nullptr; @@ -355,7 +359,7 @@ int32_t AicoreLifecycle::wait_bootstrap_complete(Runtime *runtime) { return 0; } -int32_t AicoreLifecycle::release_partition(int32_t thread_idx, bool start_execution) { +int32_t AicoreLifecycle::release_partition(Runtime *runtime, int32_t thread_idx, bool start_execution) { const int32_t lo = static_cast((static_cast(thread_idx) * core_count_) / aicpu_thread_num_); const int32_t hi = static_cast((static_cast(thread_idx + 1) * core_count_) / aicpu_thread_num_); int32_t rc = 0; @@ -363,13 +367,19 @@ int32_t AicoreLifecycle::release_partition(int32_t thread_idx, bool start_execut if (start_execution && trace != nullptr && lifecycle_timing_enabled()) trace->register_release_start_cycles = get_sys_cnt_aicpu(); wmb(); - for (int32_t i = lo; i < hi; ++i) { - if (cores_[i].reg_addr == 0) continue; - if (start_execution) { - platform_init_aicore_regs(cores_[i].reg_addr); - } else { - if (platform_deinit_aicore_regs(cores_[i].reg_addr) != 0) rc = -1; + if (start_execution) { + for (int32_t i = lo; i < hi; ++i) { + if (cores_[i].reg_addr != 0) platform_init_aicore_regs(cores_[i].reg_addr); + } + } else { + AicoreExitTarget targets[kMaxWorkers]; + size_t count = 0; + for (int32_t i = lo; i < hi; ++i) { + if (cores_[i].reg_addr == 0) continue; + targets[count] = {cores_[i].reg_addr, &runtime->get_teardown_gates()[i]}; + ++count; } + if (count != 0 && platform_retire_aicore_group(targets, count, platform_aicore_exit_deadline()) != 0) rc = -1; } if (start_execution && trace != nullptr && lifecycle_timing_enabled()) trace->register_release_end_cycles = get_sys_cnt_aicpu(); @@ -383,38 +393,45 @@ void AicoreLifecycle::signal_shutdown_partition(int32_t thread_idx) { if (trace != nullptr && lifecycle_timing_enabled()) trace->exit_signal_start_cycles = get_sys_cnt_aicpu(); for (int32_t i = lo; i < hi; ++i) { if (cores_[i].reg_addr == 0) continue; - write_reg(cores_[i].reg_addr, RegId::DATA_MAIN_BASE, AICORE_EXIT_SIGNAL); + platform_signal_aicore_exit(cores_[i].reg_addr); } + wmb(); if (trace != nullptr && lifecycle_timing_enabled()) trace->exit_signal_end_cycles = get_sys_cnt_aicpu(); } int32_t AicoreLifecycle::finish_shutdown_partition(int32_t thread_idx, Runtime *runtime) { - (void)runtime; const int32_t lo = static_cast((static_cast(thread_idx) * core_count_) / aicpu_thread_num_); const int32_t hi = static_cast((static_cast(thread_idx + 1) * core_count_) / aicpu_thread_num_); - int32_t rc = 0; - AicpuThreadLifecycleTrace *trace = thread_lifecycle_trace(thread_idx); - if (trace != nullptr && lifecycle_timing_enabled()) trace->exit_wait_start_cycles = get_sys_cnt_aicpu(); + AicoreExitTarget targets[kMaxWorkers]; + int32_t core_ids[kMaxWorkers]; + size_t count = 0; for (int32_t i = lo; i < hi; ++i) { if (cores_[i].reg_addr == 0) continue; - if (platform_deinit_aicore_regs(cores_[i].reg_addr) != 0) { - rc = -1; + targets[count] = {cores_[i].reg_addr, &runtime->get_teardown_gates()[i]}; + core_ids[count] = i; + ++count; + } + AicpuThreadLifecycleTrace *trace = thread_lifecycle_trace(thread_idx); + if (trace != nullptr && lifecycle_timing_enabled()) trace->exit_wait_start_cycles = get_sys_cnt_aicpu(); + bool released[kMaxWorkers]; + const int32_t rc = + count == 0 ? 0 : platform_retire_aicore_group(targets, count, platform_aicore_exit_deadline(), released); + if (rc != 0) { + for (size_t i = 0; i < count; ++i) { + if (!released[i]) LOG_ERROR("AICore retirement: core %d not released", core_ids[i]); } } - rmb(); if (trace != nullptr && lifecycle_timing_enabled()) { trace->exit_wait_end_cycles = get_sys_cnt_aicpu(); cache_flush_range(trace, sizeof(*trace)); } - int32_t core_ids[kMaxWorkers]{}; - int32_t count = 0; - + int32_t partition_core_ids[kMaxWorkers]; + int32_t partition_count = 0; for (int32_t i = lo; i < hi; ++i) - core_ids[count++] = i; - - if (is_chip_swimlane_enabled()) chip_swimlane_aicpu_flush(thread_idx, core_ids, count); - if (is_pmu_enabled()) pmu_aicpu_finalize(core_ids, count); + partition_core_ids[partition_count++] = i; + if (is_chip_swimlane_enabled()) chip_swimlane_aicpu_flush(thread_idx, partition_core_ids, partition_count); + if (is_pmu_enabled()) pmu_aicpu_finalize(partition_core_ids, partition_count); return rc; } diff --git a/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.h b/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.h index 72ac47d69f..9e8af2be62 100644 --- a/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.h +++ b/src/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.h @@ -19,6 +19,7 @@ class Runtime; struct AicpuThreadLifecycleTrace; +class AicoreLifecycleTestPeer; class AicoreLifecycle { public: @@ -29,12 +30,13 @@ class AicoreLifecycle { void begin_bootstrap_wait(int32_t thread_idx); void end_bootstrap_wait(int32_t thread_idx); int32_t wait_bootstrap_complete(Runtime *runtime); - int32_t release_partition(int32_t thread_idx, bool start_execution); + int32_t release_partition(Runtime *runtime, int32_t thread_idx, bool start_execution); void signal_shutdown_partition(int32_t thread_idx); int32_t finish_shutdown_partition(int32_t thread_idx, Runtime *runtime); void deinit(); private: + friend class AicoreLifecycleTestPeer; static constexpr int32_t kMaxWorkers = 108; struct CoreState { diff --git a/src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp b/src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp index d023a70068..95a5132914 100644 --- a/src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp +++ b/src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp @@ -321,7 +321,7 @@ int32_t AicpuExecutor::init(Runtime *runtime) { // The DMB register is the only execution gate. On failure the same // partitioned path sends EXIT and waits for every ACK. - if (aicore_lifecycle_.release_partition(tidx, !init_failed_.load(std::memory_order_acquire)) != 0) { + if (aicore_lifecycle_.release_partition(runtime, tidx, !init_failed_.load(std::memory_order_acquire)) != 0) { init_failed_.store(true, std::memory_order_release); } @@ -578,8 +578,7 @@ void AicpuExecutor::deinit(Runtime *runtime) { // The length is the device descriptor, not sizeof(Runtime): the host-only // tail past it is never on the device at all. It deliberately spans more // than the uploaded prefix, keeping maintenance over the whole allocated - // descriptor — including the gate array, which A5 declares but no A5 code - // reads or writes. + // descriptor — including the device-owned gate array. cache_invalidate_range(runtime, sizeof(runtime->dev)); aicore_lifecycle_.deinit(); diff --git a/src/a5/runtime/host_build_graph/aicpu/aicpu_legacy_executor.cpp b/src/a5/runtime/host_build_graph/aicpu/aicpu_legacy_executor.cpp index a4917a7415..a39a987887 100644 --- a/src/a5/runtime/host_build_graph/aicpu/aicpu_legacy_executor.cpp +++ b/src/a5/runtime/host_build_graph/aicpu/aicpu_legacy_executor.cpp @@ -403,8 +403,7 @@ int32_t LegacyAicpuExecutor::run(Runtime *runtime) { } // Always shutdown AICore — even if sched_ctx_.completed_ was already true. - // platform_deinit_aicore_regs is idempotent. - int32_t shutdown_rc = sched_ctx_.shutdown(thread_idx); + int32_t shutdown_rc = sched_ctx_.shutdown(runtime, thread_idx); // Both outcomes reach the terminal record before this thread's arrival, and // with the state that produced each: folding shutdown_rc into run_rc first // would publish a teardown failure as an execution one. @@ -471,8 +470,7 @@ void LegacyAicpuExecutor::deinit(Runtime *runtime) { // The length is the device descriptor, not sizeof(Runtime): the host-only // tail past it is never on the device at all. It deliberately spans more // than the uploaded prefix, keeping maintenance over the whole allocated - // descriptor — including the gate array, which A5 declares but no A5 code - // reads or writes. + // descriptor — including the device-owned gate array. cache_invalidate_range(runtime, sizeof(runtime->dev)); // Reset all SchedulerContext-owned state in one place. diff --git a/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_cold_path.cpp b/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_cold_path.cpp index 1ad1224ff1..05555f1719 100644 --- a/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_cold_path.cpp +++ b/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_cold_path.cpp @@ -546,11 +546,10 @@ void SchedulerContext::log_chip_swimlane_summary(int32_t thread_idx, int32_t cur #endif // ============================================================================= -// Shutdown: deinit AICore regs for this thread's cores (and PMU finalize if enabled). +// Shutdown: retire this thread's cores (and PMU finalize if enabled). // Orchestrator threads have core_trackers_[thread_idx].core_num() == 0 -> no-op. -// platform_deinit_aicore_regs is idempotent; safe to call after early completion. // ============================================================================= -int32_t SchedulerContext::shutdown(int32_t thread_idx) { +int32_t SchedulerContext::shutdown(Runtime *runtime, int32_t thread_idx) { const int32_t *cores = core_trackers_[thread_idx].core_ids(); int32_t core_num = core_trackers_[thread_idx].core_num(); if (core_num == 0) return 0; @@ -562,18 +561,30 @@ int32_t SchedulerContext::shutdown(int32_t thread_idx) { #endif LOG_INFO("Thread %d: Shutting down %d cores", thread_idx, core_num); - int32_t rc = 0; - for (int32_t i = 0; i < core_num; i++) { - int32_t core_id = cores[i]; - uint64_t reg_addr = core_exec_states_[core_id].reg_addr; - if (reg_addr != 0) { - // Timeout means AICore is unresponsive. Log and continue deiniting remaining cores. - if (platform_deinit_aicore_regs(reg_addr) != 0) { - LOG_ERROR("Thread %d: Core %d deinit timed out", thread_idx, core_id); - rc = -1; - } - } else { - LOG_ERROR("Thread %d: Core %d has invalid register address", thread_idx, core_id); + return retire_cores(runtime, cores, core_num); +} + +int32_t SchedulerContext::retire_cores(Runtime *runtime, const int32_t *core_ids, int32_t core_num) { + AicoreExitTarget targets[PLATFORM_MAX_CORES]; + int32_t claimed_ids[PLATFORM_MAX_CORES]; + size_t count = 0; + for (int32_t i = 0; i < core_num; ++i) { + const int32_t core_id = core_ids[i]; + if (core_id < 0 || core_id >= cores_total_num_) continue; + const uint64_t reg_addr = core_exec_states_[core_id].reg_addr; + if (reg_addr == 0) continue; + if (core_retired_[core_id].exchange(true, std::memory_order_acq_rel)) continue; + targets[count] = {reg_addr, &runtime->get_teardown_gates()[core_id]}; + claimed_ids[count] = core_id; + ++count; + } + if (count == 0) return 0; + + bool released[PLATFORM_MAX_CORES]; + const int32_t rc = platform_retire_aicore_group(targets, count, platform_aicore_exit_deadline(), released); + if (rc != 0) { + for (size_t i = 0; i < count; ++i) { + if (!released[i]) LOG_ERROR("AICore retirement: core %d not released", claimed_ids[i]); } } return rc; @@ -621,7 +632,7 @@ void SchedulerContext::handshake_partition(Runtime *runtime, int32_t tidx, int32 // way RegId::COND polling is. // // Servicing a core = validate its physical_core_id, then open its register - // window (platform_init_aicore_regs: FAST_PATH + DATA_MAIN_BASE=IDLE). That + // window (platform_init_aicore_regs: DATA_MAIN_BASE=IDLE). That // IDLE write is *also* the signal the core polls for to leave its // post-report wait — so opening the window IS the acknowledgement. There is // no separate aicpu_regs_ready ack and no second round-trip. AIC/AIV @@ -793,27 +804,15 @@ bool SchedulerContext::assign_cores_to_threads() { } // ============================================================================= -// Emergency shutdown: broadcast exit signal to every handshake'd core and -// deinit their AICore register blocks. Idempotent. +// Emergency shutdown claims every initialized core not already owned by a +// normal retirement, then retires all winners against one deadline. // ============================================================================= void SchedulerContext::emergency_shutdown(Runtime *runtime) { - (void)runtime; // exit is now delivered via each core's register block, not GM LOG_WARN("Emergency shutdown: sending exit signal to all initialized cores"); - int32_t timeout_count = 0; - for (int32_t i = 0; i < cores_total_num_; i++) { - // platform_deinit_aicore_regs writes DATA_MAIN_BASE=EXIT, which both - // releases a core still polling for its window to open and signals it to - // exit. Cores never opened (reg_addr==0) are reaped by the host device - // reset that follows a handshake failure. - if (core_exec_states_[i].reg_addr != 0) { - if (platform_deinit_aicore_regs(core_exec_states_[i].reg_addr) != 0) { - timeout_count++; - } - } - } - if (timeout_count > 0) { - LOG_ERROR("Emergency shutdown: %d cores did not acknowledge exit", timeout_count); - } + int32_t all[PLATFORM_MAX_CORES]; + for (int32_t i = 0; i < cores_total_num_; ++i) + all[i] = i; + (void)retire_cores(runtime, all, cores_total_num_); } // ============================================================================= @@ -866,6 +865,10 @@ int32_t SchedulerContext::pre_handshake_init(Runtime *runtime, int32_t aicpu_thr LOG_ERROR("Invalid cores_total_num %d (expected 1-%d)", cores_total_num_, RUNTIME_MAX_WORKER); return -1; } + memset(runtime->get_teardown_gates(), 0, sizeof(AicoreTeardownControl) * cores_total_num_); + for (int32_t i = 0; i < cores_total_num_; ++i) + core_retired_[i].store(false, std::memory_order_relaxed); + wmb(); aic_count_ = 0; aiv_count_ = 0; handshake_failed_.store(false, std::memory_order_release); diff --git a/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_context.h b/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_context.h index 9c440ea9b2..685e34d15d 100644 --- a/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_context.h +++ b/src/a5/runtime/host_build_graph/runtime/scheduler/scheduler_context.h @@ -153,7 +153,7 @@ class SchedulerContext { // Shutdown AICore registers for this thread's assigned cores. // Also runs PMU finalize (SIMPLER_DFX) before deinit when enabled. // Orchestrator threads (core_trackers_[thread_idx].core_num() == 0) are a no-op. - int32_t shutdown(int32_t thread_idx); + int32_t shutdown(Runtime *runtime, int32_t thread_idx); // Run all post-attach scheduler bookkeeping, once, on the boot leader: // - publishes core assignments to the perf collector (SIMPLER_DFX) @@ -195,6 +195,9 @@ class SchedulerContext { // Per-core execution state, indexed by core_id (= worker_id) CoreExecState core_exec_states_[RUNTIME_MAX_WORKER]; + // Normal and emergency teardown may race; one winner owns each register + // window and its return gate until the close has drained. + std::atomic core_retired_[RUNTIME_MAX_WORKER]{}; // Cluster-ordered core trackers, one per scheduler thread CoreTracker core_trackers_[MAX_AICPU_THREADS]; @@ -295,6 +298,7 @@ class SchedulerContext { // Emergency shutdown: broadcast exit signal to every handshake'd core and // deinit their AICore register blocks. Idempotent. void emergency_shutdown(Runtime *runtime); + int32_t retire_cores(Runtime *runtime, const int32_t *core_ids, int32_t core_num); __attribute__((noinline, cold)) void fail_scheduler(Runtime *runtime, int32_t thread_idx, int32_t error_code); diff --git a/src/a5/runtime/tensormap_and_ringbuffer/aicore/aicore_executor.cpp b/src/a5/runtime/tensormap_and_ringbuffer/aicore/aicore_executor.cpp index fd2bf72e5a..644db8d835 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/aicore/aicore_executor.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/aicore/aicore_executor.cpp @@ -94,7 +94,7 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in // Phase 2: Wait for the AICPU to open our register window. A kernel launch // resets DATA_MAIN_BASE to 0 (verified on a2a3 silicon; a5 shares this // register protocol and relies on CI); the AICPU writes DATA_MAIN_BASE = - // AICPU_IDLE_TASK_ID (non-zero) as it opens FAST_PATH, so a non-zero read + // AICPU_IDLE_TASK_ID (non-zero) to open the window, so a non-zero read // means the window is open and reads/writes are valid. The AICPU runs // assign_cores_to_threads (µs) between opening the window and the first // dispatch, so this IDLE is observed long before any task_id lands — the @@ -104,7 +104,7 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in while (read_reg(RegId::DATA_MAIN_BASE) == 0) { SPIN_WAIT_HINT(); } - // Report initial idle status via register (FAST_PATH is now open). + // Report initial idle status via register (the window is now open). write_reg(RegId::COND, AICORE_IDLE_VALUE); // The AICPU writes task after observing our report (so our CACHELINE_OUT flush @@ -266,4 +266,5 @@ __aicore__ __attribute__((weak)) void aicore_execute(__gm__ Runtime *runtime, in // Flush all dirty cache lines to HBM before kernel exit. dcci(my_hank, SINGLE_CACHE_LINE, CACHELINE_OUT); + wait_for_post_close_release(&runtime->dev.teardown_gates[s_block_idx].post_close_release); } diff --git a/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp b/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp index 5c5f15cd05..18cef7b2e1 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/aicpu/aicpu_executor.cpp @@ -1026,8 +1026,8 @@ int32_t AicpuExecutor::run(Runtime *runtime) { } // Always shutdown AICore — even if sched_ctx_.completed_ was already true. - // platform_deinit_aicore_regs is idempotent; orchestrator threads have - // core_trackers_[thread_idx].core_num() == 0 so they skip the loop harmlessly. + // The retirement claim makes repeated requests harmless; orchestrator + // threads own no group. Initializers service requests that precede READY. int32_t shutdown_rc = sched_ctx_.shutdown(thread_idx); // Both outcomes reach the terminal record before this thread's arrival, and // with the state that produced each: folding shutdown_rc into run_rc first diff --git a/src/a5/runtime/tensormap_and_ringbuffer/docs/RUNTIME_LOGIC.md b/src/a5/runtime/tensormap_and_ringbuffer/docs/RUNTIME_LOGIC.md index 2f5ed2df39..262211b5e8 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/docs/RUNTIME_LOGIC.md +++ b/src/a5/runtime/tensormap_and_ringbuffer/docs/RUNTIME_LOGIC.md @@ -647,7 +647,7 @@ Public surface (called from `AicpuExecutor::init/run/deinit`): | `post_handshake_init(runtime)` | leader, after barrier | Build worker-id lists in core order, assign cores to threads | | `bind_runtime(rt)` | device-orch only | Wire `sched_` to `rt->scheduler` once the orchestrator thread creates `rt` | | `resolve_and_dispatch(runtime, thread_idx)` | per scheduler thread | Main dispatch loop | -| `shutdown(thread_idx)` | per thread on exit | `platform_deinit_aicore_regs` for this thread's cores | +| `shutdown(thread_idx)` | per thread on exit | Claim the published retirement group, finalize PMU on healthy runs, then retire the group | | `on_orchestration_done(runtime, rt, thread_idx, total_tasks)` | orchestrator thread | Publish core assignments, latch task count, fold inline-completed tasks, flip `orchestrator_done_`, and run `emergency_shutdown` on fatal | | `deinit()` | once per run | Reset every scheduler-owned field to its post-construction default | | Read-only accessors | various | `aic_count()` / `aiv_count()` / `is_completed()` / `completed_tasks_count()` | diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h b/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h index d9f715d912..5d372d5bea 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/runtime.h @@ -43,6 +43,7 @@ #include "common/platform_config.h" #include "aicpu/platform_aicpu_affinity.h" // MAX_GATE_THREADS (aicpu_allowed_cpus bound) #include "dispatch_payload.h" +#include "aicore_teardown.h" #include "task_args.h" #include "common/launch_entry_args.h" // EntryArgsSource, LaunchEntryArgsPlan #include "tensormap_and_ringbuffer/entry_args.h" // EntryArgsStorage @@ -159,8 +160,9 @@ inline bool aicore_report_accepted(const volatile Handshake *handshake, uint64_t * Three lengths, in order. `runtime_device_copy_size` is what a steady-state run * re-publishes and stops before `workers`; `runtime_device_initialized_prefix_size` * adds `workers`, and is what the first publication onto an allocation sends so - * the handshake region starts defined. This runtime has no gate tail, so that - * second length equals `runtime_device_extent_size`. + * the handshake region starts defined. The return gates after `workers` are + * AICPU-initialized, so `runtime_device_extent_size` carries them and is the + * larger of the last two. * * Adding a field here grows the device image; adding a field to Runtime's * host-only tail does not. Keep it standard-layout (static_assert below) so the @@ -253,9 +255,10 @@ struct alignas(64) DeviceRuntimeLaunchDesc { // the AICPU the task pointer it answers with — and no host value is // consumed. A steady-state run re-uploads none of it; the first publication // onto a given allocation carries it once, which is what gives a fresh - // block a defined starting value. This runtime has no gate tail, so the - // initialized prefix ends with this array, at the end of the descriptor. + // block a defined starting value. The initialized prefix ends here; the + // AICPU initializes the isolated return gates before opening any window. Handshake workers[RUNTIME_MAX_WORKER]; + AicoreTeardownControl teardown_gates[RUNTIME_MAX_WORKER]; }; // ============================================================================= @@ -298,6 +301,7 @@ class Runtime { int get_aicpu_thread_num() const { return dev.aicpu_thread_num; } void set_aicpu_thread_num(int n) { dev.aicpu_thread_num = n; } Handshake *get_workers() { return dev.workers; } + AicoreTeardownControl *get_teardown_gates() { return dev.teardown_gates; } const Handshake *get_workers() const { return dev.workers; } int32_t get_aicpu_allowed_cpu_count() const { return dev.aicpu_allowed_cpu_count; } void set_aicpu_allowed_cpu_count(int32_t n) { dev.aicpu_allowed_cpu_count = n; } @@ -491,9 +495,17 @@ static_assert( ); static_assert( offsetof(DeviceRuntimeLaunchDesc, workers) + sizeof(DeviceRuntimeLaunchDesc::workers) == + offsetof(DeviceRuntimeLaunchDesc, teardown_gates), + "teardown_gates must immediately follow the initialized handshake prefix" +); +static_assert( + offsetof(DeviceRuntimeLaunchDesc, teardown_gates) % 64 == 0, + "return gates must not share cache lines with handshake writes" +); +static_assert( + offsetof(DeviceRuntimeLaunchDesc, teardown_gates) + sizeof(DeviceRuntimeLaunchDesc::teardown_gates) == sizeof(DeviceRuntimeLaunchDesc), - "workers must end the descriptor on this runtime: it has no gate tail, so the initialized prefix is " - "the whole extent and a field appended behind it would never be published" + "teardown_gates must end the descriptor" ); // Bytes a steady-state run uploads: the descriptor before the handshake region. @@ -502,13 +514,12 @@ static_assert( size_t runtime_device_copy_size(const Runtime &rt); // Bytes the first publication onto a device allocation uploads: through the end -// of the handshake region. A5 trb has no post-close gate array, so this equals -// the device extent below. +// of the handshake region, excluding the AICPU-initialized return gates. size_t runtime_device_initialized_prefix_size(const Runtime &rt); // Bytes of device memory a Runtime image occupies, and the size every allocation // backing a device `Runtime` must use. Never smaller than -// `runtime_device_initialized_prefix_size`; equal to it on this runtime. +// `runtime_device_initialized_prefix_size`. size_t runtime_device_extent_size(const Runtime &rt); // This run's entry-argument routing facts, captured from `rt` while it is still diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp index 8d02c57b23..d56d441b19 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_cold_path.cpp @@ -9,6 +9,7 @@ * ----------------------------------------------------------------------------------------------------------- */ #include "scheduler_context.h" +#include "utils/fatal_shutdown_latch.h" #include #include @@ -66,17 +67,13 @@ LoopAction SchedulerContext::handle_orchestrator_exit( "completed_tasks=%d, total_tasks=%d", thread_idx, orch_err, completed_tasks_.load(std::memory_order_relaxed), total_tasks_ ); - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } + emergency_shutdown(runtime); return LoopAction::BREAK_LOOP; } int32_t sched_err = header->sched_error_code.load(std::memory_order_acquire); if (sched_err != SIMPLER_ERROR_NONE) { LOG_ERROR("Thread %d: Scheduler fatal error detected (code=%d)", thread_idx, sched_err); - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } + emergency_shutdown(runtime); return LoopAction::BREAK_LOOP; } @@ -101,17 +98,13 @@ LoopAction SchedulerContext::check_idle_fatal_error(int32_t thread_idx, SharedMe int32_t orch_err = header->orch_error_code.load(std::memory_order_acquire); if (orch_err != SIMPLER_ERROR_NONE) { LOG_ERROR("Thread %d: Fatal error detected (code=%d), sending EXIT_SIGNAL to all cores", thread_idx, orch_err); - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } + emergency_shutdown(runtime); return LoopAction::BREAK_LOOP; } int32_t sched_err = header->sched_error_code.load(std::memory_order_acquire); if (sched_err != SIMPLER_ERROR_NONE) { LOG_ERROR("Thread %d: Scheduler fatal error detected (code=%d)", thread_idx, sched_err); - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } + emergency_shutdown(runtime); return LoopAction::BREAK_LOOP; } return LoopAction::NONE; @@ -474,7 +467,7 @@ int32_t SchedulerContext::handle_timeout_exit( // sees the locators above already settled. header->sched_stall_detail.store(cls.detail, std::memory_order_release); } - if (!completed_.exchange(true, std::memory_order_acq_rel)) { + if (begin_emergency_shutdown()) { log_shutdown_stall_snapshot(thread_idx, idle_iterations, last_progress_count); #if SIMPLER_DFX // Capture the in-flight kernels' partial output before signalling the @@ -500,7 +493,7 @@ int32_t SchedulerContext::handle_timeout_exit( ); } #endif - emergency_shutdown(runtime); + signal_emergency_shutdown(runtime); } #if SIMPLER_DFX uint64_t sched_timeout_ts = get_sys_cnt_aicpu(); @@ -656,40 +649,117 @@ void SchedulerContext::log_chip_swimlane_summary(int32_t thread_idx, [[maybe_unu #endif // ============================================================================= -// Shutdown: deinit AICore regs for this thread's cores. -// Orchestrator threads have core_trackers_[thread_idx].core_num() == 0 -> no-op. -// platform_deinit_aicore_regs is idempotent; safe to call after early completion. +// Shutdown: each thread retires the cores it owns, on its own way out. +// Normal shutdown and the emergency all-core sweep arbitrate per core. // ============================================================================= int32_t SchedulerContext::shutdown(int32_t thread_idx) { - const int32_t *cores = core_trackers_[thread_idx].core_ids(); - int32_t core_num = core_trackers_[thread_idx].core_num(); - if (core_num == 0) return 0; + if (thread_idx < 0 || thread_idx >= aicpu_thread_num_) return 0; #if SIMPLER_DFX - // Restore PMU CTRL registers for this thread's cores before AICore shutdown - if (is_pmu_enabled()) { - pmu_aicpu_finalize(cores, core_num); + // Restore PMU CTRL registers for this thread's cores before AICore + // shutdown. A fatal run ends in a host-side device reset, so counters read + // here would not survive into the next generation. + if (is_pmu_enabled() && !fatal_shutdown_started_.load(std::memory_order_acquire)) { + pmu_aicpu_finalize(core_trackers_[thread_idx].core_ids(), core_trackers_[thread_idx].core_num()); } #endif - LOG_INFO("Thread %d: Shutting down %d cores", thread_idx, core_num); - int32_t rc = 0; - for (int32_t i = 0; i < core_num; i++) { - int32_t core_id = cores[i]; - uint64_t reg_addr = core_exec_states_[core_id].reg_addr; - if (reg_addr != 0) { - // Timeout means AICore is unresponsive. Log and continue deiniting remaining cores. - if (platform_deinit_aicore_regs(reg_addr) != 0) { - LOG_ERROR("Thread %d: Core %d deinit timed out", thread_idx, core_id); - rc = -1; + return retire_cores(core_trackers_[thread_idx].core_ids(), core_trackers_[thread_idx].core_num()); +} + +void SchedulerContext::publish_retirement_group(int32_t owner_thread) { + int32_t ids[PLATFORM_MAX_CORES]; + const int32_t count = retirement_group_ids(owner_thread, ids); + int32_t claimed[PLATFORM_MAX_CORES]; + int32_t claimed_count = 0; + for (int32_t i = 0; i < count; ++i) { + if (__atomic_fetch_or(&retirement_state_[ids[i]], RETIREMENT_READY, __ATOMIC_ACQ_REL) == RETIREMENT_REQUESTED) { + claimed[claimed_count++] = ids[i]; + } + } + (void)retire_claimed_cores(claimed, claimed_count); +} + +int32_t SchedulerContext::retirement_group_ids(int32_t owner_thread, int32_t *ids) { + // A failed assignment can leave the tracker empty or incomplete. The + // barrier-free handshake has a fixed blocked partition independent of it. + int32_t count = 0; + if (retirement_blocked_layout_) { + if (owner_thread >= active_sched_threads_) return 0; + const int32_t aic_n = cores_total_num_ / PLATFORM_CORES_PER_BLOCKDIM; + for (int32_t ci = owner_thread; ci < aic_n; ci += active_sched_threads_) { + ids[count++] = ci; + ids[count++] = aic_n + 2 * ci; + ids[count++] = aic_n + 2 * ci + 1; + } + } else if (retirement_unassigned_) { + // The serial initializer has joined every handshake before publishing + // this fallback. One owner covers every initialized core on failure. + if (owner_thread != 0) return 0; + for (int32_t i = 0; i < cores_total_num_; ++i) + ids[count++] = i; + } else { + const auto &tracker = core_trackers_[owner_thread]; + for (int32_t i = 0; i < tracker.core_num(); ++i) + ids[count++] = tracker.core_ids()[i]; + } + return count; +} + +int32_t SchedulerContext::retire_cores(const int32_t *core_ids, int32_t core_num) { + int32_t claimed[PLATFORM_MAX_CORES]; + int32_t count = 0; + for (int32_t i = 0; i < core_num; ++i) { + const int32_t core_id = core_ids[i]; + if (core_id < 0 || core_id >= cores_total_num_) continue; + if (__atomic_fetch_or(&retirement_state_[core_id], RETIREMENT_REQUESTED, __ATOMIC_ACQ_REL) == + RETIREMENT_READY) { + claimed[count++] = core_id; + } + } + return retire_claimed_cores(claimed, count); +} + +int32_t SchedulerContext::retire_claimed_cores(const int32_t *core_ids, int32_t core_num) { + AicoreExitTarget targets[PLATFORM_MAX_CORES]; + int32_t claimed_ids[PLATFORM_MAX_CORES]; + size_t count = 0; + for (int32_t i = 0; i < core_num; ++i) { + const int32_t core_id = core_ids[i]; + if (core_id < 0 || core_id >= cores_total_num_) continue; + if (core_exec_states_[core_id].reg_addr == 0) continue; + claimed_ids[count] = core_id; + targets[count] = {core_exec_states_[core_id].reg_addr, &teardown_gates_[core_id]}; + ++count; + } + if (count == 0) return 0; + + // platform_retire_aicore_group fills every entry on every path it returns + // from, so this needs no initializer. + bool released[PLATFORM_MAX_CORES]; + const int32_t rc = platform_retire_aicore_group(targets, count, platform_aicore_exit_deadline(), released); + if (rc != 0) { + // Naming the cores is the only signal an unretired core leaves: the host + // sees just a stream timeout. + for (size_t i = 0; i < count; ++i) { + if (!released[i]) { + LOG_ERROR( + "AICore retirement: core %d not released (COND=0x%llx)", claimed_ids[i], + static_cast(read_reg(targets[i].reg_addr, RegId::COND)) + ); } - } else { - LOG_ERROR("Thread %d: Core %d has invalid register address", thread_idx, core_id); } } return rc; } +int32_t SchedulerContext::retire_all_cores() { + int32_t ids[PLATFORM_MAX_CORES]; + for (int32_t i = 0; i < cores_total_num_; ++i) + ids[i] = i; + return retire_cores(ids, cores_total_num_); +} + // ============================================================================= // Handshake a contiguous slice of AICore workers. Runs on every AICPU thread in // parallel (partitioned by tidx/nthreads); the leader's pre_handshake_init has @@ -732,7 +802,7 @@ void SchedulerContext::handshake_partition(Runtime *runtime, int32_t tidx, int32 // way RegId::COND polling is. // // Servicing a core = validate its physical_core_id, then open its register - // window (platform_init_aicore_regs: FAST_PATH + DATA_MAIN_BASE=IDLE). That + // window (platform_init_aicore_regs: DATA_MAIN_BASE=IDLE). That // IDLE write is *also* the signal the core polls for to leave its // post-report wait — so opening the window IS the acknowledgement. There is // no separate aicpu_regs_ready ack and no second round-trip. AIC/AIV @@ -959,6 +1029,7 @@ void SchedulerContext::assign_own_clusters(int32_t tidx) { CoreTracker::MAX_CLUSTERS ); handshake_failed_.store(true, std::memory_order_release); + publish_retirement_group(tidx); return; } tracker.init(own_n); @@ -1003,16 +1074,13 @@ void SchedulerContext::assign_own_clusters(int32_t tidx) { } } } + publish_retirement_group(tidx); } // Abort the run on a handshake failure discovered without the all-thread barrier // (non-DFX path): latch completion so every scheduler thread exits its dispatch // loop, and broadcast exit to whatever cores did come up. Idempotent. -void SchedulerContext::abort_and_shutdown(Runtime *runtime) { - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } -} +void SchedulerContext::abort_and_shutdown(Runtime *runtime) { emergency_shutdown(runtime); } // Profiling-subsystem init (leader-only). pmu_aicpu_init needs every core's // physical_core_id, so the barrier-free init path calls this behind an @@ -1094,23 +1162,25 @@ bool SchedulerContext::assign_cores_to_threads() { // Emergency shutdown: broadcast exit signal to every handshake'd core and // deinit their AICore register blocks. Idempotent. // ============================================================================= +bool SchedulerContext::begin_emergency_shutdown() { + return publish_fatal_shutdown(fatal_shutdown_started_, completed_); +} + +void SchedulerContext::signal_emergency_shutdown(Runtime *runtime) { + (void)runtime; // exit is delivered via each core's register block, not GM + // Sweeps every core rather than one thread's slice: a fatal run must not + // depend on ready owners reaching their own shutdown. Owners still in init + // service the request when they publish their group. The retirement + // writes DATA_MAIN_BASE=EXIT, which both releases a core still polling for + // its window to open and signals it to exit. Cores whose windows never + // opened (reg_addr==0) remain the host recovery path's responsibility. + LOG_WARN("Emergency shutdown: retiring all initialized AICores"); + (void)retire_all_cores(); +} + void SchedulerContext::emergency_shutdown(Runtime *runtime) { - (void)runtime; // exit is now delivered via each core's register block, not GM - LOG_WARN("Emergency shutdown: sending exit signal to all initialized cores"); - int32_t timeout_count = 0; - for (int32_t i = 0; i < cores_total_num_; i++) { - // platform_deinit_aicore_regs writes DATA_MAIN_BASE=EXIT, which both - // releases a core still polling for its window to open and signals it to - // exit. Cores never opened (reg_addr==0) are reaped by the host device - // reset that follows a handshake failure. - if (core_exec_states_[i].reg_addr != 0) { - if (platform_deinit_aicore_regs(core_exec_states_[i].reg_addr) != 0) { - timeout_count++; - } - } - } - if (timeout_count > 0) { - LOG_ERROR("Emergency shutdown: %d cores did not acknowledge exit", timeout_count); + if (begin_emergency_shutdown()) { + signal_emergency_shutdown(runtime); } } @@ -1124,6 +1194,10 @@ int32_t SchedulerContext::pre_handshake_init( // Zero all per-core execution state before handshake memset(core_exec_states_, 0, sizeof(core_exec_states_)); + // Reset before hs_setup_done_ releases any initializer or emergency caller. + memset(retirement_state_, 0, sizeof(retirement_state_)); + retirement_blocked_layout_ = aicpu_thread_num > 1 && !runtime->dev.serial_orch_sched; + retirement_unassigned_ = false; // Wire thread/transition configuration that handshake/assign need to read. aicpu_thread_num_ = aicpu_thread_num; @@ -1188,6 +1262,9 @@ int32_t SchedulerContext::pre_handshake_init( // their owned clusters (assign_own_clusters) without the post-handshake // discovery pass and its all-thread barrier. const int32_t cluster_num = cores_total_num_ / PLATFORM_CORES_PER_BLOCKDIM; + teardown_gates_ = runtime->get_teardown_gates(); + memset(teardown_gates_, 0, sizeof(AicoreTeardownControl) * cores_total_num_); + wmb(); aic_count_ = cluster_num * PLATFORM_AIC_CORES_PER_BLOCKDIM; aiv_count_ = cluster_num * PLATFORM_AIV_CORES_PER_BLOCKDIM; active_sched_threads_ = (sched_thread_num_ > 0) ? sched_thread_num_ : aicpu_thread_num_; @@ -1223,6 +1300,9 @@ int32_t SchedulerContext::pre_handshake_init( int32_t SchedulerContext::post_handshake_init(Runtime *runtime) { if (handshake_failed_.load(std::memory_order_acquire)) { + retirement_unassigned_ = true; + for (int32_t t = 0; t < aicpu_thread_num_; ++t) + publish_retirement_group(t); emergency_shutdown(runtime); return -1; } @@ -1250,6 +1330,10 @@ int32_t SchedulerContext::post_handshake_init(Runtime *runtime) { LOG_INFO("Core discovery complete: %d AIC, %d AIV", aic_count_, aiv_count_); if (!assign_cores_to_threads()) { + retirement_unassigned_ = true; + for (int32_t t = 0; t < aicpu_thread_num_; ++t) + publish_retirement_group(t); + emergency_shutdown(runtime); return -1; } @@ -1325,6 +1409,8 @@ int32_t SchedulerContext::post_handshake_init(Runtime *runtime) { func_id_to_addr_ = reinterpret_cast(runtime->dev.callable_table_addr_); func_id_to_addr_count_ = runtime->dev.callable_table_len_; + for (int32_t t = 0; t < aicpu_thread_num_; ++t) + publish_retirement_group(t); return 0; } @@ -1363,6 +1449,7 @@ void SchedulerContext::deinit() { total_tasks_ = 0; orchestrator_done_.store(false, std::memory_order_release); completed_.store(false, std::memory_order_release); + fatal_shutdown_started_.store(false, std::memory_order_release); // Reset core discovery and assignment state aic_count_ = 0; @@ -1430,9 +1517,7 @@ void SchedulerContext::on_orchestration_done( orch_err = sched_->sm_header->orch_error_code.load(std::memory_order_relaxed); } if (orch_err != SIMPLER_ERROR_NONE) { - if (!completed_.exchange(true, std::memory_order_acq_rel)) { - emergency_shutdown(runtime); - } + emergency_shutdown(runtime); } #if SIMPLER_DFX diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_context.h b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_context.h index fd3b540882..bd220a4934 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_context.h +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/scheduler/scheduler_context.h @@ -49,6 +49,8 @@ struct RuntimeContext; * - scheduler_dispatch.cpp (task dispatch loop and helpers) */ class SchedulerContext { + friend class SchedulerRetirementTestPeer; + public: // ========================================================================= // Lifecycle @@ -107,6 +109,12 @@ class SchedulerContext { // Orchestrator threads (core_trackers_[thread_idx].core_num() == 0) are a no-op. int32_t shutdown(int32_t thread_idx); + // Claim each core independently; requests for unpublished cores stay pending. + int32_t retire_cores(const int32_t *core_ids, int32_t core_num); + + // Request every core before retiring the ready winners in one platform batch. + int32_t retire_all_cores(); + // Run all post-orchestration scheduler bookkeeping: // - publishes core assignments to the perf collector (SIMPLER_DFX) // - latches submitted task count from shared memory @@ -192,6 +200,18 @@ class SchedulerContext { // be submitted; schedulers poll it. std::atomic orchestrator_done_{false}; std::atomic completed_{false}; + // Published before completed_, so a thread that observes completion also + // observes this and cannot enter the healthy shutdown path for a fatal run. + std::atomic fatal_shutdown_started_{false}; + // Both participants modify each core's atomic byte: the operation that sees + // the other bit alone owns retirement. READY publishes initialization; + // REQUESTED can arrive before initialization without consuming an empty set. + static constexpr uint8_t RETIREMENT_READY = 1; + static constexpr uint8_t RETIREMENT_REQUESTED = 2; + uint8_t retirement_state_[PLATFORM_MAX_CORES]{}; + AicoreTeardownControl *teardown_gates_{nullptr}; + bool retirement_blocked_layout_{false}; + bool retirement_unassigned_{false}; // The active callable's registration-owned object-address table and the // number of entries it holds, both bound from the descriptor in the cold // path. The table is in the callable's registration block, not in the @@ -245,6 +265,15 @@ class SchedulerContext { // deinit their AICore register blocks. Idempotent. void emergency_shutdown(Runtime *runtime); + // Elect exactly one thread to drive the emergency retirement. Returns true + // to the elected caller only; publishes the fatal flag before the + // completion latch. + bool begin_emergency_shutdown(); + void signal_emergency_shutdown(Runtime *runtime); + void publish_retirement_group(int32_t owner_thread); + int32_t retirement_group_ids(int32_t owner_thread, int32_t *ids); + int32_t retire_claimed_cores(const int32_t *core_ids, int32_t core_num); + // ========================================================================= // Dispatch (scheduler_dispatch.cpp) // ========================================================================= diff --git a/src/a5/runtime/tensormap_and_ringbuffer/runtime/shared/runtime.cpp b/src/a5/runtime/tensormap_and_ringbuffer/runtime/shared/runtime.cpp index b02df60a9f..55eabfe392 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/runtime/shared/runtime.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/runtime/shared/runtime.cpp @@ -194,7 +194,9 @@ LaunchEntryArgsPlan runtime_launch_entry_args_plan(const Runtime &rt) { // The first publication onto an allocation adds the handshake region, so it // starts from the ctor-zeroed host copy rather than from whatever rtMalloc -// left. This runtime has no host-uninitialized tail, so that reaches the end. -size_t runtime_device_initialized_prefix_size(const Runtime &) { return sizeof(DeviceRuntimeLaunchDesc); } +// left. Return gates are initialized by the AICPU before window-open. +size_t runtime_device_initialized_prefix_size(const Runtime &) { + return offsetof(DeviceRuntimeLaunchDesc, teardown_gates); +} size_t runtime_device_extent_size(const Runtime &) { return sizeof(DeviceRuntimeLaunchDesc); } diff --git a/src/common/host_build_graph/runtime.h b/src/common/host_build_graph/runtime.h index be59518c60..91dac328fc 100644 --- a/src/common/host_build_graph/runtime.h +++ b/src/common/host_build_graph/runtime.h @@ -125,8 +125,8 @@ struct Handshake { // The AICore owns this line's writeback: it flushes the whole line with // dcci(..., CACHELINE_OUT) on its report and again on exit. A word the AICPU // must publish independently cannot live here — a stale line writeback would -// overwrite it. The A2/A3 post-close return gates live in -// Runtime::teardown_gates, one isolated line each; A5 leaves them unused. +// overwrite it. The post-close return gates live in +// Runtime::teardown_gates, one isolated line each on both platforms. static_assert(sizeof(Handshake) == 64); static_assert(std::is_standard_layout_v && std::is_trivially_copyable_v); // The payload offsets are the device-side wire contract: AICore writes them and @@ -330,16 +330,14 @@ struct alignas(64) DeviceRuntimeLaunchDesc { // what gives a fresh block a defined starting value. Handshake workers[RUNTIME_MAX_WORKER]; - // A2/A3 post-close return gates, one isolated cache line per worker. The + // Post-close return gates, one isolated cache line per worker. The // AICPU stores here only after that worker's register window is closed; // the AICore bypass-loads its own entry and returns once it reads RELEASE. // Separate from workers[] because the AICore flushes its whole Handshake - // line, which would overwrite a gate sharing it. Unused reserved storage on - // A5 — a declared member that occupies layout, read and written by no A5 - // code, and A5 runs no gate initialization. + // line, which would overwrite a gate sharing it. // // Last, and outside both the uploaded prefix and the initialized prefix: on - // A2/A3 the AICPU zeroes every active entry in `pre_handshake_init` and + // both platforms the AICPU zeroes every active entry in `pre_handshake_init` and // executes `wmb()` before it publishes `hs_setup_done_`, and no register // window opens before that publication, so the meaningful initial value is // produced on the device ahead of every read of it. No host-supplied gate diff --git a/src/common/task_interface/aicore_teardown.h b/src/common/task_interface/aicore_teardown.h index 5247381027..bd4634baf7 100644 --- a/src/common/task_interface/aicore_teardown.h +++ b/src/common/task_interface/aicore_teardown.h @@ -17,9 +17,9 @@ constexpr uint32_t AICORE_POST_CLOSE_RELEASE = 1; -// A2/A3: AICPU resets this word before window-open and publishes it only after +// AICPU resets this word before window-open and publishes it only after // window-close. AICore bypass-loads it after EXITED; no cached stores or dcci -// may touch this line while the protocol is active. A5 leaves it unused. +// may touch this line while the protocol is active. struct alignas(64) AicoreTeardownControl { uint32_t post_close_release; }; diff --git a/tests/ut/cpp/a5/platform/CMakeLists.txt b/tests/ut/cpp/a5/platform/CMakeLists.txt index 4def80a9c1..2df8850bd8 100644 --- a/tests/ut/cpp/a5/platform/CMakeLists.txt +++ b/tests/ut/cpp/a5/platform/CMakeLists.txt @@ -44,4 +44,26 @@ add_custom_command(TARGET test_a5_aicpu_topology_fallback POST_BUILD "$/aicpu_cpu_topo_fallback.json" ) +a5_platform_case(test_aicore_retirement.cpp + SOURCES + ${SIMPLER_SRC}/a5/platform/shared/aicpu/platform_regs.cpp + ${SIMPLER_SRC}/a5/platform/sim/aicpu/inner_platform_regs.cpp + ${SIMPLER_SRC}/common/platform/sim/aicpu/device_time.cpp + ${HOST_LOG_TEST_SOURCES} + INCLUDES ${SIMPLER_SRC}/common/platform/sim/aicpu) +set_tests_properties(test_a5_aicore_retirement PROPERTIES TIMEOUT 20) + +# The AICore half of the return gate: it compiles the real a5sim +# aicore/inner_kernel.h (so the sim include directory is on the path) and the +# same production platform_regs.cpp retire path as the case above, which is what +# publishes the gate the worker waits on. +a5_platform_case(test_return_gate_wait.cpp + SOURCES + ${SIMPLER_SRC}/a5/platform/shared/aicpu/platform_regs.cpp + ${SIMPLER_SRC}/a5/platform/sim/aicpu/inner_platform_regs.cpp + ${SIMPLER_SRC}/common/platform/sim/aicpu/device_time.cpp + ${HOST_LOG_TEST_SOURCES} + INCLUDES ${SIMPLER_SRC}/common/platform/sim/aicpu ${SIMPLER_SRC}/a5/platform/sim) +set_tests_properties(test_a5_return_gate_wait PROPERTIES TIMEOUT 20) + simpler_ut_glob_cases(a5_platform_case) diff --git a/tests/ut/cpp/a5/platform/test_aicore_retirement.cpp b/tests/ut/cpp/a5/platform/test_aicore_retirement.cpp new file mode 100644 index 0000000000..27ed48e29d --- /dev/null +++ b/tests/ut/cpp/a5/platform/test_aicore_retirement.cpp @@ -0,0 +1,342 @@ +/* + * Copyright (c) PyPTO Contributors. + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of + * CANN Open Software License Agreement Version 2.0 (the "License"). + * Please refer to the License for details. You may not use this file except in compliance with the License. + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. + * See LICENSE in the root of the software repository for the full text of the License. + * ----------------------------------------------------------------------------------------------------------- + */ + +/** + * Retirement behaves the same whether every core answers or none does, and the + * cases that matter are the ones where a core does not. A run that ends cleanly + * exercises none of them, and a host-side test cannot see them either: the + * device is force-reset on the path that produces them, so the per-core report + * does not reliably reach the host log. These drive the group directly against + * simulated register blocks, where a core's acknowledgement is something the + * test decides rather than something it waits for. + * + * a5 has no fast-path window control. The post-close return gates and the group + * promises: every core signalled before any is waited on, no window closed until + * the whole group has acknowledged, an unacknowledged core left alone and named, + * and one shared deadline for the group rather than one per core. + */ + +#include + +#include +#include +#include +#include +#include +#include + +#include "aicpu/device_time.h" +#include "aicpu/platform_regs.h" +#include "common/platform_config.h" + +namespace { + +// One sparse register block per core, laid out the way sparse_reg_ptr expects. +class CoreRegs { +public: + CoreRegs() { block_.fill(0); } + uint64_t addr() { return reinterpret_cast(block_.data()); } + uint32_t dispatch() { return static_cast(read_reg(addr(), RegId::DATA_MAIN_BASE)); } + void ack() { write_reg(addr(), RegId::COND, AICORE_EXITED_VALUE); } + bool signalled() { return dispatch() == AICORE_EXIT_SIGNAL; } + bool closed() { return dispatch() == AICPU_IDLE_TASK_ID; } + +private: + alignas(64) std::array block_{}; +}; + +template +bool wait_until(Predicate predicate, std::chrono::milliseconds budget) { + const auto deadline = std::chrono::steady_clock::now() + budget; + while (!predicate()) { + if (std::chrono::steady_clock::now() >= deadline) return false; + std::this_thread::yield(); + } + return true; +} + +template +std::array +make_targets(const uint64_t (&addrs)[N], std::array &teardowns) { + std::array targets{}; + for (size_t i = 0; i < N; ++i) + targets[i] = {addrs[i], &teardowns[i]}; + return targets; +} + +// Short enough to keep the timeout cases quick, long enough that a sweep over a +// handful of cores cannot exhaust it on a loaded machine. +uint64_t short_deadline() { return get_sys_cnt_aicpu() + PLATFORM_PROF_SYS_CNT_FREQ / 10; } + +} // namespace + +TEST(AicoreRetirement, ReturnGatesReleaseOnlyAcknowledgedCoresAfterClose) { + std::array cores; + std::array gates{}; + AicoreExitTarget targets[] = {{cores[0].addr(), &gates[0]}, {cores[1].addr(), &gates[1]}}; + bool released[2]; + cores[0].ack(); + EXPECT_EQ(platform_retire_aicore_group(targets, 2, short_deadline(), released), -1); + EXPECT_TRUE(cores[0].closed()); + EXPECT_EQ(gates[0].post_close_release, AICORE_POST_CLOSE_RELEASE); + EXPECT_FALSE(cores[1].closed()); + EXPECT_EQ(gates[1].post_close_release, 0U); + EXPECT_TRUE(released[0]); + EXPECT_FALSE(released[1]); +} + +TEST(AicoreRetirement, ReturnGateStaysClosedWhilePeerAckIsPending) { + std::array cores; + std::array gates{}; + AicoreExitTarget targets[] = {{cores[0].addr(), &gates[0]}, {cores[1].addr(), &gates[1]}}; + cores[0].ack(); + int rc = -1; + std::thread caller([&] { + rc = platform_retire_aicore_group(targets, 2, platform_aicore_exit_deadline()); + }); + EXPECT_TRUE(wait_until( + [&] { + return cores[1].signalled(); + }, + std::chrono::milliseconds(500) + )); + EXPECT_EQ(__atomic_load_n(&gates[0].post_close_release, __ATOMIC_ACQUIRE), 0U); + cores[1].ack(); + caller.join(); + EXPECT_EQ(rc, 0); + for (size_t i = 0; i < cores.size(); ++i) { + EXPECT_TRUE(cores[i].closed()); + EXPECT_EQ(gates[i].post_close_release, AICORE_POST_CLOSE_RELEASE); + } +} + +TEST(AicoreRetirement, NullTeardownRejectsWholeBatchBeforeSignalling) { + std::array cores; + std::array teardowns{}; + AicoreExitTarget targets[] = {{cores[0].addr(), &teardowns[0]}, {cores[1].addr(), nullptr}}; + bool released[] = {true, true}; + EXPECT_EQ(platform_retire_aicore_group(targets, 2, short_deadline(), released), -1); + EXPECT_FALSE(cores[0].signalled()); + EXPECT_FALSE(cores[1].signalled()); + EXPECT_FALSE(released[0]); + EXPECT_FALSE(released[1]); +} + +// The broadcast is what lets the cores drain concurrently; if it were interleaved +// with the waiting, a core late to answer would hold up signalling the rest. +TEST(AicoreRetirement, SignalsEveryCoreBeforeWaitingOnAny) { + std::array cores; + uint64_t addrs[4]; + for (size_t i = 0; i < cores.size(); ++i) + addrs[i] = cores[i].addr(); + std::array teardowns{}; + auto targets = make_targets(addrs, teardowns); + + // A budget far longer than the window this test watches is what makes the + // assertion below discriminating: a retirement that signalled one core and + // waited on it before signalling the next could not have reached the last + // core within a fraction of a single budget. + const uint64_t long_budget = PLATFORM_PROF_SYS_CNT_FREQ * 2; + std::atomic done{false}; + std::thread caller([&] { + platform_retire_aicore_group(targets.data(), targets.size(), get_sys_cnt_aicpu() + long_budget); + done.store(true, std::memory_order_release); + }); + // Nothing has acknowledged yet, so the call is still sweeping. Every core + // must already carry the exit signal. + EXPECT_TRUE(wait_until( + [&] { + return cores[cores.size() - 1].signalled(); + }, + std::chrono::milliseconds(200) + )); + for (auto &core : cores) + EXPECT_TRUE(core.signalled()); + for (auto &core : cores) + core.ack(); + EXPECT_TRUE(wait_until( + [&] { + return done.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + caller.join(); +} + +// A core is never quiesced while a peer is still being waited on. +TEST(AicoreRetirement, ClosesNoWindowUntilTheGroupHasAcknowledged) { + std::array cores; + uint64_t addrs[2] = {cores[0].addr(), cores[1].addr()}; + std::array teardowns{}; + auto targets = make_targets(addrs, teardowns); + + std::atomic result{1}; + std::thread caller([&] { + // This case tests ACK ordering, not timeout. The observation thread can + // be descheduled arbitrarily without consuming the retirement budget. + // CTest bounds a broken implementation that never returns. + result.store( + platform_retire_aicore_group(targets.data(), targets.size(), std::numeric_limits::max()), + std::memory_order_release + ); + }); + // Nonfatal: both ACKs and join must run even when signal observation fails. + EXPECT_TRUE(wait_until( + [&] { + return cores[0].signalled() && cores[1].signalled(); + }, + std::chrono::seconds(2) + )); + cores[0].ack(); + // The second core has not answered, so neither window may close yet. + EXPECT_FALSE(wait_until( + [&] { + return cores[0].closed(); + }, + std::chrono::milliseconds(50) + )); + cores[1].ack(); + EXPECT_TRUE(wait_until( + [&] { + return result.load(std::memory_order_acquire) == 0; + }, + std::chrono::seconds(2) + )); + caller.join(); + EXPECT_TRUE(cores[0].closed()); + EXPECT_TRUE(cores[1].closed()); +} + +// The core that never answers keeps its exit signal: closing its window would +// hand it back while it is still whatever state it is stuck in. +TEST(AicoreRetirement, LeavesAnUnacknowledgedCoreUnclosed) { + std::array cores; + uint64_t addrs[2] = {cores[0].addr(), cores[1].addr()}; + std::array teardowns{}; + auto targets = make_targets(addrs, teardowns); + cores[1].ack(); + + EXPECT_EQ(platform_retire_aicore_group(targets.data(), targets.size(), get_sys_cnt_aicpu()), -1); + EXPECT_TRUE(cores[0].signalled()); + EXPECT_FALSE(cores[0].closed()); + EXPECT_TRUE(cores[1].closed()); +} + +// An unretired core leaves the host nothing but a stream timeout, so the caller +// needs the group to say which ones they were. +TEST(AicoreRetirement, ReportsWhichCoresWereReleased) { + std::array cores; + uint64_t addrs[2] = {cores[0].addr(), cores[1].addr()}; + std::array teardowns{}; + auto targets = make_targets(addrs, teardowns); + cores[1].ack(); + + bool released[2] = {true, false}; + EXPECT_EQ(platform_retire_aicore_group(targets.data(), targets.size(), get_sys_cnt_aicpu(), released), -1); + EXPECT_FALSE(released[0]); + EXPECT_TRUE(released[1]); + + std::array answered; + uint64_t answered_addrs[2] = {answered[0].addr(), answered[1].addr()}; + std::array answered_teardowns{}; + auto answered_targets = make_targets(answered_addrs, answered_teardowns); + for (auto &core : answered) + core.ack(); + bool all_released[2] = {}; + EXPECT_EQ( + platform_retire_aicore_group(answered_targets.data(), answered_targets.size(), short_deadline(), all_released), + 0 + ); + EXPECT_TRUE(all_released[0]); + EXPECT_TRUE(all_released[1]); +} + +// One budget for the group, not one per core: a wedged core must not cost every +// core behind it a timeout of its own. +TEST(AicoreRetirement, SpendsOneBudgetOnTheWholeGroup) { + constexpr size_t kCores = 8; + std::array cores; + uint64_t addrs[kCores]; + for (size_t i = 0; i < kCores; ++i) + addrs[i] = cores[i].addr(); + std::array teardowns{}; + auto targets = make_targets(addrs, teardowns); + + const uint64_t budget_ticks = PLATFORM_PROF_SYS_CNT_FREQ / 10; + const auto started = std::chrono::steady_clock::now(); + EXPECT_EQ(platform_retire_aicore_group(targets.data(), targets.size(), get_sys_cnt_aicpu() + budget_ticks), -1); + const auto elapsed = std::chrono::steady_clock::now() - started; + + // Two budgets of headroom absorbs scheduling noise; eight would be the cost + // of spending one budget per core. + const auto budget = std::chrono::milliseconds(100); + EXPECT_LT(elapsed, 2 * budget); + for (auto &core : cores) + EXPECT_FALSE(core.closed()); +} + +// The caller may pass an uninitialized buffer and read it on any return, so a +// rejected group still has to fill it, and must not touch a register first. +TEST(AicoreRetirement, RejectsInvalidGroupsWithoutTouchingRegisters) { + std::array cores; + std::array teardowns{}; + AicoreExitTarget targets[2] = {{cores[0].addr(), &teardowns[0]}, {0, &teardowns[1]}}; + + bool released[2] = {true, true}; + EXPECT_EQ(platform_retire_aicore_group(targets, 2, short_deadline(), released), -1); + EXPECT_FALSE(released[0]); + EXPECT_FALSE(released[1]); + EXPECT_EQ(cores[0].dispatch(), 0U); + + EXPECT_EQ(platform_retire_aicore_group(nullptr, 1, short_deadline()), -1); + EXPECT_EQ(platform_retire_aicore_group(nullptr, 0, short_deadline()), 0); + EXPECT_EQ(platform_retire_aicore_group(targets, PLATFORM_MAX_CORES + 1, short_deadline()), -1); +} + +// A one-target group follows the same deferred close path as a larger group. +TEST(AicoreRetirement, SingleCoreTargetUsesTheGroupPath) { + CoreRegs acked; + AicoreTeardownControl acked_teardown{}; + AicoreExitTarget acked_target{acked.addr(), &acked_teardown}; + acked.ack(); + EXPECT_EQ(platform_retire_aicore_group(&acked_target, 1, short_deadline()), 0); + EXPECT_TRUE(acked.closed()); + + // Signalling is not licence to quiesce a core that has not answered yet. + // Watching the window stay open across a settle interval checks that defer. + CoreRegs core; + AicoreTeardownControl teardown{}; + AicoreExitTarget target{core.addr(), &teardown}; + std::atomic done{false}; + std::thread caller([&] { + platform_retire_aicore_group(&target, 1, platform_aicore_exit_deadline()); + done.store(true, std::memory_order_release); + }); + EXPECT_TRUE(wait_until( + [&] { + return core.signalled(); + }, + std::chrono::milliseconds(200) + )); + std::this_thread::sleep_for(std::chrono::milliseconds(50)); + EXPECT_FALSE(core.closed()); + EXPECT_FALSE(done.load(std::memory_order_acquire)); + + core.ack(); + EXPECT_TRUE(wait_until( + [&] { + return done.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + caller.join(); + EXPECT_TRUE(core.closed()); +} diff --git a/tests/ut/cpp/a5/platform/test_return_gate_wait.cpp b/tests/ut/cpp/a5/platform/test_return_gate_wait.cpp new file mode 100644 index 0000000000..0ed1ec8c19 --- /dev/null +++ b/tests/ut/cpp/a5/platform/test_return_gate_wait.cpp @@ -0,0 +1,236 @@ +/* + * Copyright (c) PyPTO Contributors. + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of + * CANN Open Software License Agreement Version 2.0 (the "License"). + * Please refer to the License for details. You may not use this file except in compliance with the License. + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. + * See LICENSE in the root of the software repository for the full text of the License. + * ----------------------------------------------------------------------------------------------------------- + */ + +/** + * The post-close return gate has an AICore half and an AICPU half. The AICPU + * half is covered by test_aicore_retirement.cpp; this case covers the AICore + * half: the real a5sim startup EXIT and post-close waits, run on a host thread + * against a simulated register block and production gate word. The production + * retire path publishes EXIT, closes the window, and releases the gate. + * + * That wait has no timeout, so each case publishes the release before it joins + * and uses non-fatal assertions, so a failed expectation can never skip the + * cleanup and leave the worker spinning. + */ + +#include + +#include +#include +#include +#include +#include + +#include "aicpu/device_time.h" +#include "aicpu/platform_regs.h" +#include "common/memory_barrier.h" +#include "common/platform_config.h" +// inner_kernel.h needs RegId / reg_offset / sparse_reg_ptr from +// common/platform_config.h and sys_cnt_now_ticks from aicpu/device_time.h, so it +// is included after both rather than on its own. +#include "aicore/inner_kernel.h" + +namespace { + +// One sparse register block per core, laid out the way sparse_reg_ptr expects. +class CoreRegs { +public: + CoreRegs() { block_.fill(0); } + uint64_t addr() { return reinterpret_cast(block_.data()); } + volatile uint8_t *base() { return block_.data(); } + uint32_t dispatch() { return static_cast(read_reg(addr(), RegId::DATA_MAIN_BASE)); } + void ack() { write_reg(addr(), RegId::COND, AICORE_EXITED_VALUE); } + bool signalled() { return dispatch() == AICORE_EXIT_SIGNAL; } + bool closed() { return dispatch() == AICPU_IDLE_TASK_ID; } + +private: + alignas(64) std::array block_{}; +}; + +template +bool wait_until(Predicate predicate, std::chrono::milliseconds budget) { + const auto deadline = std::chrono::steady_clock::now() + budget; + while (!predicate()) { + if (std::chrono::steady_clock::now() >= deadline) return false; + std::this_thread::yield(); + } + return true; +} + +void publish_release(AicoreTeardownControl &gate) { + __atomic_store_n(&gate.post_close_release, AICORE_POST_CLOSE_RELEASE, __ATOMIC_RELEASE); +} + +} // namespace + +namespace { +thread_local volatile uint8_t *sim_reg_base = nullptr; +} + +volatile uint8_t *sim_get_reg_base() { return sim_reg_base; } + +// Only the AICPU's post-close publication ends the wait; the AICore's own exit +// write does not. +TEST(ReturnGateWait, RealSimWorkerReturnsOnlyAfterReleaseIsPublished) { + AicoreTeardownControl gate{}; + std::atomic returned{false}; + std::thread worker([&] { + wait_for_post_close_release(&gate.post_close_release); + returned.store(true, std::memory_order_release); + }); + + EXPECT_FALSE(wait_until( + [&] { + return returned.load(std::memory_order_acquire); + }, + std::chrono::milliseconds(50) + )); + + publish_release(gate); + EXPECT_TRUE(wait_until( + [&] { + return returned.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + + // Unconditional second publish and join: the wait has no timeout, so this is + // what keeps a failure above from leaving the thread spinning. + publish_release(gate); + worker.join(); +} + +// A stale release from the prior run cannot let a resident-startup failure ACK +// before AICPU publishes EXIT and resets this run's gate. +TEST(ReturnGateWait, StartupFailureWaitsForExitBeforeAckAndPostCloseReturn) { + CoreRegs core; + AicoreTeardownControl gate{}; + gate.post_close_release = AICORE_POST_CLOSE_RELEASE; + AicoreExitTarget target{core.addr(), &gate}; + std::atomic entered_exit_wait{false}; + std::atomic acked{false}; + std::atomic returned{false}; + std::atomic saw_closed_window{false}; + + std::thread worker([&] { + sim_reg_base = core.base(); + entered_exit_wait.store(true, std::memory_order_release); + wait_for_aicpu_exit_signal(); + write_reg(RegId::COND, AICORE_EXITED_VALUE); + acked.store(true, std::memory_order_release); + wait_for_post_close_release(&gate.post_close_release); + saw_closed_window.store(core.closed(), std::memory_order_release); + returned.store(true, std::memory_order_release); + }); + + EXPECT_TRUE(wait_until( + [&] { + return entered_exit_wait.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + std::this_thread::sleep_for(std::chrono::milliseconds(50)); + EXPECT_FALSE(acked.load(std::memory_order_acquire)); + EXPECT_FALSE(returned.load(std::memory_order_acquire)); + EXPECT_EQ(__atomic_load_n(&gate.post_close_release, __ATOMIC_ACQUIRE), AICORE_POST_CLOSE_RELEASE); + + // Model pre_handshake_init: reset the reused descriptor before the AICPU + // retirement path can publish this run's EXIT register signal. + __atomic_store_n(&gate.post_close_release, 0U, __ATOMIC_RELEASE); + wmb(); + int32_t rc = -1; + std::thread caller([&] { + rc = platform_retire_aicore_group(&target, 1, platform_aicore_exit_deadline()); + }); + caller.join(); + if (rc != 0 && !returned.load(std::memory_order_acquire)) { + platform_signal_aicore_exit(core.addr()); + wmb(); + core.ack(); + publish_release(gate); + } + EXPECT_EQ(rc, 0); + EXPECT_TRUE(acked.load(std::memory_order_acquire)); + EXPECT_TRUE(core.closed()); + EXPECT_TRUE(wait_until( + [&] { + return returned.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + + publish_release(gate); + worker.join(); + EXPECT_TRUE(saw_closed_window.load(std::memory_order_acquire)); +} + +// The production retire path publishes the gate, and it must do so only after +// the whole group has acknowledged and every window has closed. The real sim +// wait is the consumer of that publication, so this pins the two halves +// together. +TEST(ReturnGateWait, RealSimWorkerIsHeldUntilTheGroupAcknowledges) { + std::array cores; + std::array gates{}; + AicoreExitTarget targets[] = {{cores[0].addr(), &gates[0]}, {cores[1].addr(), &gates[1]}}; + std::atomic worker_returned{false}; + std::atomic worker_saw_closed_windows{false}; + + std::thread worker([&] { + wait_for_post_close_release(&gates[0].post_close_release); + const bool both_windows_closed = cores[0].closed() && cores[1].closed(); + worker_saw_closed_windows.store(both_windows_closed, std::memory_order_release); + worker_returned.store(true, std::memory_order_release); + }); + + // HBG's resident shutdown has already broadcast EXIT before retirement + // starts collecting acknowledgements and closing windows. + platform_signal_aicore_exit(cores[0].addr()); + platform_signal_aicore_exit(cores[1].addr()); + wmb(); + cores[0].ack(); + int32_t rc = -1; + std::thread caller([&] { + rc = platform_retire_aicore_group(targets, 2, platform_aicore_exit_deadline()); + }); + + // One pre-signalled core is still awaiting ACK. No target may publish its + // return gate while the group remains in the collection phase. + EXPECT_TRUE(wait_until( + [&] { + return cores[1].signalled(); + }, + std::chrono::milliseconds(500) + )); + // One core has not acknowledged, so no gate may be published yet and the + // worker must still be waiting. + EXPECT_FALSE(worker_returned.load(std::memory_order_acquire)); + EXPECT_EQ(__atomic_load_n(&gates[0].post_close_release, __ATOMIC_ACQUIRE), 0U); + EXPECT_EQ(__atomic_load_n(&gates[1].post_close_release, __ATOMIC_ACQUIRE), 0U); + + cores[1].ack(); + caller.join(); + EXPECT_EQ(rc, 0); + EXPECT_TRUE(cores[0].closed()); + EXPECT_TRUE(cores[1].closed()); + EXPECT_EQ(__atomic_load_n(&gates[0].post_close_release, __ATOMIC_ACQUIRE), AICORE_POST_CLOSE_RELEASE); + EXPECT_EQ(__atomic_load_n(&gates[1].post_close_release, __ATOMIC_ACQUIRE), AICORE_POST_CLOSE_RELEASE); + EXPECT_TRUE(wait_until( + [&] { + return worker_returned.load(std::memory_order_acquire); + }, + std::chrono::seconds(2) + )); + + publish_release(gates[0]); + publish_release(gates[1]); + worker.join(); + EXPECT_TRUE(worker_saw_closed_windows.load(std::memory_order_acquire)); +} diff --git a/tests/ut/cpp/a5/runtime/host_build_graph/CMakeLists.txt b/tests/ut/cpp/a5/runtime/host_build_graph/CMakeLists.txt index 445d1f7111..6ffbe52d45 100644 --- a/tests/ut/cpp/a5/runtime/host_build_graph/CMakeLists.txt +++ b/tests/ut/cpp/a5/runtime/host_build_graph/CMakeLists.txt @@ -21,6 +21,24 @@ a5_hbg_case(test_hbg_scheduler_bootstrap.cpp INCLUDES ${SIMPLER_SRC}/a5/runtime/host_build_graph/aicpu SOURCES ${HBG_COMMON_SHARED_DIR}/runtime.cpp) +# Exercise the HBG owners against a recording retirement boundary. The platform +# register tests separately drive the real ACK/close/read-back/drain sequence. +if(APPLE) + set(_retirement_gc_link_option -Wl,-dead_strip) +else() + set(_retirement_gc_link_option -Wl,--gc-sections) +endif() +a5_hbg_case(test_hbg_retirement_wiring.cpp + INCLUDES ${SIMPLER_SRC}/a5/runtime/host_build_graph/aicpu + SOURCES + ${SIMPLER_SRC}/a5/runtime/host_build_graph/aicpu/aicore_lifecycle.cpp + ${SIMPLER_SRC}/a5/runtime/host_build_graph/runtime/scheduler/scheduler_cold_path.cpp + ${HBG_COMMON_SHARED_DIR}/runtime.cpp + ${SIMPLER_SRC}/common/platform/shared/aicpu/chip_swimlane_collector_aicpu.cpp + ${SIMPLER_SRC}/a5/platform/shared/aicpu/pmu_collector_aicpu.cpp + COMPILE_OPTIONS -ffunction-sections -fdata-sections + LINK_OPTIONS ${_retirement_gc_link_option}) + # Drives the real orchestrator submit path. a5_hbg_case(test_hbg_submit_poison.cpp ORCH) diff --git a/tests/ut/cpp/a5/runtime/host_build_graph/test_hbg_retirement_wiring.cpp b/tests/ut/cpp/a5/runtime/host_build_graph/test_hbg_retirement_wiring.cpp new file mode 100644 index 0000000000..486ee7f51c --- /dev/null +++ b/tests/ut/cpp/a5/runtime/host_build_graph/test_hbg_retirement_wiring.cpp @@ -0,0 +1,208 @@ +/* + * Copyright (c) PyPTO Contributors. + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of + * CANN Open Software License Agreement Version 2.0 (the "License"). + * Please refer to the License for details. You may not use this file except in compliance with the License. + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. + * See LICENSE in the root of the software repository for the full text of the License. + * ----------------------------------------------------------------------------------------------------------- + */ + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "aicore_lifecycle.h" +#include "aicpu/platform_regs.h" +#include "runtime.h" +#include "scheduler/scheduler_context.h" + +// The register-layer tests exercise the real ACK, close, read-back, drain and +// release sequence. This seam verifies that both HBG owners pass the real per-core +// gate to that protocol, including their failure and overlapping-exit paths. +namespace { +struct RetireCall { + std::vector addrs; + std::vector gates; +}; + +std::vector retire_calls; +std::vector exit_signals; +int32_t retirement_result = 0; +std::mutex retire_mutex; +} // namespace + +uint64_t platform_aicore_exit_deadline() { return std::numeric_limits::max(); } + +int32_t platform_retire_aicore_group(const AicoreExitTarget *targets, size_t count, uint64_t, bool *released) { + RetireCall call; + for (size_t i = 0; i < count; ++i) { + call.addrs.push_back(targets[i].reg_addr); + call.gates.push_back(targets[i].teardown); + if (released != nullptr) released[i] = retirement_result == 0; + } + { + std::lock_guard lock(retire_mutex); + retire_calls.push_back(std::move(call)); + } + return retirement_result; +} + +void platform_signal_aicore_exit(uint64_t reg_addr) { exit_signals.push_back(reg_addr); } +void platform_init_aicore_regs(uint64_t) {} +void write_reg(uint64_t, RegId, uint64_t) {} +extern "C" bool is_dump_args_enabled() { return false; } + +class SchedulerContextTestPeer { +public: + static void seed(SchedulerContext &context, const std::array &addrs) { + context.cores_total_num_ = 3; + context.core_trackers_[0].init(1); + context.core_trackers_[0].set_cluster(0, 0, 1, 2); + for (int i = 0; i < 3; ++i) { + context.core_exec_states_[i].reg_addr = addrs[i]; + context.core_retired_[i].store(false, std::memory_order_relaxed); + } + } + static void emergency(SchedulerContext &context, Runtime *runtime) { context.emergency_shutdown(runtime); } +}; + +class AicoreLifecycleTestPeer { +public: + static void seed(AicoreLifecycle &lifecycle, const std::array &addrs) { + lifecycle.core_count_ = 3; + lifecycle.aicpu_thread_num_ = 1; + for (int i = 0; i < 3; ++i) + lifecycle.cores_[i].reg_addr = addrs[i]; + } +}; + +namespace { +constexpr std::array kAddrs = {0x1000, 0x2000, 0x3000}; + +class HbgRetirementWiring : public testing::Test { +protected: + void SetUp() override { + retire_calls.clear(); + exit_signals.clear(); + retirement_result = 0; + } + + static void expect_group(const Runtime &runtime) { + ASSERT_EQ(retire_calls.size(), 1u); + EXPECT_EQ(retire_calls[0].addrs, std::vector(kAddrs.begin(), kAddrs.end())); + for (size_t i = 0; i < kAddrs.size(); ++i) + EXPECT_EQ(retire_calls[0].gates[i], &runtime.dev.teardown_gates[i]); + } +}; + +TEST_F(HbgRetirementWiring, SchedulerNormalShutdownClaimsEachCoreBeforeEmergency) { + auto runtime = std::make_unique(); + auto scheduler = std::make_unique(); + SchedulerContextTestPeer::seed(*scheduler, kAddrs); + + EXPECT_EQ(scheduler->shutdown(runtime.get(), 0), 0); + SchedulerContextTestPeer::emergency(*scheduler, runtime.get()); + expect_group(*runtime); +} + +TEST_F(HbgRetirementWiring, SchedulerClearsStaleGatesBeforeHandshake) { + auto runtime = std::make_unique(); + runtime->set_worker_count(3); + for (auto &gate : runtime->dev.teardown_gates) + gate.post_close_release = AICORE_POST_CLOSE_RELEASE; + auto scheduler = std::make_unique(); + + ASSERT_EQ(scheduler->pre_handshake_init(runtime.get(), 1, 0), 0); + for (size_t i = 0; i < kAddrs.size(); ++i) + EXPECT_EQ(runtime->dev.teardown_gates[i].post_close_release, 0u); +} + +TEST_F(HbgRetirementWiring, SchedulerEmergencyClaimsEachCoreBeforeNormalShutdown) { + auto runtime = std::make_unique(); + auto scheduler = std::make_unique(); + SchedulerContextTestPeer::seed(*scheduler, kAddrs); + + SchedulerContextTestPeer::emergency(*scheduler, runtime.get()); + EXPECT_EQ(scheduler->shutdown(runtime.get(), 0), 0); + expect_group(*runtime); +} + +TEST_F(HbgRetirementWiring, ConcurrentNormalAndEmergencyShutdownRetireEachCoreOnce) { + auto runtime = std::make_unique(); + auto scheduler = std::make_unique(); + SchedulerContextTestPeer::seed(*scheduler, kAddrs); + std::atomic go{false}; + auto wait_for_start = [&] { + while (!go.load(std::memory_order_acquire)) + std::this_thread::yield(); + }; + std::thread normal([&] { + wait_for_start(); + EXPECT_EQ(scheduler->shutdown(runtime.get(), 0), 0); + }); + std::thread emergency([&] { + wait_for_start(); + SchedulerContextTestPeer::emergency(*scheduler, runtime.get()); + }); + go.store(true, std::memory_order_release); + normal.join(); + emergency.join(); + std::array claims{}; + size_t target_count = 0; + for (const RetireCall &call : retire_calls) { + for (size_t target = 0; target < call.addrs.size(); ++target) { + ++target_count; + for (size_t core = 0; core < kAddrs.size(); ++core) { + if (call.addrs[target] != kAddrs[core]) continue; + ++claims[core]; + EXPECT_EQ(call.gates[target], &runtime->dev.teardown_gates[core]); + } + } + } + EXPECT_EQ(target_count, kAddrs.size()); + EXPECT_EQ(claims, (std::array{1, 1, 1})); +} + +TEST_F(HbgRetirementWiring, LegacyShutdownSignalsAndRetiresTheSameGatedPartition) { + auto runtime = std::make_unique(); + AicoreLifecycle lifecycle; + AicoreLifecycleTestPeer::seed(lifecycle, kAddrs); + + lifecycle.signal_shutdown_partition(0); + EXPECT_EQ(exit_signals, std::vector(kAddrs.begin(), kAddrs.end())); + EXPECT_EQ(lifecycle.finish_shutdown_partition(0, runtime.get()), 0); + expect_group(*runtime); +} + +TEST_F(HbgRetirementWiring, LegacyClearsStaleGatesBeforeHandshake) { + auto runtime = std::make_unique(); + runtime->set_worker_count(3); + for (auto &gate : runtime->dev.teardown_gates) + gate.post_close_release = AICORE_POST_CLOSE_RELEASE; + AicoreLifecycle lifecycle; + + ASSERT_EQ(lifecycle.pre_handshake_init(runtime.get(), 1, 0), 0); + for (size_t i = 0; i < kAddrs.size(); ++i) + EXPECT_EQ(runtime->dev.teardown_gates[i].post_close_release, 0u); +} + +TEST_F(HbgRetirementWiring, FailedLegacyStartupStillRetiresThroughPerCoreGates) { + auto runtime = std::make_unique(); + AicoreLifecycle lifecycle; + AicoreLifecycleTestPeer::seed(lifecycle, kAddrs); + retirement_result = -1; + + EXPECT_EQ(lifecycle.release_partition(runtime.get(), 0, false), -1); + expect_group(*runtime); +} +} // namespace diff --git a/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/CMakeLists.txt b/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/CMakeLists.txt index 2b3d48c1e6..c4fbaf2989 100644 --- a/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/CMakeLists.txt +++ b/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/CMakeLists.txt @@ -17,6 +17,21 @@ macro(a5_tmr_case) tmr_case(${ARGV} ARCHS a5) endmacro() +# Exercise the production cold path with an observable platform retirement sink. +a5_tmr_case(test_scheduler_retirement.cpp + SOURCES + ${A5_RUNTIME_DIR}/scheduler/scheduler_cold_path.cpp + ${A5_RUNTIME_DIR}/shared/runtime.cpp + ${SIMPLER_SRC}/common/platform/shared/aicpu/aicpu_device_config.cpp + DEFINES SIMPLER_DFX=0 + COMPILE_OPTIONS -ffunction-sections -fdata-sections + TIMEOUT 20) +if(APPLE) + target_link_options(test_a5_tmr_scheduler_retirement PRIVATE -Wl,-dead_strip) +else() + target_link_options(test_a5_tmr_scheduler_retirement PRIVATE -Wl,--gc-sections) +endif() + # Everything not named above is a plain case: one file, this runtime's whole # object library, nothing else to say. Dropping a new test_*.cpp into this # directory is all it takes — declaring one by hand above is what removes it diff --git a/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/test_scheduler_retirement.cpp b/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/test_scheduler_retirement.cpp new file mode 100644 index 0000000000..01c7fb731b --- /dev/null +++ b/tests/ut/cpp/a5/runtime/tensormap_and_ringbuffer/test_scheduler_retirement.cpp @@ -0,0 +1,254 @@ +/* + * Copyright (c) PyPTO Contributors. + * This program is free software, you can redistribute it and/or modify it under the terms and conditions of + * CANN Open Software License Agreement Version 2.0 (the "License"). + * Please refer to the License for details. You may not use this file except in compliance with the License. + * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, + * INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. + * See LICENSE in the root of the software repository for the full text of the License. + * ----------------------------------------------------------------------------------------------------------- + */ + +#include + +#include +#include +#include +#include + +#include "runtime.h" +#include "scheduler/scheduler_context.h" + +namespace { +constexpr int kCores = 6; +std::array, kCores> retirements{}; +std::atomic retirement_batches{0}; +std::atomic deadlines{0}; +bool silent_core = false; +} // namespace + +uint64_t platform_aicore_exit_deadline() { return ++deadlines; } +uint64_t read_reg(uint64_t, RegId) { return AICORE_EXITED_VALUE; } +int32_t platform_retire_aicore_group(const AicoreExitTarget *targets, size_t count, uint64_t, bool *released) { + ++retirement_batches; + for (size_t i = 0; i < count; ++i) { + EXPECT_GE(targets[i].reg_addr, 1U); + EXPECT_LE(targets[i].reg_addr, kCores); + if (targets[i].reg_addr >= 1 && targets[i].reg_addr <= kCores) ++retirements[targets[i].reg_addr - 1]; + if (released != nullptr) released[i] = !(silent_core && targets[i].reg_addr == 1); + EXPECT_NE(targets[i].teardown, nullptr); + if (targets[i].teardown != nullptr && !(silent_core && targets[i].reg_addr == 1)) { + __atomic_store_n(&targets[i].teardown->post_close_release, AICORE_POST_CLOSE_RELEASE, __ATOMIC_RELAXED); + } + } + return silent_core ? -1 : 0; +} + +class SchedulerRetirementTestPeer { +public: + static void open_cores(SchedulerContext &context) { + for (int i = 0; i < kCores; ++i) + context.core_exec_states_[i].reg_addr = i + 1; + } + static void start_assignment(SchedulerContext &context, int owner) { context.core_trackers_[owner].init(1); } + static bool fatal(const SchedulerContext &context) { return context.fatal_shutdown_started_.load(); } + static void fail_handshake(SchedulerContext &context) { context.handshake_failed_.store(true); } + static void set_core_types(SchedulerContext &context) { + for (int i = 0; i < kCores; ++i) { + context.core_type_compact_[i] = static_cast(i < 2 ? CoreType::AIC : CoreType::AIV); + } + } +}; + +class SchedulerRetirement : public testing::Test { +protected: + std::unique_ptr context = std::make_unique(); + std::unique_ptr runtime = std::make_unique(); + + void SetUp() override { + retirement_batches = 0; + deadlines = 0; + silent_core = false; + for (auto &count : retirements) + count.store(0); + runtime->dev.worker_count = kCores; + runtime->dev.serial_orch_sched = false; + ASSERT_EQ(context->pre_handshake_init(runtime.get(), 3, 2, 0), 0); + } + void expect_once() { + for (int i = 0; i < kCores; ++i) + EXPECT_EQ(retirements[i].load(), 1) << "core " << i; + } +}; + +TEST_F(SchedulerRetirement, EmergencyUsesOneBatchAndDeadlineAcrossOwners) { + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + silent_core = true; + EXPECT_EQ(context->retire_all_cores(), -1); + expect_once(); + EXPECT_EQ(retirement_batches.load(), 1); + EXPECT_EQ(deadlines.load(), 1); +} + +TEST_F(SchedulerRetirement, OverlappingCoreSetsAreClaimedIndependently) { + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + const int32_t subset[] = {0, 3, 0}; + EXPECT_EQ(context->retire_cores(subset, 3), 0); + EXPECT_EQ(context->retire_all_cores(), 0); + expect_once(); + EXPECT_EQ(retirement_batches.load(), 2); +} + +TEST_F(SchedulerRetirement, NewGenerationClearsEveryReturnGateBeforePublication) { + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + ASSERT_EQ(context->retire_all_cores(), 0); + for (int i = 0; i < kCores; ++i) + ASSERT_EQ(runtime->dev.teardown_gates[i].post_close_release, AICORE_POST_CLOSE_RELEASE); + context->deinit(); + ASSERT_EQ(context->pre_handshake_init(runtime.get(), 3, 2, 0), 0); + for (int i = 0; i < kCores; ++i) + EXPECT_EQ(runtime->dev.teardown_gates[i].post_close_release, 0U); +} + +TEST_F(SchedulerRetirement, EmergencyBeforeInitializationRetiresLateCores) { + context->abort_and_shutdown(runtime.get()); + ASSERT_TRUE(context->is_completed()); + ASSERT_TRUE(SchedulerRetirementTestPeer::fatal(*context)); + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + // Failed init need not reach run()/shutdown(): assignment must honor the request. + expect_once(); + EXPECT_EQ(context->shutdown(0), 0); + EXPECT_EQ(context->shutdown(1), 0); + expect_once(); +} + +TEST_F(SchedulerRetirement, EmergencyDoesNotReadPartiallyAssignedTrackers) { + SchedulerRetirementTestPeer::open_cores(*context); + SchedulerRetirementTestPeer::start_assignment(*context, 1); + context->abort_and_shutdown(runtime.get()); + for (auto &count : retirements) + EXPECT_EQ(count.load(), 0); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + expect_once(); +} + +TEST_F(SchedulerRetirement, NormalAndEmergencyRetireEachReadyCoreOnce) { + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + std::thread normal([&] { + context->shutdown(0); + }); + std::thread emergency([&] { + context->abort_and_shutdown(runtime.get()); + }); + context->shutdown(1); + normal.join(); + emergency.join(); + expect_once(); +} + +TEST_F(SchedulerRetirement, AssignmentAndEmergencyCanPublishConcurrently) { + SchedulerRetirementTestPeer::open_cores(*context); + std::thread first([&] { + context->assign_own_clusters(0); + }); + std::thread second([&] { + context->assign_own_clusters(1); + }); + context->abort_and_shutdown(runtime.get()); + first.join(); + second.join(); + expect_once(); +} + +TEST_F(SchedulerRetirement, SerialHandshakeFailureUsesOneFallbackOwner) { + runtime->dev.serial_orch_sched = true; + ASSERT_EQ(context->pre_handshake_init(runtime.get(), 3, 2, 0), 0); + SchedulerRetirementTestPeer::open_cores(*context); + SchedulerRetirementTestPeer::fail_handshake(*context); + EXPECT_EQ(context->post_handshake_init(runtime.get()), -1); + expect_once(); + context->retire_all_cores(); + expect_once(); +} + +TEST_F(SchedulerRetirement, SerialInitializationPublishesAssignedGroups) { + runtime->dev.serial_orch_sched = true; + ASSERT_EQ(context->pre_handshake_init(runtime.get(), 3, 2, 0), 0); + SchedulerRetirementTestPeer::open_cores(*context); + SchedulerRetirementTestPeer::set_core_types(*context); + ASSERT_EQ(context->post_handshake_init(runtime.get()), 0); + context->shutdown(0); + context->shutdown(1); + context->shutdown(2); + expect_once(); +} + +TEST_F(SchedulerRetirement, NewGenerationDoesNotInheritPendingRetirement) { + context->abort_and_shutdown(runtime.get()); + context->deinit(); + ASSERT_EQ(context->pre_handshake_init(runtime.get(), 3, 2, 0), 0); + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + for (auto &count : retirements) + EXPECT_EQ(count.load(), 0); + EXPECT_FALSE(context->is_completed()); + EXPECT_FALSE(SchedulerRetirementTestPeer::fatal(*context)); + context->shutdown(0); + context->shutdown(1); + expect_once(); +} + +TEST_F(SchedulerRetirement, CompletionObserverSeesFatalPublication) { + std::thread emergency([&] { + context->abort_and_shutdown(runtime.get()); + }); + while (!context->is_completed()) {} + EXPECT_TRUE(SchedulerRetirementTestPeer::fatal(*context)); + emergency.join(); +} + +// Claim is per core, not per caller: concurrent calls naming arbitrary +// overlapping sets must each win only the cores no other call has claimed. A +// per-caller claim lets two overlapping callers both retire a shared core, which +// shows up below as a count above 1 for that core. +TEST_F(SchedulerRetirement, ConcurrentOverlappingSetsClaimEachCoreOnce) { + SchedulerRetirementTestPeer::open_cores(*context); + context->assign_own_clusters(0); + context->assign_own_clusters(1); + + const int32_t sets[][6] = { + {0, 1, 2, 3, 4, 5}, {5, 4, 3, 2, 1, 0}, {0, 2, 4, 0, 2, 4}, {1, 3, 5, 1, 3, 5}, {2, 3, 0, 5, 4, 1}, + }; + constexpr int kCallers = static_cast(sizeof(sets) / sizeof(sets[0])); + + // A start line, so the calls contend on the per-core claim instead of being + // serialized by thread creation order. + std::atomic go{false}; + std::array callers; + for (int t = 0; t < kCallers; ++t) { + callers[t] = std::thread([&, t] { + while (!go.load(std::memory_order_acquire)) + std::this_thread::yield(); + context->retire_cores(sets[t], 6); + }); + } + go.store(true, std::memory_order_release); + for (auto &caller : callers) + caller.join(); + + // Exactly once per core is the property: a bare total would hide a double + // claim offset by a miss. + expect_once(); +} diff --git a/tests/ut/cpp/common/host_build_graph/support/stall_dump_level_a5_stubs.cpp b/tests/ut/cpp/common/host_build_graph/support/stall_dump_level_a5_stubs.cpp index b5a5c5949a..a38cc2ed78 100644 --- a/tests/ut/cpp/common/host_build_graph/support/stall_dump_level_a5_stubs.cpp +++ b/tests/ut/cpp/common/host_build_graph/support/stall_dump_level_a5_stubs.cpp @@ -13,8 +13,9 @@ #include "aicpu/platform_regs.h" -// a5 retires one core at a time through its own register entry, where a2a3 retires a -// group against a deadline — so the emergency-shutdown half the cold path carries -// needs a different stub per architecture. The stall-dump tests never reach it; only -// the link does. -int32_t __attribute__((weak)) platform_deinit_aicore_regs(uint64_t) { return 0; } +// The stall-dump tests never reach retirement; this stub only satisfies the link. +uint64_t __attribute__((weak)) platform_aicore_exit_deadline() { return 0; } + +int32_t __attribute__((weak)) platform_retire_aicore_group(const AicoreExitTarget *, size_t, uint64_t, bool *) { + return 0; +}