Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 37 additions & 17 deletions src/a5/runtime/host_build_graph/aicore/aicore_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@
#include "aicore/aicore.h"
#include "aicore/aicore_profiling_state.h"
// AICore dependency scheduling uses one device-side protocol.
#include "scheduler/scheduler_dispatch.h"
#include "scheduler/scheduler_mix.h"
#include "common/platform_config.h"
#include "dispatch_payload.h"
#include "runtime.h"
Expand Down Expand Up @@ -149,7 +149,7 @@ __aicore__ bool bootstrap_ready_graph(
return false;
const bool phase_timing_enabled = scheduler_phase_timing_enabled(profiling_level);
if (phase_timing_enabled) stats->bootstrap_start_cycles = scheduler_cycles();
SchedulerReadyBatch batches[SCHEDULER_CORE_TYPE_COUNT]{};
SchedulerReadyBatch batches[SCHEDULER_READY_QUEUE_COUNT]{};
uint64_t tasks_per_scheduler = graph.task_count / scheduler_count;
uint64_t remainder = graph.task_count % scheduler_count;
uint64_t task_begin =
Expand All @@ -170,8 +170,7 @@ __aicore__ bool bootstrap_ready_graph(
SchedulerRouteResult::READY_TO_ENQUEUE;
if (route == SchedulerRouteResult::ERROR) return false;
if (route == SchedulerRouteResult::READY_TO_ENQUEUE) {
const uint32_t core_type =
scheduler_metadata_core_type_index(scheduler_metadata_single_subtask_slot(metadata->active_mask));
const uint32_t core_type = scheduler_task_ready_queue(metadata->flags, metadata->active_mask);
if (!scheduler_bootstrap_ready_batch_append(
scheduler_state_base, scheduler, static_cast<int64_t>(task_id), &batches[core_type],
phase_timing_enabled ? &stats->ready : nullptr, profiling_level
Expand All @@ -182,7 +181,7 @@ __aicore__ bool bootstrap_ready_graph(
}
scheduler_cache_barrier();
uint64_t ready_types = 0;
for (uint32_t type = 0; type < SCHEDULER_CORE_TYPE_COUNT; ++type) {
for (uint32_t type = 0; type < SCHEDULER_READY_QUEUE_COUNT; ++type) {
if (!scheduler_bootstrap_ready_batch_publish(
scheduler_state_base, scheduler, type, scheduler->config.scheduler_index, &batches[type],
phase_timing_enabled ? &stats->ready : nullptr, &ready_types
Expand Down Expand Up @@ -233,11 +232,16 @@ __aicore__ bool bootstrap_ready_graph(

// Prepare the first executable wave while the sole DMB launch gate is
// still closed.
uint64_t ready_victim_cursors[SCHEDULER_CORE_TYPE_COUNT]{
uint64_t ready_victim_cursors[SCHEDULER_READY_QUEUE_COUNT]{
(scheduler->config.scheduler_index + 1) % scheduler_count,
(scheduler->config.scheduler_index + 1) % scheduler_count,
};
bool fill_failed = false;
if (!scheduler_fill_cluster_mix_slots(
graph, scheduler_state_base, scheduler, run_control, &ready_victim_cursors[SCHEDULER_MIX_QUEUE],
phase_timing_enabled ? &stats->ready : nullptr, profiling_level, true, ssbuf_region, nullptr
))
return false;
(void)scheduler_fill_cluster_normal_slots(
graph, scheduler_state_base, scheduler, run_control, ready_victim_cursors,
phase_timing_enabled ? &stats->ready : nullptr, profiling_level, 0, deferred_aiv, &fill_failed, ssbuf_region
Expand Down Expand Up @@ -276,13 +280,13 @@ __aicore__ bool run_ready_dispatch_loop(
bool scheduler_worker = context->is_scheduler();
if (scheduler_count == 0) return false;
if (ssbuf_region == nullptr || cluster_lane >= PLATFORM_CORES_PER_BLOCKDIM) return false;
uint64_t ready_victim_cursors[SCHEDULER_CORE_TYPE_COUNT]{
uint64_t ready_victim_cursors[SCHEDULER_READY_QUEUE_COUNT]{
scheduler_worker ? (context->config.scheduler_index + 1) % scheduler_count : 0,
scheduler_worker ? (context->config.scheduler_index + 1) % scheduler_count : 0,
};
uint32_t seen_generation[SCHEDULER_PENDING_SLOT_COUNT]{};
uint64_t completion_publication = 0;
uint32_t scan_start = 0;
uint64_t next_dispatch_sequence = 1;
uint32_t backoff_iterations = kInitialBackoffIterations;
uint32_t scheduler_error_poll_count = 0;
if (phase_timing_enabled) context->profiling->loop_iter = UINT32_MAX;
Expand Down Expand Up @@ -317,7 +321,7 @@ __aicore__ bool run_ready_dispatch_loop(
return false;
}
uint32_t published_local_slot = UINT32_MAX;
if (scheduler_worker && deferred_aiv != nullptr && deferred_aiv->count != 0) {
if (scheduler_worker && !context->has_mix && deferred_aiv != nullptr && deferred_aiv->count != 0) {
const uint32_t deferred_before = deferred_aiv->count;
if (!scheduler_drain_deferred_aiv_to_peer(
graph, scheduler_state_base, context, run_control, deferred_aiv,
Expand All @@ -334,12 +338,26 @@ __aicore__ bool run_ready_dispatch_loop(
}
if (scheduler_worker && published_local_slot == UINT32_MAX) {
uint64_t direct_refilled_slot_mask = 0;
SchedulerRefillCandidates refill_candidates(context->has_mix);
const bool completion_progress = scheduler_service_cluster_completions(
graph, scheduler_state_base, context, run_control, phase_timing_enabled ? &stats->wake : nullptr,
phase_timing_enabled ? &stats->ready : nullptr, phase_timing_enabled ? &stats->completion : nullptr,
ready_victim_cursors, profiling_level, &direct_refilled_slot_mask, ssbuf_region
ready_victim_cursors, profiling_level, &direct_refilled_slot_mask, ssbuf_region,
context->has_mix ? &refill_candidates : nullptr
);
scheduler_progress = completion_progress;
if (!scheduler_fill_mix_after_completions(
graph, scheduler_state_base, context, run_control, &refill_candidates.mix_ready,
&ready_victim_cursors[SCHEDULER_MIX_QUEUE], phase_timing_enabled ? &stats->ready : nullptr,
profiling_level, deferred_aiv == nullptr || deferred_aiv->count == 0, ssbuf_region,
&scheduler_progress
))
return false;
if (context->has_mix && !scheduler_publish_refill_candidates(
graph, scheduler_state_base, context, run_control, &refill_candidates,
phase_timing_enabled ? &stats->ready : nullptr, profiling_level, ssbuf_region
))
return false;
bool fill_failed = false;
const bool dispatch_progress = scheduler_fill_cluster_normal_slots(
graph, scheduler_state_base, context, run_control, ready_victim_cursors,
Expand Down Expand Up @@ -377,10 +395,10 @@ __aicore__ bool run_ready_dispatch_loop(
if (scheduler_worker) {
uint32_t local_slot = UINT32_MAX;
uint64_t local_generation = 0;
if (scheduler_local_ready_pop(context, scan_start, &local_slot, &local_generation)) {
if (scheduler_local_ready_pop(context, &local_slot, &local_generation)) {
if (scheduler_dispatch_state(local_generation) != SchedulerDispatchSlotState::READY ||
scheduler_dispatch_generation(local_generation) == 0 ||
scheduler_dispatch_generation(local_generation) == seen_generation[local_slot]) {
scheduler_dispatch_generation(local_generation) != next_dispatch_sequence) {
scheduler_record_error(
run_control, context->slots[cluster_lane][local_slot].task_id,
SchedulerGraphResult::INVALID_ARGUMENTS, &graph, context,
Expand All @@ -398,7 +416,7 @@ __aicore__ bool run_ready_dispatch_loop(
scheduler_ssbuf_load_relaxed(&ssbuf_region->lanes[cluster_lane].dispatch[slot_index].publication);
const uint32_t generation = static_cast<uint32_t>(publication);
if (phase_timing_enabled) ++stats->task_state_poll_count;
if (generation != 0 && generation != seen_generation[slot_index]) {
if (generation == next_dispatch_sequence) {
ready_slot = static_cast<int32_t>(slot_index);
ready_generation = generation;
ready_publication = publication;
Expand All @@ -424,9 +442,11 @@ __aicore__ bool run_ready_dispatch_loop(
scheduler_append_idle_activity(scheduler_state_base, context, idle_start_cycles, ready_observe);
idle_active = false;
}
scheduler_observe_dispatch_payload_control(payload);
scheduler_observe_dispatch_payload_arguments(payload);
scheduler_observe_dispatch_payload_barrier();
if (!scheduler_worker) {
scheduler_observe_dispatch_payload_control(payload);
scheduler_observe_dispatch_payload_arguments(payload);
scheduler_observe_dispatch_payload_barrier();
}
const int64_t task_id = scheduler_worker ? local_slot->task_id : ssbuf_control->task_id;
if (task_id < 0 || static_cast<uint64_t>(task_id) >= graph.task_count || ready_generation == 0) {
scheduler_record_error(
Expand All @@ -435,7 +455,7 @@ __aicore__ bool run_ready_dispatch_loop(
);
return false;
}
seen_generation[slot_index] = ready_generation;
++next_dispatch_sequence;
SchedulerExecutorTaskTrace execution_trace{};
if (commit_scheduler_trace || commit_task_timing) {
uint64_t local_completion_index = stats->completion.enqueue_count;
Expand Down
19 changes: 13 additions & 6 deletions src/a5/runtime/host_build_graph/aicpu/aicpu_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -179,16 +179,23 @@ static void publish_aicore_task_timing(Runtime *runtime) {
auto *metadata = scheduler_state_at<SchedulerTaskMetadata>(scheduler_state_base, context->task_metadata_offset);
auto *traces = scheduler_state_at<SchedulerTaskTrace>(scheduler_state_base, context->trace_cells_offset);
cache_invalidate_range(metadata, static_cast<size_t>(context->graph_task_count) * sizeof(*metadata));
cache_invalidate_range(traces, static_cast<size_t>(context->graph_task_count) * sizeof(*traces));

for (uint64_t task_id = 0; task_id < context->graph_task_count; ++task_id) {
const int32_t slot = metadata[task_id].timing_slot;
if (slot < 0 || slot >= NUM_TASK_TIMING_SLOTS) continue;
const uint64_t start = traces[task_id].kernel_start_cycles;
const uint64_t end = traces[task_id].kernel_end_cycles;
if (start == 0 || end <= start) continue;
if (start < records[slot].dispatch_cycle) records[slot].dispatch_cycle = start;
if (end > records[slot].finish_cycle) records[slot].finish_cycle = end;
if (!scheduler_task_is_executable(metadata[task_id].flags)) continue;
for (uint8_t subtask = 0; subtask < 3; ++subtask) {
if ((metadata[task_id].active_mask & (1U << subtask)) == 0) continue;
auto &trace = traces[scheduler_task_trace_index(
metadata[task_id].trace_index_base, metadata[task_id].active_mask, subtask
)];
cache_invalidate_range(&trace, sizeof(trace));
const uint64_t start = trace.kernel_start_cycles;
const uint64_t end = trace.kernel_end_cycles;
if (start == 0 || end <= start) continue;
if (start < records[slot].dispatch_cycle) records[slot].dispatch_cycle = start;
if (end > records[slot].finish_cycle) records[slot].finish_cycle = end;
}
}
aicpu_publish_task_timing_tail_usage(1);
}
Expand Down
6 changes: 3 additions & 3 deletions src/a5/runtime/host_build_graph/common/intrinsic.h
Original file line number Diff line number Diff line change
Expand Up @@ -62,8 +62,8 @@
* issue #900 (PR #899 spmd_paged_attention_highperf); the kernel
* compiled, ran without error, and produced wrong output. Use
* `get_sub_block_id(args)` instead, which reads from the runtime's
* `GlobalContext.sub_block_id`. Resident graph materialization sets it
* from the selected AIV task subslot before every dispatch publication.
* `GlobalContext.sub_block_id`. Resident Mix dispatch sets it
* from the final physical AIV placement before publication.
*
* - `get_block_idx()` and `get_block_num()` are not redirected to
* simpler's LocalContext either — use the `(args)` variants below
Expand Down Expand Up @@ -107,7 +107,7 @@ static constexpr int32_t PAYLOAD_GLOBAL_CONTEXT_INDEX = SPMD_GLOBAL_CONTEXT_INDE

/**
* Per-dispatch global context, stored in DispatchPayload. Resident graph
* materialization sets sub_block_id from the selected AIV task subslot; the
* Mix dispatch sets sub_block_id from the final physical AIV placement; the
* legacy scheduler seeds the equivalent per-core value during startup.
*/
struct GlobalContext {
Expand Down
41 changes: 39 additions & 2 deletions src/a5/runtime/host_build_graph/docs/RUNTIME_LOGIC.md
Original file line number Diff line number Diff line change
Expand Up @@ -420,7 +420,7 @@ READY with the same generation until the local Executor claims it, so the ready
token is reconstructed from the slot. Completion generation validation still
prevents stale notifications from freeing or refilling a pending slot.

The local configuration occupies 88 bytes and the base local state 256 bytes
The local configuration occupies 88 bytes and the base local state 376 bytes
under the 64-bit ABI, including its optional profiling pointer. A separate
240-byte profiling state holds six timing slots, two self-execution traces,
worker trace caches, profiling offsets, and the loop counter/valid mask.
Expand All @@ -430,7 +430,7 @@ The host records whether any task requests sampled timing in the run control.
Only a run with chip profiling or sampled timing enters the resident function
specialization that allocates profiling state; the plain specialization does
not allocate it. These functions do not inline into the common entry. The
combined local state with profiling is 496 bytes. These sizes exclude other
combined local state with profiling is 616 bytes. These sizes exclude other
function locals, worker statistics and compiler spills. Compile-time assertions
anchor the 64-bit configuration, slot, base and profiling state sizes; they do
not establish the dynamic AICore stack high-water mark.
Expand All @@ -453,6 +453,10 @@ writeback and ready publication share a release barrier. Tokens are SPSC and
use no SSBUF read-modify-write atomics. The Scheduler initializes every token
before publishing the header, and each invocation validates the region.
Self-execution uses local notifications, completion generations and trace storage.
Each slot's embedded context pointers and constant single-block fields are
initialized in the host-created resident payload image. Self-execution consumes
its own payload writes after the publication barrier, without an additional
payload invalidation on pickup. Remote Executors retain payload invalidation.

The Scheduler and Executor share the SSBUF structure definitions in the same
runtime build. Each run initializes the region before use.
Expand Down Expand Up @@ -488,6 +492,39 @@ therefore avoids periodic publication on the busy path; the continuous-busy
timeout is an accepted limit outside that expected duration, not evidence that
busy execution has stopped making progress.

Single-block ordinary Mix uses the resident path. Each participating lane
owns an independent GM DispatchPayload and one slot. A local six-entry tracker
joins completions: each lane releases its slot immediately; only the final lane
marks the task done and resolves dependencies. Kernels implement their own
cross-lane synchronization. There is no cluster waiting-count gate or ACK state.

Ordinary and Mix dispatches share a per-lane sequence in the publication word.
Preparation reserves resources without consuming a sequence; publication commits
it after the final lane is known. Executors consume dispatch order across both
slots, including Scheduler self-execution. Mix prepares every participating lane
before publishing any lane, and finishes the group's publications before the
next Mix. Published ordinary tasks retain their position.
One local metadata snapshot and one argument-region parse serve all lanes of a
Mix. Shared scalar values and tensor descriptor addresses are read or computed
once, then written into each lane's independent payload. After all payloads
are written back, one barrier precedes their individual READY publications.

Mix admission checks all participating slots and a tracker before claiming the
ready inbox head. With ordinary deferred reservations, local Mix may be claimed
but remote Mix cannot be stolen. Mix never enters the deferred queue. On graphs
containing Mix, a scheduling pass consumes completions, attempts the unlocked
direct Mix candidate, admits queued Mix, and then refills ordinary work.
Displaced direct successors return to ready queues;
ordinary-only graphs keep immediate direct refill. Task traces use metadata's
32-bit base index plus the logical subtask offset, independently of placement.
An unlocked Mix may remain a candidate within the current scheduling pass when
no Mix work was queued at its ready-queue snapshot. Complete lane and tracker
capacity permit direct preparation before the next queue claim; otherwise the
candidate joins its ready queue first. Queue arrivals between that snapshot
and a failed placement can precede the candidate after it joins the queue.
Phase profiling records ready transitions for every active subtask. With
profiling disabled, refill candidates do not compute trace indices.

Directory queries are skipped when no unreserved FREE slot can accept
work. Idle polling backs off from 8 to at most 32 iterations. Dispatch trace
GM writes follow ready publication; Host tooling computes ready-to-kernel
Expand Down
Loading
Loading