Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 1 addition & 4 deletions .github/workflows/codeql.yml
Original file line number Diff line number Diff line change
Expand Up @@ -30,11 +30,8 @@ jobs:
- name: Install OS requirements
timeout-minutes: 10
run: |
apt update
apt install -y software-properties-common
apt-add-repository -y ppa:git-core/ppa
apt-get update
apt install -y git
DEBIAN_FRONTEND=noninteractive apt-get install -y git

- name: Checkout repository
uses: actions/checkout@v4
Expand Down
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,14 @@
Documentation for TransferBench is available at
[https://rocm.docs.amd.com/projects/TransferBench](https://rocm.docs.amd.com/projects/TransferBench).

## v1.70.02
### Modified
- rings preset defaults `NUM_SUB_EXEC` to 0, which uses all available subexecutors per Transfer
- smoketest disables broadcast, gather, and all-to-all by default on multi-node. Set `TEST_LIST` to enable them

## v1.70.01
Skipped due to misnaming of 1.70.00

## v1.70.00
### Added
- Adding support for Tensor Data Mover (TDM)-based executor [T] on supported hardware (gfx1250, NVIDIA sm_90+ via TMA).
Expand Down
2 changes: 1 addition & 1 deletion CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -104,7 +104,7 @@ set(ENV{ROCM_PATH} "${ROCM_PATH}")
#==================================================================================================
set(TRANSFERBENCH_VERSION_MAJOR 1)
set(TRANSFERBENCH_VERSION_MINOR 70)
set(TRANSFERBENCH_VERSION_PATCH_FALLBACK "00")
set(TRANSFERBENCH_VERSION_PATCH_FALLBACK "02")

# Auto-compute patch from git: count commits since the last v<MAJOR>.<MINOR>.* tag.
# Falls back to TRANSFERBENCH_VERSION_PATCH_FALLBACK when git is unavailable,
Expand Down
6 changes: 3 additions & 3 deletions docs/reference/presets.rst
Original file line number Diff line number Diff line change
Expand Up @@ -518,8 +518,8 @@ To modify the behavior of the rings preset, use the following environment variab
- ``0``

* - ``NUM_SUB_EXEC``
- SubExecutors (CUs for GFX, batch items for DMA) per transfer.
- ``8``
- SubExecutors (CUs for GFX, batch items for DMA) per transfer. ``0`` uses all available subexecutors on the Executor.
- ``0``

* - ``RING_SIZE``
- Number of GPUs per ring. Must evenly divide ``numRanks x NUM_GPU_DEVICES``. Default is one ring covering the whole pod.
Expand Down Expand Up @@ -550,7 +550,7 @@ Example output
MEM_TYPE = 0 : Using default GPU memory (0=default, 1=fine-grained, 2=uncached, 3=managed)
NUM_GPU_DEVICES = 4 : Using 4 GPUs
NUM_QUEUE_PAIRS = 0 : Using 0 queue pairs for NIC transfers
NUM_SUB_EXEC = 8 : Using 8 subExecutors/CUs per Transfer
NUM_SUB_EXEC = 0 : Using all available subexecutors/CUs per Transfer
USE_DMA_EXEC = 0 : Using GFX executor
USE_REMOTE_READ = 0 : Using SRC as executor
STRIDE = 1 : Reordering devices by taking 1 steps
Expand Down
2 changes: 1 addition & 1 deletion src/client/EnvVars.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,7 @@ THE SOFTWARE.
#include <random>
#include <time.h>

#define CLIENT_VERSION "01"
#define CLIENT_VERSION "02"

#include "TransferBench.hpp"
using namespace TransferBench;
Expand Down
40 changes: 30 additions & 10 deletions src/client/Presets/Rings.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@ int RingsPreset(EnvVars& ev,
int memTypeIdx = EnvVars::GetEnvVar("MEM_TYPE" , 0);
int numGpus = EnvVars::GetEnvVar("NUM_GPU_DEVICES", numDetectedGpus);
int numQueuePairs = EnvVars::GetEnvVar("NUM_QUEUE_PAIRS", 0);
int numSubExecs = EnvVars::GetEnvVar("NUM_SUB_EXEC" , 8);
int numSubExecs = EnvVars::GetEnvVar("NUM_SUB_EXEC" , 0);
int showDetails = EnvVars::GetEnvVar("SHOW_DETAILS" , 0);
int useDmaExec = EnvVars::GetEnvVar("USE_DMA_EXEC" , 0);
int useTdmExec = EnvVars::GetEnvVar("USE_TDM_EXEC" , 0);
Expand Down Expand Up @@ -78,6 +78,10 @@ int RingsPreset(EnvVars& ev,
Utils::Print("[ERROR] Num queue pairs must be non-negative\n");
return ERR_FATAL;
}
if (numSubExecs < 0) {
Utils::Print("[ERROR] NUM_SUB_EXEC must be non-negative (0 uses all available subexecutors)\n");
return ERR_FATAL;
}

int totalGpus = numRanks * numGpus;
if (totalGpus % ringSize) {
Expand All @@ -95,7 +99,10 @@ int RingsPreset(EnvVars& ev,
ev.Print("MEM_TYPE" , memTypeIdx , "Using %s GPU memory (%s)", devMemTypeStr.c_str(), Utils::GetAllGpuMemTypeStr().c_str());
ev.Print("NUM_GPU_DEVICES", numGpus , "Using %d GPUs", numGpus);
ev.Print("NUM_QUEUE_PAIRS", numQueuePairs, "Using %d queue pairs for NIC transfers", numQueuePairs);
ev.Print("NUM_SUB_EXEC" , numSubExecs , "Using %d subexecutors/CUs per Transfer", numSubExecs);
if (numSubExecs == 0)
ev.Print("NUM_SUB_EXEC" , numSubExecs, "Using all available subexecutors/CUs per Transfer");
else
ev.Print("NUM_SUB_EXEC" , numSubExecs, "Using %d subexecutors/CUs per Transfer", numSubExecs);
ev.Print("USE_DMA_EXEC" , useDmaExec , "Using %s executor", useDmaExec ? "DMA" : "GFX");
ev.Print("USE_TDM_EXEC" , useTdmExec , "Using %s executor", useTdmExec ? "TDM" : "GFX");
ev.Print("USE_REMOTE_READ", useRemoteRead, "Using %s as executor", useRemoteRead ? "DST" : "SRC");
Expand All @@ -105,16 +112,9 @@ int RingsPreset(EnvVars& ev,
}
}

Utils::Print("GPU-%s Rings benchmark:\n", execName);
Utils::Print("==============================\n");
Utils::Print("[%lu bytes per Transfer] [%s:%d] [MemType:%s] [NIC QueuePairs:%d] [#Ranks:%d]\n",
numBytesPerTransfer, execName, numSubExecs,
devMemTypeStr.c_str(), numQueuePairs, numRanks);

TransferBench::ConfigOptions cfg = ev.ToConfigOptions();

int numRings = totalGpus / ringSize;
Utils::Print("Running %d parallel ring(s) each of %d devices. All numbers in GB/s:\n", numRings, ringSize);

// Determine ordering of GPUs for the rings based on stride
std::vector<int> indices(totalGpus);
Expand Down Expand Up @@ -143,7 +143,7 @@ int RingsPreset(EnvVars& ev,
t.srcs = {memDevices[srcIdx]};
t.dsts = {memDevices[dstIdx]};
t.exeDevice = {exeType, memDevices[exeIdx].memIndex, memDevices[exeIdx].memRank};
t.numSubExecs = numSubExecs;
t.numSubExecs = (numSubExecs > 0) ? numSubExecs : TransferBench::GetNumSubExecutors(t.exeDevice);
transfers.push_back(t);

// Build NIC transfers between these GPUs as well if requested
Expand All @@ -157,6 +157,26 @@ int RingsPreset(EnvVars& ev,
}
}

// NUM_SUB_EXEC=0: each hop already took its own executor's count.
// Report it in the header only if every selected GPU executor agrees, otherwise "var"
std::string seStr = std::to_string(numSubExecs);
if (numSubExecs == 0) {
int seCount = 0;
for (auto const& t : transfers) {
if (t.exeDevice.exeType != exeType) continue; // skip NIC overlays
if (seCount == 0) seCount = t.numSubExecs;
else if (t.numSubExecs != seCount) { seCount = -1; break; }
}
seStr = (seCount < 0) ? "variable" : std::to_string(seCount);
}

Utils::Print("GPU-%s Rings benchmark:\n", execName);
Utils::Print("==============================\n");
Utils::Print("[%lu bytes per Transfer] [%s:%s] [MemType:%s] [NIC QueuePairs:%d] [#Ranks:%d]\n",
numBytesPerTransfer, execName, seStr.c_str(),
devMemTypeStr.c_str(), numQueuePairs, numRanks);
Utils::Print("Running %d parallel ring(s) each of %d devices. All numbers in GB/s:\n", numRings, ringSize);

TransferBench::TestResults results;
if (!TransferBench::RunTransfers(cfg, transfers, results)) {
for (auto const& err : results.errResults)
Expand Down
20 changes: 18 additions & 2 deletions src/client/Presets/SmokeTest.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -269,9 +269,17 @@ int SmokeTestPreset(EnvVars& ev,
MemType cpuMemType = Utils::GetCpuMemType(cpuMemTypeIdx);
MemType gpuMemType = Utils::GetGpuMemType(gpuMemTypeIdx);
std::set<int> testsToRun(testList.begin(), testList.end());
bool skipCollectiveByDefault = false;
if (testList.empty()) {
for (int testIdx = 1; testIdx <= NUM_SMOKE_TESTS; testIdx++)
testsToRun.insert(testIdx);
// Multi-node default: H2D/D2H/D2D only.
// Broadcast, gather, and all-to-all remain available via TEST_LIST
Comment thread
AtlantaPepsi marked this conversation as resolved.
if (numRanks > 1) {
skipCollectiveByDefault = true;
for (int t : {5, 6, 7, 8, 13, 14, 15, 16})
testsToRun.erase(t);
}
}

vector<size_t> sizeList;
Expand Down Expand Up @@ -312,8 +320,13 @@ int SmokeTestPreset(EnvVars& ev,
ev.Print("RUN_PARALLEL", runParallel, "Running GPUs %s", runParallel ? "in parallel" : "serially");
ev.Print("SIZE_LIST" , sizeStrList.size(), "Transfer sizes tested: %s", ev.GetStr(sizeStrList).c_str());
ev.Print("SE_MAX_BYTES", seMaxBytesStr, "Each SubExecutor can work on at most %lu bytes", seMaxBytes);
ev.Print("TEST_LIST" , testsToRun.size(), testList.empty() ? "Running all tests (can also filter with 'dma','gfx','fast')"
: "Running Tests: %s", ev.GetStr(testList).c_str());
if (!testList.empty())
ev.Print("TEST_LIST", testsToRun.size(), "Running Tests: %s", ev.GetStr(testList).c_str());
else if (skipCollectiveByDefault)
ev.Print("TEST_LIST", testsToRun.size(),
"Running H2D/D2H/D2D only (broadcast/gather/a2a off by default on multi-node; set TEST_LIST to enable)");
else
ev.Print("TEST_LIST", testsToRun.size(), "Running all tests (can also filter with 'dma','gfx','fast')");
ev.Print("USE_BDMA" , useBdma, "Using %s dma executor", useBdma ? "batched" : "standard");
printf("\n");
}
Expand Down Expand Up @@ -417,6 +430,9 @@ int SmokeTestPreset(EnvVars& ev,
} else {
Utils::Print("All tests passed\n");
}
if (skipCollectiveByDefault) {
Utils::Print("[WARN] Broadcast / Gather / AllToAll tests are disabled by default on multi-node; set TEST_LIST to enable them\n");
}
if (numRanks > 1 && Utils::GetRankPerPodMap().size() != 1) {
Utils::Print("[WARN] Copy (D2D) / Broadcast / Gather / AllToAll tests are skipped if ranks are not in same pod\n");
}
Expand Down
Loading