mirror of
https://github.com/NVIDIA/nccl-tests.git
synced 2026-10-01 14:37:28 +00:00
Add per-iteration summary skip option
Allow -K/--per_iter_skip to omit leading samples from the per-iteration summary while retaining all raw timing data. Record skipped samples in JSON output and expose the setting in the offline analyzer. Bump JSON output to version 3. Signed-off-by: David Addison <[email protected]>
This commit is contained in:
@@ -76,6 +76,8 @@ All tests support the same set of arguments :
|
||||
* `-I,--per_iter_timing <0/1>` collect per-iteration CUDA event timings and print summary columns.
|
||||
`i_p99` uses nearest-rank percentile and may equal `i_max` with fewer than 100 samples.
|
||||
Incompatible with CUDA graph capture (`-G`). Default : 0.
|
||||
* `-K,--per_iter_skip <count>` exclude leading samples from `-I` summary statistics.
|
||||
Raw per-iteration JSON data remains complete. Default : 0.
|
||||
* Test operation
|
||||
* `-p,--parallel_init <0/1>` use threads to initialize NCCL in parallel. Default : 0.
|
||||
* `-c,--check <check iteration count>` perform count iterations, checking correctness of results on each iteration. This can be quite slow on large numbers of GPUs. Default : 1.
|
||||
|
||||
+18
-3
@@ -104,6 +104,7 @@ static int ncclroot = 0;
|
||||
int parallel_init = 0;
|
||||
int blocking_coll = 0;
|
||||
int per_iter_timing = 0;
|
||||
int per_iter_skip = 0;
|
||||
static int streamnull = 0;
|
||||
static int timeout = 0;
|
||||
int cudaGraphLaunches = 0;
|
||||
@@ -879,9 +880,10 @@ testResult_t BenchTime(struct threadArgs* args, ncclDataType_t type, ncclRedOp_t
|
||||
if (allProcessTimes && nProcessRows > 1)
|
||||
ReduceProcessMaxIterTimes(iterTimes, allProcessTimes, nProcessRows, iters);
|
||||
|
||||
int summaryIters = iters - per_iter_skip;
|
||||
struct IterStats stats;
|
||||
computeIterStats(iterTimes, iters, &stats);
|
||||
writePerIterReport(&stats, iterTimes, iters, in_place==0, allProcessTimes, nProcessRows);
|
||||
computeIterStats(iterTimes + per_iter_skip, summaryIters, &stats);
|
||||
writePerIterReport(&stats, iterTimes, iters, per_iter_skip, in_place==0, allProcessTimes, nProcessRows);
|
||||
|
||||
if (in_place && report_timestamps) writeTimestamp();
|
||||
|
||||
@@ -1193,13 +1195,14 @@ int main(int argc, char* argv[], char **envp) {
|
||||
{"memory", required_argument, 0, 'M'},
|
||||
{"unalign", required_argument, 0, 'u'},
|
||||
{"per_iter_timing", required_argument, 0, 'I'},
|
||||
{"per_iter_skip", required_argument, 0, 'K'},
|
||||
{"help", no_argument, 0, 'h'},
|
||||
{}
|
||||
};
|
||||
|
||||
while(1) {
|
||||
int c;
|
||||
c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex);
|
||||
c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:K:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex);
|
||||
|
||||
if (c == -1)
|
||||
break;
|
||||
@@ -1314,6 +1317,9 @@ int main(int argc, char* argv[], char **envp) {
|
||||
case 'I':
|
||||
per_iter_timing = (int)strtol(optarg, NULL, 0);
|
||||
break;
|
||||
case 'K':
|
||||
per_iter_skip = (int)strtol(optarg, NULL, 0);
|
||||
break;
|
||||
case 'x':
|
||||
#if NCCL_VERSION_CODE >= NCCL_VERSION(2,27,0)
|
||||
ctaPolicy = (int)strtol(optarg, NULL, 0);
|
||||
@@ -1392,6 +1398,7 @@ int main(int argc, char* argv[], char **envp) {
|
||||
" (best with -z 0, -z 2, or -z 3; with -z 1 includes barrier wait;\n\t"
|
||||
" i_p99 uses nearest-rank percentile and may equal i_max\n\t"
|
||||
" with <100 samples) (default: 0)] \n\t"
|
||||
"[-K,--per_iter_skip <count> exclude leading samples from -I summary stats (default: 0)] \n\t"
|
||||
"[-h,--help]\n",
|
||||
programName);
|
||||
return 0;
|
||||
@@ -1419,6 +1426,14 @@ int main(int argc, char* argv[], char **envp) {
|
||||
fprintf(stderr, "blocking mode (-z 3) is incompatible with CUDA graph mode (-G). Disabling.\n");
|
||||
blocking_coll = 0;
|
||||
}
|
||||
if (per_iter_skip < 0 || per_iter_skip >= iters) {
|
||||
fprintf(stderr, "per-iteration skip (-K) must be between 0 and %d. Disabling.\n", iters - 1);
|
||||
per_iter_skip = 0;
|
||||
}
|
||||
if (per_iter_skip && !per_iter_timing) {
|
||||
fprintf(stderr, "per-iteration skip (-K) requires per-iteration timing (-I 1). Disabling.\n");
|
||||
per_iter_skip = 0;
|
||||
}
|
||||
|
||||
#ifdef MPI_SUPPORT
|
||||
MPI_Init(&argc, &argv);
|
||||
|
||||
+10
-5
@@ -42,13 +42,14 @@ extern int agg_iters;
|
||||
extern int parallel_init;
|
||||
extern int blocking_coll;
|
||||
extern int per_iter_timing;
|
||||
extern int per_iter_skip;
|
||||
extern int cudaGraphLaunches;
|
||||
extern int unalign;
|
||||
|
||||
static FILE *json_report_fp;
|
||||
static thread_local bool write_json;
|
||||
|
||||
#define JSON_FILE_VERSION 2
|
||||
#define JSON_FILE_VERSION 3
|
||||
|
||||
#define TIME_STRING_FORMAT "%Y-%m-%d %H:%M:%S"
|
||||
|
||||
@@ -560,7 +561,7 @@ void computeIterStats(const double* times, int n, struct IterStats* stats) {
|
||||
free(sorted);
|
||||
}
|
||||
|
||||
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs) {
|
||||
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs) {
|
||||
double cvPct = stats->avg > 0 ? (stats->stdev / stats->avg) * 100.0 : 0.0;
|
||||
|
||||
char minStr[8], maxStr[8], p99Str[8], cvStr[8];
|
||||
@@ -575,6 +576,7 @@ void writePerIterReport(const struct IterStats* stats, const double* iterTimes,
|
||||
const char* key = out_of_place ? "out_of_place_per_iter" : "in_place_per_iter";
|
||||
jsonKey(key);
|
||||
jsonStartObject();
|
||||
jsonKey("skipped_iterations"); jsonInt(skippedIters);
|
||||
jsonKey("min_us"); jsonDouble(stats->min * 1e6);
|
||||
jsonKey("max_us"); jsonDouble(stats->max * 1e6);
|
||||
jsonKey("avg_us"); jsonDouble(stats->avg * 1e6);
|
||||
@@ -634,9 +636,11 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota
|
||||
if (blocking_coll == 3)
|
||||
PRINT("# Blocking Enabled: wait and barrier after each outer iteration (-n); "
|
||||
"time excludes barrier \n");
|
||||
if (per_iter_timing)
|
||||
PRINT("# Per-Iteration Report: CUDA event timing; "
|
||||
"i_* uses max over process rows \n");
|
||||
if (per_iter_timing) {
|
||||
PRINT("# Per-Iteration Report: CUDA event timing; i_* uses max over process rows");
|
||||
if (per_iter_skip) PRINT("; summary skips first %d iterations", per_iter_skip);
|
||||
PRINT(" \n");
|
||||
}
|
||||
if (parallel_init) PRINT("# Parallel Init Enabled: threads call into NcclInitRank concurrently \n");
|
||||
PRINT("#\n");
|
||||
|
||||
@@ -661,6 +665,7 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota
|
||||
jsonKey("graph"); jsonInt(cudaGraphLaunches);
|
||||
jsonKey("blocking_collectives"); jsonBool(blocking_coll);
|
||||
jsonKey("per_iter_timing"); jsonBool(per_iter_timing);
|
||||
jsonKey("per_iter_skip"); jsonInt(per_iter_skip);
|
||||
jsonKey("parallel_init"); jsonBool(parallel_init);
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@ void writeBenchmarkLinePreamble(size_t nBytes, size_t nElem, const char typeName
|
||||
void writeBenchmarkLineTerminator(int actualIters, const char *name);
|
||||
void writeBenchMarkLineNullBody();
|
||||
void writeBenchmarkLineBody(double timeUsec, double algBw, double busBw, bool reportErrors, int64_t wrongElts, bool report_cputime, bool report_timestamps, bool out_of_place);
|
||||
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs);
|
||||
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs);
|
||||
void writeTimestamp();
|
||||
testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int totalProcs, int color, const char hostname[], const char *program_name);
|
||||
void writeResultHeader(bool report_cputime, bool report_timestamps);
|
||||
|
||||
@@ -65,6 +65,8 @@ def print_overview(data, placement_key, placement_name):
|
||||
print(f" Results: {len(data.get('results', []))}")
|
||||
cfg = data.get('config', {})
|
||||
print(f" Iterations: {cfg.get('iterations')} (agg: {cfg.get('aggregated_iterations', 1)})")
|
||||
if cfg.get('per_iter_skip', 0):
|
||||
print(f" Summary skip: {cfg['per_iter_skip']} leading iterations")
|
||||
print(f" Placement: {placement_name}")
|
||||
r0 = data['results'][0] if data.get('results') else {}
|
||||
per_iter = r0.get(placement_key, {})
|
||||
@@ -190,7 +192,7 @@ def print_spikes(data, placement_key, placement_label, threshold=1.5, sizes=None
|
||||
|
||||
def print_iter0(data, placement_key, placement_label, sizes=None):
|
||||
print("=" * 80)
|
||||
print(f"ITERATION 0 WARMUP IMPACT ({placement_label})")
|
||||
print(f"ITERATION 0 WARMUP IMPACT (RAW, {placement_label})")
|
||||
print("=" * 80)
|
||||
print(f"{'Size':>14} {'iter0 (us)':>10} {'iter1+ avg':>10} {'spike':>6} {'CV w/ iter0':>11} {'CV w/o':>8}")
|
||||
print("-" * 80)
|
||||
|
||||
Reference in New Issue
Block a user