From 90d56b4cc7c3f0062b487f0e03a2ca1a9a24fd43 Mon Sep 17 00:00:00 2001 From: David Addison Date: Tue, 30 Jun 2026 18:23:54 -0700 Subject: [PATCH] Add per-iteration summary skip option Allow -K/--per_iter_skip to omit leading samples from the per-iteration summary while retaining all raw timing data. Record skipped samples in JSON output and expose the setting in the offline analyzer. Bump JSON output to version 3. Signed-off-by: David Addison --- README.md | 2 ++ src/common.cu | 21 ++++++++++++++++++--- src/util.cu | 15 ++++++++++----- src/util.h | 2 +- tools/analyze_perf_json.py | 4 +++- 5 files changed, 34 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index ef67d19..2d17b5e 100644 --- a/README.md +++ b/README.md @@ -76,6 +76,8 @@ All tests support the same set of arguments : * `-I,--per_iter_timing <0/1>` collect per-iteration CUDA event timings and print summary columns. `i_p99` uses nearest-rank percentile and may equal `i_max` with fewer than 100 samples. Incompatible with CUDA graph capture (`-G`). Default : 0. + * `-K,--per_iter_skip ` exclude leading samples from `-I` summary statistics. + Raw per-iteration JSON data remains complete. Default : 0. * Test operation * `-p,--parallel_init <0/1>` use threads to initialize NCCL in parallel. Default : 0. * `-c,--check ` perform count iterations, checking correctness of results on each iteration. This can be quite slow on large numbers of GPUs. Default : 1. diff --git a/src/common.cu b/src/common.cu index 39f13c5..dec49f3 100644 --- a/src/common.cu +++ b/src/common.cu @@ -104,6 +104,7 @@ static int ncclroot = 0; int parallel_init = 0; int blocking_coll = 0; int per_iter_timing = 0; +int per_iter_skip = 0; static int streamnull = 0; static int timeout = 0; int cudaGraphLaunches = 0; @@ -879,9 +880,10 @@ testResult_t BenchTime(struct threadArgs* args, ncclDataType_t type, ncclRedOp_t if (allProcessTimes && nProcessRows > 1) ReduceProcessMaxIterTimes(iterTimes, allProcessTimes, nProcessRows, iters); + int summaryIters = iters - per_iter_skip; struct IterStats stats; - computeIterStats(iterTimes, iters, &stats); - writePerIterReport(&stats, iterTimes, iters, in_place==0, allProcessTimes, nProcessRows); + computeIterStats(iterTimes + per_iter_skip, summaryIters, &stats); + writePerIterReport(&stats, iterTimes, iters, per_iter_skip, in_place==0, allProcessTimes, nProcessRows); if (in_place && report_timestamps) writeTimestamp(); @@ -1193,13 +1195,14 @@ int main(int argc, char* argv[], char **envp) { {"memory", required_argument, 0, 'M'}, {"unalign", required_argument, 0, 'u'}, {"per_iter_timing", required_argument, 0, 'I'}, + {"per_iter_skip", required_argument, 0, 'K'}, {"help", no_argument, 0, 'h'}, {} }; while(1) { int c; - c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex); + c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:K:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex); if (c == -1) break; @@ -1314,6 +1317,9 @@ int main(int argc, char* argv[], char **envp) { case 'I': per_iter_timing = (int)strtol(optarg, NULL, 0); break; + case 'K': + per_iter_skip = (int)strtol(optarg, NULL, 0); + break; case 'x': #if NCCL_VERSION_CODE >= NCCL_VERSION(2,27,0) ctaPolicy = (int)strtol(optarg, NULL, 0); @@ -1392,6 +1398,7 @@ int main(int argc, char* argv[], char **envp) { " (best with -z 0, -z 2, or -z 3; with -z 1 includes barrier wait;\n\t" " i_p99 uses nearest-rank percentile and may equal i_max\n\t" " with <100 samples) (default: 0)] \n\t" + "[-K,--per_iter_skip exclude leading samples from -I summary stats (default: 0)] \n\t" "[-h,--help]\n", programName); return 0; @@ -1419,6 +1426,14 @@ int main(int argc, char* argv[], char **envp) { fprintf(stderr, "blocking mode (-z 3) is incompatible with CUDA graph mode (-G). Disabling.\n"); blocking_coll = 0; } + if (per_iter_skip < 0 || per_iter_skip >= iters) { + fprintf(stderr, "per-iteration skip (-K) must be between 0 and %d. Disabling.\n", iters - 1); + per_iter_skip = 0; + } + if (per_iter_skip && !per_iter_timing) { + fprintf(stderr, "per-iteration skip (-K) requires per-iteration timing (-I 1). Disabling.\n"); + per_iter_skip = 0; + } #ifdef MPI_SUPPORT MPI_Init(&argc, &argv); diff --git a/src/util.cu b/src/util.cu index 364722f..665d854 100644 --- a/src/util.cu +++ b/src/util.cu @@ -42,13 +42,14 @@ extern int agg_iters; extern int parallel_init; extern int blocking_coll; extern int per_iter_timing; +extern int per_iter_skip; extern int cudaGraphLaunches; extern int unalign; static FILE *json_report_fp; static thread_local bool write_json; -#define JSON_FILE_VERSION 2 +#define JSON_FILE_VERSION 3 #define TIME_STRING_FORMAT "%Y-%m-%d %H:%M:%S" @@ -560,7 +561,7 @@ void computeIterStats(const double* times, int n, struct IterStats* stats) { free(sorted); } -void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs) { +void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs) { double cvPct = stats->avg > 0 ? (stats->stdev / stats->avg) * 100.0 : 0.0; char minStr[8], maxStr[8], p99Str[8], cvStr[8]; @@ -575,6 +576,7 @@ void writePerIterReport(const struct IterStats* stats, const double* iterTimes, const char* key = out_of_place ? "out_of_place_per_iter" : "in_place_per_iter"; jsonKey(key); jsonStartObject(); + jsonKey("skipped_iterations"); jsonInt(skippedIters); jsonKey("min_us"); jsonDouble(stats->min * 1e6); jsonKey("max_us"); jsonDouble(stats->max * 1e6); jsonKey("avg_us"); jsonDouble(stats->avg * 1e6); @@ -634,9 +636,11 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota if (blocking_coll == 3) PRINT("# Blocking Enabled: wait and barrier after each outer iteration (-n); " "time excludes barrier \n"); - if (per_iter_timing) - PRINT("# Per-Iteration Report: CUDA event timing; " - "i_* uses max over process rows \n"); + if (per_iter_timing) { + PRINT("# Per-Iteration Report: CUDA event timing; i_* uses max over process rows"); + if (per_iter_skip) PRINT("; summary skips first %d iterations", per_iter_skip); + PRINT(" \n"); + } if (parallel_init) PRINT("# Parallel Init Enabled: threads call into NcclInitRank concurrently \n"); PRINT("#\n"); @@ -661,6 +665,7 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota jsonKey("graph"); jsonInt(cudaGraphLaunches); jsonKey("blocking_collectives"); jsonBool(blocking_coll); jsonKey("per_iter_timing"); jsonBool(per_iter_timing); + jsonKey("per_iter_skip"); jsonInt(per_iter_skip); jsonKey("parallel_init"); jsonBool(parallel_init); } diff --git a/src/util.h b/src/util.h index c5f3060..b8dfbdb 100644 --- a/src/util.h +++ b/src/util.h @@ -40,7 +40,7 @@ void writeBenchmarkLinePreamble(size_t nBytes, size_t nElem, const char typeName void writeBenchmarkLineTerminator(int actualIters, const char *name); void writeBenchMarkLineNullBody(); void writeBenchmarkLineBody(double timeUsec, double algBw, double busBw, bool reportErrors, int64_t wrongElts, bool report_cputime, bool report_timestamps, bool out_of_place); -void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs); +void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs); void writeTimestamp(); testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int totalProcs, int color, const char hostname[], const char *program_name); void writeResultHeader(bool report_cputime, bool report_timestamps); diff --git a/tools/analyze_perf_json.py b/tools/analyze_perf_json.py index 2bab058..ccd9052 100644 --- a/tools/analyze_perf_json.py +++ b/tools/analyze_perf_json.py @@ -65,6 +65,8 @@ def print_overview(data, placement_key, placement_name): print(f" Results: {len(data.get('results', []))}") cfg = data.get('config', {}) print(f" Iterations: {cfg.get('iterations')} (agg: {cfg.get('aggregated_iterations', 1)})") + if cfg.get('per_iter_skip', 0): + print(f" Summary skip: {cfg['per_iter_skip']} leading iterations") print(f" Placement: {placement_name}") r0 = data['results'][0] if data.get('results') else {} per_iter = r0.get(placement_key, {}) @@ -190,7 +192,7 @@ def print_spikes(data, placement_key, placement_label, threshold=1.5, sizes=None def print_iter0(data, placement_key, placement_label, sizes=None): print("=" * 80) - print(f"ITERATION 0 WARMUP IMPACT ({placement_label})") + print(f"ITERATION 0 WARMUP IMPACT (RAW, {placement_label})") print("=" * 80) print(f"{'Size':>14} {'iter0 (us)':>10} {'iter1+ avg':>10} {'spike':>6} {'CV w/ iter0':>11} {'CV w/o':>8}") print("-" * 80)