Add per-iteration summary skip option

Allow -K/--per_iter_skip to omit leading samples from the
per-iteration summary while retaining all raw timing data.

Record skipped samples in JSON output and expose the setting in the
offline analyzer. Bump JSON output to version 3.

Signed-off-by: David Addison <[email protected]>
This commit is contained in:
David Addison
2026-07-07 10:54:17 -07:00
parent 206db923d4
commit 90d56b4cc7
5 changed files with 34 additions and 10 deletions
+2
View File
@@ -76,6 +76,8 @@ All tests support the same set of arguments :
* `-I,--per_iter_timing <0/1>` collect per-iteration CUDA event timings and print summary columns.
`i_p99` uses nearest-rank percentile and may equal `i_max` with fewer than 100 samples.
Incompatible with CUDA graph capture (`-G`). Default : 0.
* `-K,--per_iter_skip <count>` exclude leading samples from `-I` summary statistics.
Raw per-iteration JSON data remains complete. Default : 0.
* Test operation
* `-p,--parallel_init <0/1>` use threads to initialize NCCL in parallel. Default : 0.
* `-c,--check <check iteration count>` perform count iterations, checking correctness of results on each iteration. This can be quite slow on large numbers of GPUs. Default : 1.
+18 -3
View File
@@ -104,6 +104,7 @@ static int ncclroot = 0;
int parallel_init = 0;
int blocking_coll = 0;
int per_iter_timing = 0;
int per_iter_skip = 0;
static int streamnull = 0;
static int timeout = 0;
int cudaGraphLaunches = 0;
@@ -879,9 +880,10 @@ testResult_t BenchTime(struct threadArgs* args, ncclDataType_t type, ncclRedOp_t
if (allProcessTimes && nProcessRows > 1)
ReduceProcessMaxIterTimes(iterTimes, allProcessTimes, nProcessRows, iters);
int summaryIters = iters - per_iter_skip;
struct IterStats stats;
computeIterStats(iterTimes, iters, &stats);
writePerIterReport(&stats, iterTimes, iters, in_place==0, allProcessTimes, nProcessRows);
computeIterStats(iterTimes + per_iter_skip, summaryIters, &stats);
writePerIterReport(&stats, iterTimes, iters, per_iter_skip, in_place==0, allProcessTimes, nProcessRows);
if (in_place && report_timestamps) writeTimestamp();
@@ -1193,13 +1195,14 @@ int main(int argc, char* argv[], char **envp) {
{"memory", required_argument, 0, 'M'},
{"unalign", required_argument, 0, 'u'},
{"per_iter_timing", required_argument, 0, 'I'},
{"per_iter_skip", required_argument, 0, 'K'},
{"help", no_argument, 0, 'h'},
{}
};
while(1) {
int c;
c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex);
c = getopt_long(argc, argv, "t:g:b:e:i:f:n:m:w:N:p:I:K:c:o:d:r:z:y:T:hG:C:a:R:x:D:V:J:S:M:u:", longopts, &longindex);
if (c == -1)
break;
@@ -1314,6 +1317,9 @@ int main(int argc, char* argv[], char **envp) {
case 'I':
per_iter_timing = (int)strtol(optarg, NULL, 0);
break;
case 'K':
per_iter_skip = (int)strtol(optarg, NULL, 0);
break;
case 'x':
#if NCCL_VERSION_CODE >= NCCL_VERSION(2,27,0)
ctaPolicy = (int)strtol(optarg, NULL, 0);
@@ -1392,6 +1398,7 @@ int main(int argc, char* argv[], char **envp) {
" (best with -z 0, -z 2, or -z 3; with -z 1 includes barrier wait;\n\t"
" i_p99 uses nearest-rank percentile and may equal i_max\n\t"
" with <100 samples) (default: 0)] \n\t"
"[-K,--per_iter_skip <count> exclude leading samples from -I summary stats (default: 0)] \n\t"
"[-h,--help]\n",
programName);
return 0;
@@ -1419,6 +1426,14 @@ int main(int argc, char* argv[], char **envp) {
fprintf(stderr, "blocking mode (-z 3) is incompatible with CUDA graph mode (-G). Disabling.\n");
blocking_coll = 0;
}
if (per_iter_skip < 0 || per_iter_skip >= iters) {
fprintf(stderr, "per-iteration skip (-K) must be between 0 and %d. Disabling.\n", iters - 1);
per_iter_skip = 0;
}
if (per_iter_skip && !per_iter_timing) {
fprintf(stderr, "per-iteration skip (-K) requires per-iteration timing (-I 1). Disabling.\n");
per_iter_skip = 0;
}
#ifdef MPI_SUPPORT
MPI_Init(&argc, &argv);
+10 -5
View File
@@ -42,13 +42,14 @@ extern int agg_iters;
extern int parallel_init;
extern int blocking_coll;
extern int per_iter_timing;
extern int per_iter_skip;
extern int cudaGraphLaunches;
extern int unalign;
static FILE *json_report_fp;
static thread_local bool write_json;
#define JSON_FILE_VERSION 2
#define JSON_FILE_VERSION 3
#define TIME_STRING_FORMAT "%Y-%m-%d %H:%M:%S"
@@ -560,7 +561,7 @@ void computeIterStats(const double* times, int n, struct IterStats* stats) {
free(sorted);
}
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs) {
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs) {
double cvPct = stats->avg > 0 ? (stats->stdev / stats->avg) * 100.0 : 0.0;
char minStr[8], maxStr[8], p99Str[8], cvStr[8];
@@ -575,6 +576,7 @@ void writePerIterReport(const struct IterStats* stats, const double* iterTimes,
const char* key = out_of_place ? "out_of_place_per_iter" : "in_place_per_iter";
jsonKey(key);
jsonStartObject();
jsonKey("skipped_iterations"); jsonInt(skippedIters);
jsonKey("min_us"); jsonDouble(stats->min * 1e6);
jsonKey("max_us"); jsonDouble(stats->max * 1e6);
jsonKey("avg_us"); jsonDouble(stats->avg * 1e6);
@@ -634,9 +636,11 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota
if (blocking_coll == 3)
PRINT("# Blocking Enabled: wait and barrier after each outer iteration (-n); "
"time excludes barrier \n");
if (per_iter_timing)
PRINT("# Per-Iteration Report: CUDA event timing; "
"i_* uses max over process rows \n");
if (per_iter_timing) {
PRINT("# Per-Iteration Report: CUDA event timing; i_* uses max over process rows");
if (per_iter_skip) PRINT("; summary skips first %d iterations", per_iter_skip);
PRINT(" \n");
}
if (parallel_init) PRINT("# Parallel Init Enabled: threads call into NcclInitRank concurrently \n");
PRINT("#\n");
@@ -661,6 +665,7 @@ testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int tota
jsonKey("graph"); jsonInt(cudaGraphLaunches);
jsonKey("blocking_collectives"); jsonBool(blocking_coll);
jsonKey("per_iter_timing"); jsonBool(per_iter_timing);
jsonKey("per_iter_skip"); jsonInt(per_iter_skip);
jsonKey("parallel_init"); jsonBool(parallel_init);
}
+1 -1
View File
@@ -40,7 +40,7 @@ void writeBenchmarkLinePreamble(size_t nBytes, size_t nElem, const char typeName
void writeBenchmarkLineTerminator(int actualIters, const char *name);
void writeBenchMarkLineNullBody();
void writeBenchmarkLineBody(double timeUsec, double algBw, double busBw, bool reportErrors, int64_t wrongElts, bool report_cputime, bool report_timestamps, bool out_of_place);
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, bool out_of_place, const double* allProcessTimes, int nProcs);
void writePerIterReport(const struct IterStats* stats, const double* iterTimes, int nIters, int skippedIters, bool out_of_place, const double* allProcessTimes, int nProcs);
void writeTimestamp();
testResult_t writeDeviceReport(size_t *maxMem, int localRank, int proc, int totalProcs, int color, const char hostname[], const char *program_name);
void writeResultHeader(bool report_cputime, bool report_timestamps);
+3 -1
View File
@@ -65,6 +65,8 @@ def print_overview(data, placement_key, placement_name):
print(f" Results: {len(data.get('results', []))}")
cfg = data.get('config', {})
print(f" Iterations: {cfg.get('iterations')} (agg: {cfg.get('aggregated_iterations', 1)})")
if cfg.get('per_iter_skip', 0):
print(f" Summary skip: {cfg['per_iter_skip']} leading iterations")
print(f" Placement: {placement_name}")
r0 = data['results'][0] if data.get('results') else {}
per_iter = r0.get(placement_key, {})
@@ -190,7 +192,7 @@ def print_spikes(data, placement_key, placement_label, threshold=1.5, sizes=None
def print_iter0(data, placement_key, placement_label, sizes=None):
print("=" * 80)
print(f"ITERATION 0 WARMUP IMPACT ({placement_label})")
print(f"ITERATION 0 WARMUP IMPACT (RAW, {placement_label})")
print("=" * 80)
print(f"{'Size':>14} {'iter0 (us)':>10} {'iter1+ avg':>10} {'spike':>6} {'CV w/ iter0':>11} {'CV w/o':>8}")
print("-" * 80)