From 1c42c2cb145184eac7bb1a35dae23d36187dff6d Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Thu, 1 Oct 2026 14:16:42 +0100 Subject: [PATCH 1/7] Add harness-ractor-mem memory harness A harness for ractor memory pathology benchmarks. The scenario block runs on the main Ractor and receives the ractor count; it may return a cleanup proc, which the harness calls after measuring retained RSS so a scenario can park idling ractors that outlive the measurement window. Retained RSS is measured against a single pristine process-start baseline after GC settle. A per-trial baseline would collapse toward zero, because memory retained by dead or idle ractors is reused by later allocation rather than returned to the OS. Peak RSS is sampled by a background thread and verified against VmHWM. Scenario wall times are reported so result tables keep working. --- README.md | 1 + harness-ractor-mem/harness.rb | 109 ++++++++++++++++++++++++++++++++++ 2 files changed, 110 insertions(+) create mode 100644 harness-ractor-mem/harness.rb diff --git a/README.md b/README.md index db115f02..a05988e9 100644 --- a/README.md +++ b/README.md @@ -222,6 +222,7 @@ You can find several test harnesses in this repository: * harness-stats - count method calls and loop iterations * harness-vernier - a harness to profile the benchmark with vernier * harness-warmup - a harness which runs as long as needed to find warmed up (peak) performance +* harness-ractor-mem - a harness for ractor memory pathology benchmarks, measuring retained and peak RSS across ractor counts To use it, run a benchmark script directly, specifying a harness directory with `-I`: diff --git a/harness-ractor-mem/harness.rb b/harness-ractor-mem/harness.rb new file mode 100644 index 00000000..c180bc83 --- /dev/null +++ b/harness-ractor-mem/harness.rb @@ -0,0 +1,109 @@ +require "etc" +require_relative "../harness/harness-common" + +Warning[:experimental] = false + +default_ractors = [1, 2, 4, 6, 8] +if rs = ENV["RUBY_BENCH_RACTORS"] + rs = rs.split(",").map(&:to_i) + rs = rs.sort.uniq + rs -= [0] + ractors = rs unless rs.empty? +end +RACTORS = (ractors || default_ractors).freeze + +MAX_ITERS = Integer(ENV.fetch("MAX_BENCH_ITRS", 3)) +SETTLE_SLEEP = Float(ENV.fetch("RACTOR_MEM_SETTLE_SLEEP", 0.1)) +PEAK_SAMPLE_INTERVAL = Float(ENV.fetch("RACTOR_MEM_PEAK_SAMPLE_INTERVAL", 0.005)) + +puts RUBY_DESCRIPTION + +def gc_settle + 2.times { GC.start(full_mark: true, immediate_sweep: true) } + sleep SETTLE_SLEEP +end + +PAGE_SIZE = Etc.sysconf(Etc::SC_PAGESIZE) + +def statm_rss + PAGE_SIZE * Integer(File.read("/proc/self/statm").split(" ")[1]) +end + +def measure_peak_rss + stop = false + peak = 0 + sampler = Thread.new do + Thread.current.report_on_exception = false + until stop + begin + rss = statm_rss + peak = rss if rss > peak + rescue StandardError + break + end + sleep PEAK_SAMPLE_INTERVAL + end + end + result = yield + stop = true + sampler.join + [result, peak] +end + +def run_benchmark(num_itrs_hint, &scenario) + bench_itrs = Integer(ENV.fetch("MIN_BENCH_ITRS", num_itrs_hint)) + bench_itrs = MAX_ITERS if bench_itrs > MAX_ITERS + bench_itrs = 1 if bench_itrs < 1 + + gc_settle + base_rss = get_rss + puts format("base RSS: %.1f MiB", base_rss / 2**20.0) + + metrics = Hash.new { |h, count| h[count] = { retained: [], peak: [] } } + times = [] + + RACTORS.each do |count| + puts "ractors: #{count}" + itr = 0 + while itr < bench_itrs + t0 = Process.clock_gettime(Process::CLOCK_MONOTONIC) + finish, peak = measure_peak_rss { scenario.call(count) } + times << Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0 + gc_settle + retained = get_rss - base_rss + finish&.call + gc_settle + metrics[count][:retained] << retained + metrics[count][:peak] << peak + puts format(" itr #%d: retained %.1f MiB, peak %.1f MiB", + itr + 1, retained / 2**20.0, peak / 2**20.0) + itr += 1 + end + end + + medians = {} + RACTORS.each do |count| + medians[count] = { + retained: Stats.new(metrics[count][:retained]).median, + peak: Stats.new(metrics[count][:peak]).median, + } + end + worst_retained_count = medians.max_by { |_, m| m[:retained] }.first + worst_retained = medians[worst_retained_count][:retained] + worst_peak = medians.values.map { |m| m[:peak] }.max + + RACTORS.each do |count| + m = medians[count] + puts format("BENCH_METRIC retained_mib_r%d=%.1f", count, m[:retained] / 2**20.0) + puts format("BENCH_METRIC peak_mib_r%d=%.1f", count, m[:peak] / 2**20.0) + end + puts format("BENCH_METRIC retained_mib=%.1f", worst_retained / 2**20.0) + puts format("BENCH_METRIC peak_mib=%.1f", worst_peak / 2**20.0) + puts format("BENCH_METRIC worst_ractor_count=%d", worst_retained_count) + + return_results([], times, + ractor_mem_base_rss: base_rss, + ractor_mem_medians: medians, + ractor_mem_samples: metrics, + ) +end From 9c47dd29ca4e30a3be0614b8a6fa7991cc18374a Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Thu, 1 Oct 2026 14:16:55 +0100 Subject: [PATCH 2/7] Add ractor memory pathology benchmarks Add three benchmarks that expose pathological memory behaviour with multiple ractors, for GC work that reclaims ractor-local memory: - ractor-dead-set: every ractor builds a large live set and terminates; their final live sets cannot be reclaimed by a full GC. - ractor-idle-garbage: every ractor builds and drops a large set, then idles without allocating; the garbage cannot be swept while they idle. - ractor-msg-backlog: unshareable payloads flood the queues of gated consumer ractors, duplicating data per consumer. benchmarks.yml marks them ractor_only with default_harness harness-ractor-mem, which overrides the --category ractor default. Pin that selection with a regression test, and document the benchmarks in the README. --- README.md | 28 +++++++++++++- benchmarks.yml | 15 ++++++++ benchmarks/ractor-dead-set/benchmark.rb | 27 +++++++++++++ benchmarks/ractor-idle-garbage/benchmark.rb | 42 +++++++++++++++++++++ benchmarks/ractor-msg-backlog/benchmark.rb | 42 +++++++++++++++++++++ test/benchmark_suite_test.rb | 19 ++++++++++ 6 files changed, 172 insertions(+), 1 deletion(-) create mode 100644 benchmarks/ractor-dead-set/benchmark.rb create mode 100644 benchmarks/ractor-idle-garbage/benchmark.rb create mode 100644 benchmarks/ractor-msg-backlog/benchmark.rb diff --git a/README.md b/README.md index a05988e9..d28c5418 100644 --- a/README.md +++ b/README.md @@ -138,7 +138,9 @@ There are two Ractor-related categories: * **`--category ractor`** - Runs both regular benchmarks marked with `ractor: true` in `benchmarks.yml` AND all benchmarks from the `benchmarks-ractor` - directory. The `harness-ractor` harness is used for both types of benchmark. + directory. The `harness-ractor` harness is used unless a benchmark sets + `default_harness` in `benchmarks.yml` (the Ractor memory benchmarks set + `harness-ractor-mem`). * **`--category ractor-only`** - Runs ONLY benchmarks from the `benchmarks-ractor` directory, ignoring regular benchmarks even if they are @@ -165,6 +167,30 @@ intended to be used with any harness except `harness-ractor`. Note: The `harness-ractor` harness is automatically selected when using these categories, so there's no need to specify `--harness` manually. +### Ractor Memory Benchmarks + +Three benchmarks measure pathological memory behaviour with multiple ractors, +for GC work that reclaims ractor-local memory. They use the +`harness-ractor-mem` harness, which measures retained RSS against a pristine +process baseline after a full GC, plus a sampled peak RSS: + +* **`ractor-dead-set`** - Every ractor builds a large live set and terminates. + The final live sets of dead ractors cannot be reclaimed by a full GC. +* **`ractor-idle-garbage`** - Every ractor builds a large set, drops all + references, then idles without allocating. The garbage cannot be swept + while the ractor idles. +* **`ractor-msg-backlog`** - Unshareable payloads flood the queues of gated + consumer ractors, duplicating the payload data per consumer. + +```bash +ruby -Iharness-ractor-mem benchmarks/ractor-dead-set/benchmark.rb +``` + +The harness prints `BENCH_METRIC retained_mib=` and +`BENCH_METRIC peak_mib=...` lines, plus one pair per ractor count. The ractor +counts and trials are controlled with `RUBY_BENCH_RACTORS` (default `1,2,4,6,8`) +and `MIN_BENCH_ITRS` (default 3). + ## Ruby options By default, ruby-bench benchmarks the Ruby used for `run_benchmarks.rb`. diff --git a/benchmarks.yml b/benchmarks.yml index 154aefaa..aa9c06a2 100644 --- a/benchmarks.yml +++ b/benchmarks.yml @@ -127,6 +127,21 @@ protoboeuf-encode: ractor: true rack: desc: test the performance of the Rack framework with barely any routing. +ractor-dead-set: + desc: multiple ractors each build a large live set and terminate, then memory retention after their death is measured with the ractor memory harness. + ractor: true + ractor_only: true + default_harness: harness-ractor-mem +ractor-idle-garbage: + desc: multiple ractors each build and drop a large garbage set, then idle uncollected while retention is measured with the ractor memory harness. + ractor: true + ractor_only: true + default_harness: harness-ractor-mem +ractor-msg-backlog: + desc: unshareable message payloads flood the queues of gated consumer ractors, duplicating data per consumer and driving peak RSS and retention. + ractor: true + ractor_only: true + default_harness: harness-ractor-mem ruby-json: desc: an optimized version of the json_pure gem's pure Ruby JSON parser. ractor: true diff --git a/benchmarks/ractor-dead-set/benchmark.rb b/benchmarks/ractor-dead-set/benchmark.rb new file mode 100644 index 00000000..947d1e7b --- /dev/null +++ b/benchmarks/ractor-dead-set/benchmark.rb @@ -0,0 +1,27 @@ +# Multiple ractors each build a large live set and then terminate. +# The final live sets of dead ractors are garbage after death. Measured +# retention shows how much of that memory a full GC fails to reclaim. + +Warning[:experimental] = false + +require_relative "../../harness/loader" + +DEAD_SET_ITEMS = Integer(ENV.fetch("RACTOR_DEAD_SET_ITEMS", 100_000)) + +run_benchmark(3) do |num_ractors| + workers = num_ractors.times.map do |worker| + Ractor.new(worker, DEAD_SET_ITEMS) do |worker_id, items| + keep = [] + i = 0 + while i < items + keep << "worker #{worker_id} item #{i} " + ("y" * 100) + i += 1 + end + keep.size + end + end + + total = workers.sum(&:value) + raise "unexpected dead-set size" unless total == num_ractors * DEAD_SET_ITEMS + nil +end diff --git a/benchmarks/ractor-idle-garbage/benchmark.rb b/benchmarks/ractor-idle-garbage/benchmark.rb new file mode 100644 index 00000000..e031a083 --- /dev/null +++ b/benchmarks/ractor-idle-garbage/benchmark.rb @@ -0,0 +1,42 @@ +# Multiple ractors each build a large live set, drop all references, then park +# on Ractor.receive without allocating again. Their garbage cannot be swept by +# the main Ractor's GC while they idle, so it is retained until each worker +# is released. The scenario returns a cleanup proc that the harness calls +# after the retention measurement. + +Warning[:experimental] = false + +require_relative "../../harness/loader" + +IDLE_GARBAGE_ITEMS = Integer(ENV.fetch("RACTOR_IDLE_GARBAGE_ITEMS", 100_000)) + +run_benchmark(3) do |num_ractors| + drained = Ractor.new(num_ractors) do |count| + count.times { Ractor.receive } + :all_drained + end + + workers = num_ractors.times.map do |worker| + Ractor.new(drained, worker, IDLE_GARBAGE_ITEMS) do |ack, worker_id, items| + keep = [] + i = 0 + while i < items + keep << "worker #{worker_id} garbage #{i} " + ("y" * 100) + i += 1 + end + keep = nil + ack.send :drained + Ractor.receive + :worker_done + end + end + + raise "unexpected drain barrier result" unless drained.value == :all_drained + + proc do + workers.each do |worker| + worker.send :stop + raise "unexpected worker result" unless worker.value == :worker_done + end + end +end diff --git a/benchmarks/ractor-msg-backlog/benchmark.rb b/benchmarks/ractor-msg-backlog/benchmark.rb new file mode 100644 index 00000000..42bec6e5 --- /dev/null +++ b/benchmarks/ractor-msg-backlog/benchmark.rb @@ -0,0 +1,42 @@ +# The main Ractor floods the incoming queues of multiple gated consumer +# ractors with unshareable string payloads. The copies made on send pile up +# in the queues while the consumers sleep, which duplicates the payload data +# per consumer and drives peak RSS. Measured retention after the queues are +# drained shows how much of the copied memory a full GC fails to reclaim. +# This file intentionally omits frozen_string_literal: frozen payloads are +# shareable and would not be copied on send. + +Warning[:experimental] = false + +require_relative "../../harness/loader" + +BACKLOG_MESSAGES = Integer(ENV.fetch("RACTOR_BACKLOG_MESSAGES", 20_000)) +BACKLOG_GATE_SLEEP = Float(ENV.fetch("RACTOR_BACKLOG_GATE_SLEEP", 1.0)) + +run_benchmark(3) do |num_ractors| + consumers = num_ractors.times.map do |consumer| + Ractor.new(consumer, BACKLOG_GATE_SLEEP) do |consumer_id, gate| + sleep gate + taken = 0 + loop do + message = Ractor.receive + break if message == :done + taken += 1 + end + [consumer_id, taken] + end + end + + consumers.each_with_index do |consumer, consumer_id| + i = 0 + while i < BACKLOG_MESSAGES + consumer.send "consumer #{consumer_id} message #{i} " + ("x" * 200) + i += 1 + end + consumer.send :done + end + + results = consumers.map(&:value) + raise "unexpected backlog drain" unless results.sum { |_, taken| taken } == num_ractors * BACKLOG_MESSAGES + nil +end diff --git a/test/benchmark_suite_test.rb b/test/benchmark_suite_test.rb index f3cdb613..8be09400 100644 --- a/test/benchmark_suite_test.rb +++ b/test/benchmark_suite_test.rb @@ -154,6 +154,25 @@ assert_equal 'custom-harness', suite.send(:benchmark_harness_for, 'custom_harness_bench') assert_equal 'custom-harness', suite.send(:benchmark_harness_for, 'simple') end + + it 'keeps a custom default_harness on ractor category runs' do + metadata = { + 'simple' => { 'category' => 'micro' }, + 'ractor_mem_bench' => { 'ractor' => true, 'ractor_only' => true, 'default_harness' => 'harness-ractor-mem' } + } + File.write('benchmarks.yml', YAML.dump(metadata)) + + suite = BenchmarkSuite.new( + categories: ['ractor'], + name_filters: [], + out_path: @out_path, + harness: 'harness-ractor', + harness_explicit: false + ) + + assert_equal 'harness-ractor-mem', suite.send(:benchmark_harness_for, 'ractor_mem_bench') + assert_equal 'harness-ractor', suite.send(:benchmark_harness_for, 'simple') + end end describe '#ractor_category_run?' do From bda4b77a4db3f718b68339d77fbde1dc73280da4 Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 6 Oct 2026 16:52:10 +0100 Subject: [PATCH 3/7] Share the Ractor GC support check and series in GCStats Move check_ractor_gc_support and the per-count GC series map from harness-ractor into lib/gc_stats.rb, so other harnesses can reuse them. --- harness-ractor/harness.rb | 26 ++------------------------ lib/gc_stats.rb | 22 ++++++++++++++++++++++ 2 files changed, 24 insertions(+), 24 deletions(-) diff --git a/harness-ractor/harness.rb b/harness-ractor/harness.rb index f7c42667..1c73f3b8 100644 --- a/harness-ractor/harness.rb +++ b/harness-ractor/harness.rb @@ -48,7 +48,7 @@ def run_benchmark(num_itrs_hint, ractor_args: [], &block) bench_itrs = MAX_ITERS if bench_itrs > MAX_ITERS if RACTOR_GC_ENABLED - check_ractor_gc_support + GCStats.check_ractor_gc_support! gc_config = GC.config.transform_keys(&:to_s) if GC.respond_to?(:config) GCStats.with_measure_total_time do run_warmup(warmup_itrs, ractor_args, &block) @@ -61,18 +61,6 @@ def run_benchmark(num_itrs_hint, ractor_args: [], &block) end end -def check_ractor_gc_support - unless GCStats.ractor_local_gc_supported? - raise NotImplementedError, "Ractor GC metrics require Ruby 4.1 or newer" - end - unless GC.respond_to?(:total_time) && GC.respond_to?(:measure_total_time) && GC.respond_to?(:measure_total_time=) - raise NotImplementedError, "Ractor GC metrics require GC.total_time and GC.measure_total_time=" - end - unless GCStats.global_gc_attributed? - raise NotImplementedError, "Ractor GC metrics require per-Ractor global GC attribution (ruby/ruby#19147)" - end -end - def run_warmup(warmup_itrs, ractor_args, &block) warmup_itrs.times do args = ractor_args.empty? ? [] : ractor_deep_dup(ractor_args) @@ -110,16 +98,6 @@ def run_benchmark_timing(bench_itrs, ractor_args, &block) return_results([], stats.values.flatten, bench_by_ractors: stats) end -RACTOR_GC_SERIES = { - "gc_count_bench" => "gc_count", - "gc_global_count_bench" => "gc_global_count", - "gc_major_count_bench" => "gc_major_count", - "gc_minor_count_bench" => "gc_minor_count", - "gc_marking_time_bench" => "gc_marking_time", - "gc_sweeping_time_bench" => "gc_sweeping_time", - "gc_total_time_bench" => "gc_total_time_ns", -}.freeze - def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:) stats = Hash.new { |h,k| h[k] = [] } gc_by_ractors = {} @@ -144,7 +122,7 @@ def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:) agg = GCStats.aggregate(worker_samples) total_ms = agg["gc_total_time_ns"]&.fdiv(1_000_000) - RACTOR_GC_SERIES.each do |series_name, field| + GCStats::RACTOR_SERIES.each do |series_name, field| series[series_name] << (field == "gc_total_time_ns" ? total_ms : agg[field]) end CONTROLLER_GC_SERIES.each_key { |series_name| series[series_name] << controller_deltas[series_name] } diff --git a/lib/gc_stats.rb b/lib/gc_stats.rb index 3a39186a..1106bef3 100644 --- a/lib/gc_stats.rb +++ b/lib/gc_stats.rb @@ -16,6 +16,16 @@ module GCStats SCALAR_FIELD_NAMES = (SCALAR_FIELDS.map(&:first) + [TOTAL_TIME_FIELD]).freeze + RACTOR_SERIES = { + "gc_count_bench" => "gc_count", + "gc_global_count_bench" => "gc_global_count", + "gc_major_count_bench" => "gc_major_count", + "gc_minor_count_bench" => "gc_minor_count", + "gc_marking_time_bench" => "gc_marking_time", + "gc_sweeping_time_bench" => "gc_sweeping_time", + "gc_total_time_bench" => TOTAL_TIME_FIELD, + }.freeze + def stat_available?(key) GC.stat(key).is_a?(Numeric) rescue ArgumentError @@ -48,6 +58,18 @@ def global_gc_attributed? status.success? end + def check_ractor_gc_support! + unless ractor_local_gc_supported? + raise NotImplementedError, "Ractor GC metrics require Ruby 4.1 or newer" + end + unless GC.respond_to?(:total_time) && GC.respond_to?(:measure_total_time) && GC.respond_to?(:measure_total_time=) + raise NotImplementedError, "Ractor GC metrics require GC.total_time and GC.measure_total_time=" + end + unless global_gc_attributed? + raise NotImplementedError, "Ractor GC metrics require per-Ractor global GC attribution (ruby/ruby#19147)" + end + end + def heap_snapshot return {} unless GC.respond_to?(:stat_heap) GC.stat_heap From 6cb8115495815234cee8f95fd9bf767935386724 Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 6 Oct 2026 16:55:29 +0100 Subject: [PATCH 4/7] Settle with a global GC in harness-ractor-mem when supported On Ruby builds with Ractor-local GC, a plain GC.start only collects the main Ractor's object space, so the retention settle no longer covered the whole process. Pass global: true when GC.start accepts it and record the settle mode as ractor_mem_settle in the results. --- README.md | 9 ++++++++- harness-ractor-mem/harness.rb | 11 ++++++++++- 2 files changed, 18 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index d28c5418..eeb820f7 100644 --- a/README.md +++ b/README.md @@ -175,7 +175,8 @@ for GC work that reclaims ractor-local memory. They use the process baseline after a full GC, plus a sampled peak RSS: * **`ractor-dead-set`** - Every ractor builds a large live set and terminates. - The final live sets of dead ractors cannot be reclaimed by a full GC. + Retention shows how much of the dead ractors' final live sets a full GC + leaves resident. * **`ractor-idle-garbage`** - Every ractor builds a large set, drops all references, then idles without allocating. The garbage cannot be swept while the ractor idles. @@ -191,6 +192,12 @@ The harness prints `BENCH_METRIC retained_mib=` and counts and trials are controlled with `RUBY_BENCH_RACTORS` (default `1,2,4,6,8`) and `MIN_BENCH_ITRS` (default 3). +The harness collects with `GC.start(global: true)` when the target Ruby's +`GC.start` accepts the `global:` keyword. Some Ruby 4.1 builds do not accept it. +On a target with Ractor-local GC, a plain `GC.start` collects only the main +Ractor's object space. The JSON field `ractor_mem_settle` records `global` or +`default`. + ## Ruby options By default, ruby-bench benchmarks the Ruby used for `run_benchmarks.rb`. diff --git a/harness-ractor-mem/harness.rb b/harness-ractor-mem/harness.rb index c180bc83..3af25510 100644 --- a/harness-ractor-mem/harness.rb +++ b/harness-ractor-mem/harness.rb @@ -16,10 +16,18 @@ SETTLE_SLEEP = Float(ENV.fetch("RACTOR_MEM_SETTLE_SLEEP", 0.1)) PEAK_SAMPLE_INTERVAL = Float(ENV.fetch("RACTOR_MEM_PEAK_SAMPLE_INTERVAL", 0.005)) +GLOBAL_GC_START = GC.method(:start).parameters.include?([:key, :global]) + puts RUBY_DESCRIPTION def gc_settle - 2.times { GC.start(full_mark: true, immediate_sweep: true) } + 2.times do + if GLOBAL_GC_START + GC.start(full_mark: true, immediate_sweep: true, global: true) + else + GC.start(full_mark: true, immediate_sweep: true) + end + end sleep SETTLE_SLEEP end @@ -102,6 +110,7 @@ def run_benchmark(num_itrs_hint, &scenario) puts format("BENCH_METRIC worst_ractor_count=%d", worst_retained_count) return_results([], times, + ractor_mem_settle: GLOBAL_GC_START ? "global" : "default", ractor_mem_base_rss: base_rss, ractor_mem_medians: medians, ractor_mem_samples: metrics, From 3c6f9a3246c6982e03b0651e5b4a84d507915848 Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 6 Oct 2026 16:57:09 +0100 Subject: [PATCH 5/7] Report per-count timings from harness-ractor-mem Write bench_by_ractors from harness-ractor-mem, and choose the per-count summary layout whenever the results carry bench_by_ractors, not only for harness-ractor. The ractor memory benchmarks now get one row per ractor count in the runner's tables. --- harness-ractor-mem/harness.rb | 7 ++++--- lib/benchmark_runner/cli.rb | 2 +- test/benchmark_runner_cli_test.rb | 35 +++++++++++++++++++++++++++++++ 3 files changed, 40 insertions(+), 4 deletions(-) diff --git a/harness-ractor-mem/harness.rb b/harness-ractor-mem/harness.rb index 3af25510..51e1a554 100644 --- a/harness-ractor-mem/harness.rb +++ b/harness-ractor-mem/harness.rb @@ -68,7 +68,7 @@ def run_benchmark(num_itrs_hint, &scenario) puts format("base RSS: %.1f MiB", base_rss / 2**20.0) metrics = Hash.new { |h, count| h[count] = { retained: [], peak: [] } } - times = [] + times = Hash.new { |h, count| h[count] = [] } RACTORS.each do |count| puts "ractors: #{count}" @@ -76,7 +76,7 @@ def run_benchmark(num_itrs_hint, &scenario) while itr < bench_itrs t0 = Process.clock_gettime(Process::CLOCK_MONOTONIC) finish, peak = measure_peak_rss { scenario.call(count) } - times << Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0 + times[count] << Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0 gc_settle retained = get_rss - base_rss finish&.call @@ -109,7 +109,8 @@ def run_benchmark(num_itrs_hint, &scenario) puts format("BENCH_METRIC peak_mib=%.1f", worst_peak / 2**20.0) puts format("BENCH_METRIC worst_ractor_count=%d", worst_retained_count) - return_results([], times, + return_results([], times.values.flatten, + bench_by_ractors: times, ractor_mem_settle: GLOBAL_GC_START ? "global" : "default", ractor_mem_base_rss: base_rss, ractor_mem_medians: medians, diff --git a/lib/benchmark_runner/cli.rb b/lib/benchmark_runner/cli.rb index ed59d2d1..f212942d 100644 --- a/lib/benchmark_runner/cli.rb +++ b/lib/benchmark_runner/cli.rb @@ -181,7 +181,7 @@ def build_output_sections(executable_names, bench_data, bench_harnesses, bench_f def build_output_section(executable_names, bench_data, bench_failures, harness, bench_names) section_data = slice_bench_data(bench_data, bench_names) breakdown = RactorBreakdown.expand(section_data) - use_ractor_layout = harness == BenchmarkSuite::RACTOR_HARNESS && !breakdown.groups.empty? + use_ractor_layout = !breakdown.groups.empty? layout = use_ractor_layout ? RactorRowLayout.new(groups: breakdown.groups) : FlatRowLayout.new display_data = use_ractor_layout ? breakdown.bench_data : section_data diff --git a/test/benchmark_runner_cli_test.rb b/test/benchmark_runner_cli_test.rb index 7baf82f3..9852bdaf 100644 --- a/test/benchmark_runner_cli_test.rb +++ b/test/benchmark_runner_cli_test.rb @@ -195,6 +195,41 @@ def create_args(overrides = {}) assert_equal 'ractor-local-workload', section[:gc_scope], 'the raw blob scope must tag the section; expansion drops gc_scope from counts without GC data' end + + it 'renders per-count timing and GC rows for a ractor memory harness section' do + args = create_args + cli = BenchmarkRunner::CLI.new(args) + gc_group = { + 'gc_count_bench' => [3], + 'gc_major_count_bench' => [1], + 'gc_minor_count_bench' => [2], + 'gc_marking_time_bench' => [1.0], + 'gc_sweeping_time_bench' => [1.0], + 'gc_total_time_bench' => [2.0], + 'gc_worker_samples' => [[{ 'gc_count' => 3, 'worker_index' => 0 }]] + } + bench_data = { + 'ruby' => { + 'ractor-dead-set' => { + 'warmup' => [], + 'bench' => [1.0, 2.0], + 'rss' => 10 * 1024 * 1024, + 'gc_scope' => 'ractor-local-workload', + 'bench_by_ractors' => { '1' => [1.0], '2' => [2.0] }, + 'gc_by_ractors' => { '1' => gc_group, '2' => gc_group } + } + } + } + + sections = cli.send(:build_output_sections, ['ruby'], bench_data, { 'ractor-dead-set' => 'harness-ractor-mem' }, {}) + section = sections.first + + assert_equal ['bench', 'ractors', 'ruby (ms)'], section[:table][0] + assert_equal ['ractor-dead-set', '1', '1000.0 ± 0.0%'], section[:table][1] + assert_equal ['', '2', '2000.0 ± 0.0%'], section[:table][2] + assert_equal ['ractor-dead-set', '1'], section[:gc_table][1][0..1] + assert_equal ['ractor-dead-set', '2'], section[:gc_table][2][0..1] + end end describe '#run integration test' do From 7d30503e39ca0e621024be29ea41cbda5f07197f Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 6 Oct 2026 16:57:52 +0100 Subject: [PATCH 6/7] Collect Ractor GC samples in harness-ractor-mem With --ractor-gc (RUBY_BENCH_RACTOR_GC=1), harness-ractor-mem checks target support before the baseline and writes gc_by_ractors in the same shape as harness-ractor. Scenarios spawn their own workers, so each worker wraps its body in measure_worker_gc and the main Ractor passes the sample to record_worker_gc. A trial fails when its recorded worker indexes are not 0...count. ractor-idle-garbage takes its sample while the set is still live and sends a shareable message, so sampling cannot sweep the garbage before the worker parks. --- README.md | 16 +++- benchmarks/ractor-dead-set/benchmark.rb | 21 +++-- benchmarks/ractor-idle-garbage/benchmark.rb | 20 +++-- benchmarks/ractor-msg-backlog/benchmark.rb | 25 ++++-- harness-ractor-mem/harness.rb | 69 ++++++++++++++- test/ractor_gc_harness_test.rb | 94 ++++++++++++++++++++- 6 files changed, 215 insertions(+), 30 deletions(-) diff --git a/README.md b/README.md index eeb820f7..2ba99ce9 100644 --- a/README.md +++ b/README.md @@ -198,6 +198,19 @@ On a target with Ractor-local GC, a plain `GC.start` collects only the main Ractor's object space. The JSON field `ractor_mem_settle` records `global` or `default`. +With `--ractor-gc` (`RUBY_BENCH_RACTOR_GC=1`), every worker samples its own +Ractor-local GC activity. The results then carry the same `gc_by_ractors` data +as `harness-ractor`. A scenario wraps each worker body in +`measure_worker_gc { ... }`, which returns `[result, sample]`. The main Ractor +passes each sample to `record_worker_gc(worker_index, sample)`. A trial fails +when its recorded worker indexes are not `0...count`. + +Worker samples cover only the workers' own object spaces during the scenario. +They do not include allocation by the main Ractor, such as the payloads that +`ractor-msg-backlog` sends. They also do not include the GCs that the harness +runs to measure retention. The JSON field `gc_controller_samples` covers the +main Ractor during the scenario. + ## Ruby options By default, ruby-bench benchmarks the Ruby used for `run_benchmarks.rb`. @@ -337,7 +350,8 @@ process's lifetime peak from `getrusage`. ## Measuring Ractor GC activity The `--ractor-gc` option of `run_benchmarks.rb` collects Ractor-local GC -metrics for benchmarks that use the Ractor harness (`--category ractor`). +metrics for benchmarks that use `harness-ractor` or `harness-ractor-mem` +(`--category ractor`). The target must use Ruby 4.1 or newer with per-Ractor global GC attribution ([ruby/ruby#19147](https://github.com/ruby/ruby/pull/19147)); older targets fail before warmup. diff --git a/benchmarks/ractor-dead-set/benchmark.rb b/benchmarks/ractor-dead-set/benchmark.rb index 947d1e7b..02a4b8a7 100644 --- a/benchmarks/ractor-dead-set/benchmark.rb +++ b/benchmarks/ractor-dead-set/benchmark.rb @@ -11,17 +11,24 @@ run_benchmark(3) do |num_ractors| workers = num_ractors.times.map do |worker| Ractor.new(worker, DEAD_SET_ITEMS) do |worker_id, items| - keep = [] - i = 0 - while i < items - keep << "worker #{worker_id} item #{i} " + ("y" * 100) - i += 1 + measure_worker_gc do + keep = [] + i = 0 + while i < items + keep << "worker #{worker_id} item #{i} " + ("y" * 100) + i += 1 + end + keep.size end - keep.size end end - total = workers.sum(&:value) + total = 0 + workers.each_with_index do |worker, worker_id| + size, sample = worker.value + record_worker_gc(worker_id, sample) + total += size + end raise "unexpected dead-set size" unless total == num_ractors * DEAD_SET_ITEMS nil end diff --git a/benchmarks/ractor-idle-garbage/benchmark.rb b/benchmarks/ractor-idle-garbage/benchmark.rb index e031a083..d6165832 100644 --- a/benchmarks/ractor-idle-garbage/benchmark.rb +++ b/benchmarks/ractor-idle-garbage/benchmark.rb @@ -12,26 +12,30 @@ run_benchmark(3) do |num_ractors| drained = Ractor.new(num_ractors) do |count| - count.times { Ractor.receive } - :all_drained + Array.new(count) { Ractor.receive } end workers = num_ractors.times.map do |worker| Ractor.new(drained, worker, IDLE_GARBAGE_ITEMS) do |ack, worker_id, items| keep = [] - i = 0 - while i < items - keep << "worker #{worker_id} garbage #{i} " + ("y" * 100) - i += 1 + _, sample = measure_worker_gc do + i = 0 + while i < items + keep << "worker #{worker_id} garbage #{i} " + ("y" * 100) + i += 1 + end end + message = Ractor.make_shareable([worker_id, sample]) keep = nil - ack.send :drained + ack.send message Ractor.receive :worker_done end end - raise "unexpected drain barrier result" unless drained.value == :all_drained + drained_workers = drained.value + raise "unexpected drain barrier result" unless drained_workers.size == num_ractors + drained_workers.each { |worker_id, sample| record_worker_gc(worker_id, sample) } proc do workers.each do |worker| diff --git a/benchmarks/ractor-msg-backlog/benchmark.rb b/benchmarks/ractor-msg-backlog/benchmark.rb index 42bec6e5..72bbf078 100644 --- a/benchmarks/ractor-msg-backlog/benchmark.rb +++ b/benchmarks/ractor-msg-backlog/benchmark.rb @@ -16,14 +16,16 @@ run_benchmark(3) do |num_ractors| consumers = num_ractors.times.map do |consumer| Ractor.new(consumer, BACKLOG_GATE_SLEEP) do |consumer_id, gate| - sleep gate - taken = 0 - loop do - message = Ractor.receive - break if message == :done - taken += 1 + measure_worker_gc do + sleep gate + taken = 0 + loop do + message = Ractor.receive + break if message == :done + taken += 1 + end + taken end - [consumer_id, taken] end end @@ -36,7 +38,12 @@ consumer.send :done end - results = consumers.map(&:value) - raise "unexpected backlog drain" unless results.sum { |_, taken| taken } == num_ractors * BACKLOG_MESSAGES + total = 0 + consumers.each_with_index do |consumer, consumer_id| + taken, sample = consumer.value + record_worker_gc(consumer_id, sample) + total += taken + end + raise "unexpected backlog drain" unless total == num_ractors * BACKLOG_MESSAGES nil end diff --git a/harness-ractor-mem/harness.rb b/harness-ractor-mem/harness.rb index 51e1a554..871e1254 100644 --- a/harness-ractor-mem/harness.rb +++ b/harness-ractor-mem/harness.rb @@ -3,6 +3,9 @@ Warning[:experimental] = false +RACTOR_GC_ENABLED = ENV["RUBY_BENCH_RACTOR_GC"] == "1" +require_relative "../lib/gc_stats" if RACTOR_GC_ENABLED + default_ractors = [1, 2, 4, 6, 8] if rs = ENV["RUBY_BENCH_RACTORS"] rs = rs.split(",").map(&:to_i) @@ -17,6 +20,7 @@ PEAK_SAMPLE_INTERVAL = Float(ENV.fetch("RACTOR_MEM_PEAK_SAMPLE_INTERVAL", 0.005)) GLOBAL_GC_START = GC.method(:start).parameters.include?([:key, :global]) +WORKER_GC_SAMPLES = [] puts RUBY_DESCRIPTION @@ -58,33 +62,65 @@ def measure_peak_rss [result, peak] end +def measure_worker_gc + return [yield, nil] unless RACTOR_GC_ENABLED + + result = nil + sample = GCStats.measure { result = yield } + [result, sample] +end + +def record_worker_gc(worker_index, sample) + return unless sample + + WORKER_GC_SAMPLES << sample.merge("worker_index" => worker_index) +end + def run_benchmark(num_itrs_hint, &scenario) bench_itrs = Integer(ENV.fetch("MIN_BENCH_ITRS", num_itrs_hint)) bench_itrs = MAX_ITERS if bench_itrs > MAX_ITERS bench_itrs = 1 if bench_itrs < 1 + if RACTOR_GC_ENABLED + GCStats.check_ractor_gc_support! + gc_config = GC.config.transform_keys(&:to_s) if GC.respond_to?(:config) + GCStats.with_measure_total_time { run_memory_trials(bench_itrs, gc_config, &scenario) } + else + run_memory_trials(bench_itrs, nil, &scenario) + end +end + +def run_memory_trials(bench_itrs, gc_config, &scenario) gc_settle base_rss = get_rss puts format("base RSS: %.1f MiB", base_rss / 2**20.0) metrics = Hash.new { |h, count| h[count] = { retained: [], peak: [] } } times = Hash.new { |h, count| h[count] = [] } + gc_groups = Hash.new do |h, count| + h[count] = { "gc_worker_samples" => [], "gc_controller_samples" => [] } + end RACTORS.each do |count| puts "ractors: #{count}" itr = 0 while itr < bench_itrs + WORKER_GC_SAMPLES.clear + controller_before = GCStats.snapshot if RACTOR_GC_ENABLED t0 = Process.clock_gettime(Process::CLOCK_MONOTONIC) finish, peak = measure_peak_rss { scenario.call(count) } times[count] << Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0 + controller_sample = GCStats.delta(controller_before, GCStats.snapshot) if RACTOR_GC_ENABLED gc_settle retained = get_rss - base_rss finish&.call gc_settle metrics[count][:retained] << retained metrics[count][:peak] << peak - puts format(" itr #%d: retained %.1f MiB, peak %.1f MiB", + line = format(" itr #%d: retained %.1f MiB, peak %.1f MiB", itr + 1, retained / 2**20.0, peak / 2**20.0) + line << record_trial_gc(gc_groups[count], count, controller_sample) if RACTOR_GC_ENABLED + puts line itr += 1 end end @@ -109,11 +145,38 @@ def run_benchmark(num_itrs_hint, &scenario) puts format("BENCH_METRIC peak_mib=%.1f", worst_peak / 2**20.0) puts format("BENCH_METRIC worst_ractor_count=%d", worst_retained_count) - return_results([], times.values.flatten, + extra = { bench_by_ractors: times, ractor_mem_settle: GLOBAL_GC_START ? "global" : "default", ractor_mem_base_rss: base_rss, ractor_mem_medians: medians, ractor_mem_samples: metrics, - ) + } + if RACTOR_GC_ENABLED + extra[:gc_scope] = "ractor-local-workload" + extra[:gc_stat_scope] = "ractor-local" + extra[:gc_measure_total_time_scope] = "ractor-local" + extra[:gc_by_ractors] = gc_groups + extra[:gc_config] = gc_config if gc_config + end + return_results([], times.values.flatten, **extra) +end + +def record_trial_gc(group, count, controller_sample) + workers = WORKER_GC_SAMPLES.sort_by { |s| s["worker_index"] } + unless workers.map { |s| s["worker_index"] } == (0...count).to_a + raise "scenario recorded worker GC samples #{workers.map { |s| s["worker_index"] }.inspect} for #{count} ractors" + end + + group["gc_worker_samples"] << workers + group["gc_controller_samples"] << controller_sample + agg = GCStats.aggregate(workers) + GCStats::RACTOR_SERIES.each do |series, field| + value = field == GCStats::TOTAL_TIME_FIELD ? agg[field]&.fdiv(1_000_000) : agg[field] + (group[series] ||= []) << value + end + + format(", worker gc %s (minor %s major %s global %s) %s", + agg["gc_count"], agg["gc_minor_count"], agg["gc_major_count"], agg["gc_global_count"], + agg["gc_total_time_ns"] ? format("%.1fms", agg["gc_total_time_ns"] / 1_000_000.0) : "N/A") end diff --git a/test/ractor_gc_harness_test.rb b/test/ractor_gc_harness_test.rb index 7ad82e38..6a991a47 100644 --- a/test/ractor_gc_harness_test.rb +++ b/test/ractor_gc_harness_test.rb @@ -103,6 +103,32 @@ def GCStats.global_gc_attributed?(*) = false puts "workload_ran=#{workload_ran}" RUBY + MEM_WORKLOAD_BODY = <<~'RUBY' + run_benchmark(2) do |count| + workers = count.times.map do |worker| + Ractor.new(worker) do |worker_id| + measure_worker_gc do + 20_000.times { Object.new } + GC.start(full_mark: true, immediate_sweep: true) + worker_id + end + end + end + workers.each_with_index do |worker, worker_id| + _, sample = worker.value + record_worker_gc(worker_id, sample) + end + nil + end + RUBY + + MEM_UNRECORDED_BODY = <<~'RUBY' + run_benchmark(2) do |count| + count.times.map { Ractor.new { 10_000.times { Object.new } } }.each(&:value) + nil + end + RUBY + before do @explicit_target = !ENV['RACTOR_GC_TEST_RUBY'].nil? @ruby = ENV['RACTOR_GC_TEST_RUBY'] || RbConfig.ruby @@ -132,7 +158,7 @@ def GCStats.global_gc_attributed?(*) = false end end - def run_workload(body) + def run_workload(body, harness: 'harness-ractor') Dir.mktmpdir do |dir| result_path = File.join(dir, 'results.json') script = File.join(dir, 'workload.rb') @@ -146,7 +172,7 @@ def run_workload(body) 'MIN_BENCH_TIME' => '0', 'RESULT_JSON_PATH' => result_path ) - stdout, stderr, status = Open3.capture3(env, @ruby, "-I#{File.join(ROOT, 'harness-ractor')}", script, chdir: ROOT) + stdout, stderr, status = Open3.capture3(env, @ruby, "-I#{File.join(ROOT, harness)}", script, chdir: ROOT) yield stdout, stderr, status, result_path end end @@ -294,6 +320,70 @@ def run_workload(body) end end + it 'collects per-worker GC samples from a ractor memory scenario' do + run_workload(MEM_WORKLOAD_BODY, harness: 'harness-ractor-mem') do |stdout, stderr, status, result_path| + assert status.success?, "workload failed:\n#{stdout}\n#{stderr}" + + data = JSON.parse(File.read(result_path)) + assert_equal 'ractor-local-workload', data['gc_scope'] + assert_equal 'global', data['ractor_mem_settle'] + assert_equal %w[1 2], data['bench_by_ractors'].keys.sort + assert_equal %w[1 2], data['gc_by_ractors'].keys.sort + + data['gc_by_ractors'].each do |count, group| + assert_equal 2, data['bench_by_ractors'][count].length, "count #{count} timing samples" + assert_equal 2, group['gc_worker_samples'].length, "count #{count} measured iterations" + assert_equal 2, group['gc_controller_samples'].length, "count #{count} controller samples" + group['gc_worker_samples'].each_with_index do |workers, i| + assert_equal (0...count.to_i).to_a, workers.map { |w| w['worker_index'] }, 'spawn-index order' + workers.each { |w| assert_operator w['gc_count'], :>, 0, 'every worker records GC activity' } + assert_equal workers.sum { |w| w['gc_count'] }, group['gc_count_bench'][i] + assert_in_delta workers.sum { |w| w['gc_total_time_ns'] } / 1_000_000.0, group['gc_total_time_bench'][i], 1e-9 + end + end + end + end + + it 'fails a ractor memory scenario that does not record its worker GC samples' do + run_workload(MEM_UNRECORDED_BODY, harness: 'harness-ractor-mem') do |stdout, stderr, status, result_path| + refute status.success?, "expected an unrecorded scenario to fail:\n#{stdout}\n#{stderr}" + assert_includes stderr, 'scenario recorded worker GC samples [] for 1 ractors' + refute File.exist?(result_path), 'no results file may be written for a failed benchmark' + end + end + + it 'reports every worker of each ractor memory benchmark' do + %w[ractor-dead-set ractor-idle-garbage ractor-msg-backlog].each do |bench| + Dir.mktmpdir do |dir| + result_path = File.join(dir, 'results.json') + env = CLEAN_ENV.merge( + 'RUBY_BENCH_RACTOR_GC' => '1', + 'RUBY_BENCH_RACTORS' => '1,2', + 'MIN_BENCH_ITRS' => '1', + 'MAX_BENCH_ITRS' => '1', + 'RACTOR_MEM_SETTLE_SLEEP' => '0', + 'RACTOR_DEAD_SET_ITEMS' => '5000', + 'RACTOR_IDLE_GARBAGE_ITEMS' => '5000', + 'RACTOR_BACKLOG_MESSAGES' => '500', + 'RACTOR_BACKLOG_GATE_SLEEP' => '0.05', + 'RESULT_JSON_PATH' => result_path + ) + script = File.join(ROOT, 'benchmarks', bench, 'benchmark.rb') + stdout, stderr, status = Open3.capture3(env, @ruby, "-I#{File.join(ROOT, 'harness-ractor-mem')}", script, chdir: ROOT) + assert status.success?, "#{bench} failed:\n#{stdout}\n#{stderr}" + + data = JSON.parse(File.read(result_path)) + assert_equal %w[1 2], data['gc_by_ractors'].keys.sort, bench + data['gc_by_ractors'].each do |count, group| + assert_equal 1, group['gc_controller_samples'].length, "#{bench} count #{count} controller samples" + workers = group['gc_worker_samples'].fetch(0) + assert_equal (0...count.to_i).to_a, workers.map { |w| w['worker_index'] }, "#{bench} count #{count} workers" + workers.each { |w| assert_kind_of Integer, w['gc_count'], "#{bench} count #{count} worker gc_count" } + end + end + end + end + it 'propagates a worker failure without writing partial or dummy results' do run_workload(FAILING_WORKLOAD_BODY) do |stdout, stderr, status, result_path| refute status.success?, "expected worker failure to fail the run:\n#{stdout}\n#{stderr}" From 929e2331172753fedd292ff752fbc102d76f9f99 Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 6 Oct 2026 16:58:26 +0100 Subject: [PATCH 7/7] Describe harness-ractor-mem in the Ractor GC scope note The note said that the main Ractor performs the count-0 workload and that Ractors created by the workload are not sampled. Both are false for harness-ractor-mem, which samples the Ractors each scenario spawns. Record the harness on each output section and print those caveats only for harness-ractor. For harness-ractor-mem, state that main-Ractor allocation and the retention GCs are not in the worker sums. --- lib/benchmark_runner.rb | 14 +++++++++++--- lib/benchmark_runner/cli.rb | 1 + test/benchmark_runner_cli_test.rb | 5 +++++ 3 files changed, 17 insertions(+), 3 deletions(-) diff --git a/lib/benchmark_runner.rb b/lib/benchmark_runner.rb index 38f9222e..a640b3ae 100644 --- a/lib/benchmark_runner.rb +++ b/lib/benchmark_runner.rb @@ -96,12 +96,20 @@ def build_output_text(ruby_descriptions, table, format, bench_failures, include_ output_str << "- controller compacts/iter*: the main Ractor's GC.stat(:compact_count) delta per iteration. Every global compacting cycle increments compact_count in every object space. Do not sum it across workers.#{other_names.empty? ? '' : " Comparison tables show #{base_name} → comparison values."}\n" end - if sections.any? { |section| section[:gc_scope] == 'ractor-local-workload' && section[:gc_table] } + ractor_gc_sections = sections.select { |section| section[:gc_scope] == 'ractor-local-workload' && section[:gc_table] } + unless ractor_gc_sections.empty? + memory_only = ractor_gc_sections.all? { |section| section[:harness] == 'harness-ractor-mem' } output_str << "Ractor GC scope note:\n" - scope_columns = +"- (worker sum) columns add Ractor-local counters across the sampled workers of each iteration; the main Ractor performs the count-0 workload." + scope_columns = +"- (worker sum) columns add Ractor-local counters across the sampled workers of each iteration" + scope_columns << (memory_only ? "." : "; under harness-ractor, the main Ractor performs the count-0 workload.") scope_columns << " GC ms/worker divides each iteration's worker-sum GC time by its sampled worker count, then averages." if other_names.empty? output_str << "#{scope_columns} Controller snapshots and per-worker heap detail are in the JSON output, not this table.\n" - output_str << "- Ruby's Ractor-retirement GC (after a worker's stack is torn down) and Ractors created by the workload itself are not sampled.\n" + output_str << "- Ruby's Ractor-retirement GC (after a worker's stack is torn down) is not sampled." + output_str << " Under harness-ractor, Ractors created by the workload itself are not sampled." unless memory_only + output_str << "\n" + if ractor_gc_sections.any? { |section| section[:harness] == 'harness-ractor-mem' } + output_str << "- Under harness-ractor-mem, the sampled workers are the Ractors that each scenario spawns. Main-Ractor allocation during the scenario and the retention-measurement GCs are not in the worker sums; gc_controller_samples in the JSON output cover the main Ractor during the scenario.\n" + end output_str << "- GC time is CPU-time accounting, not elapsed pause time; summed across Ractors it can exceed wall time.\n" output_str << "- Per-GC ratios divide by recorded GC counts, not complete process-wide GC cycles. Phase times are integer milliseconds; total GC time is kept at nanosecond resolution in the raw worker samples.\n" end diff --git a/lib/benchmark_runner/cli.rb b/lib/benchmark_runner/cli.rb index f212942d..85c49888 100644 --- a/lib/benchmark_runner/cli.rb +++ b/lib/benchmark_runner/cli.rb @@ -197,6 +197,7 @@ def build_output_section(executable_names, bench_data, bench_failures, harness, section = { title: harness, + harness: harness, table: table, format: format, failures: slice_failures(bench_failures, bench_names), diff --git a/test/benchmark_runner_cli_test.rb b/test/benchmark_runner_cli_test.rb index 9852bdaf..dd53351e 100644 --- a/test/benchmark_runner_cli_test.rb +++ b/test/benchmark_runner_cli_test.rb @@ -229,6 +229,11 @@ def create_args(overrides = {}) assert_equal ['', '2', '2000.0 ± 0.0%'], section[:table][2] assert_equal ['ractor-dead-set', '1'], section[:gc_table][1][0..1] assert_equal ['ractor-dead-set', '2'], section[:gc_table][2][0..1] + + output = BenchmarkRunner.build_output_text({ 'ruby' => 'ruby 4.1.0dev' }, nil, nil, {}, sections: sections) + assert_match(/Under harness-ractor-mem, the sampled workers are the Ractors that each scenario spawns/, output) + refute_match(/created by the workload itself are not sampled/, output) + refute_match(/count-0 workload/, output) end end