From ca0fa5bb07432bb6f1ace87e84f0fd55f214933e Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:21:32 +0100 Subject: [PATCH 1/6] Extract shared GC sampling helpers from harness-gc --- harness-gc/harness.rb | 37 ++++------------ harness/gc-stats.rb | 58 ++++++++++++++++++++++++ test/gc_stats_test.rb | 100 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 167 insertions(+), 28 deletions(-) create mode 100644 harness/gc-stats.rb create mode 100644 test/gc_stats_test.rb diff --git a/harness-gc/harness.rb b/harness-gc/harness.rb index 9398fdcf..1e98cd36 100644 --- a/harness-gc/harness.rb +++ b/harness-gc/harness.rb @@ -1,4 +1,5 @@ require_relative "../harness/harness-common" +require_relative "../harness/gc-stats" WARMUP_ITRS = Integer(ENV.fetch('WARMUP_ITRS', 15)) MIN_BENCH_ITRS = Integer(ENV.fetch('MIN_BENCH_ITRS', 10)) @@ -12,24 +13,6 @@ def realtime Process.clock_gettime(Process::CLOCK_MONOTONIC) - r0 end -def gc_stat_heap_snapshot - return {} unless GC.respond_to?(:stat_heap) - GC.stat_heap -end - -def gc_stat_heap_delta(before, after) - delta = {} - after.each do |heap_idx, after_stats| - before_stats = before[heap_idx] || {} - heap_delta = {} - after_stats.each do |key, val| - next unless val.is_a?(Numeric) && before_stats.key?(key) - heap_delta[key] = val - before_stats[key] - end - delta[heap_idx] = heap_delta unless heap_delta.empty? - end - delta -end def run_benchmark(_num_itrs_hint, **, &block) times = [] @@ -56,21 +39,19 @@ def run_benchmark(_num_itrs_hint, **, &block) puts header begin - gc_before = GC.stat - heap_before = gc_stat_heap_snapshot + gc_before = GCStats.snapshot time = realtime(&block) num_itrs += 1 - gc_after = GC.stat - heap_after = gc_stat_heap_snapshot + sample = GCStats.delta(gc_before, GCStats.snapshot) time_ms = (1000 * time).to_i - mark_delta = has_marking ? gc_after[:marking_time] - gc_before[:marking_time] : 0 - sweep_delta = has_sweeping ? gc_after[:sweeping_time] - gc_before[:sweeping_time] : 0 - count_delta = gc_after[:count] - gc_before[:count] - major_delta = gc_after[:major_gc_count] - gc_before[:major_gc_count] - minor_delta = gc_after[:minor_gc_count] - gc_before[:minor_gc_count] + mark_delta = has_marking ? sample["gc_marking_time"] : 0 + sweep_delta = has_sweeping ? sample["gc_sweeping_time"] : 0 + count_delta = sample["gc_count"] + major_delta = sample["gc_major_count"] + minor_delta = sample["gc_minor_count"] ratio_str = minor_delta > 0 ? "%.2f" % (major_delta.to_f / minor_delta) : "-" itr_str = "%4s %6s" % ["##{num_itrs}:", "#{time_ms}ms"] @@ -89,7 +70,7 @@ def run_benchmark(_num_itrs_hint, **, &block) gc_counts << count_delta major_counts << major_delta minor_counts << minor_delta - gc_heap_deltas << gc_stat_heap_delta(heap_before, heap_after) + gc_heap_deltas << sample["gc_stat_heap_delta"] total_time += time end until num_itrs >= WARMUP_ITRS + MIN_BENCH_ITRS and total_time >= MIN_BENCH_TIME diff --git a/harness/gc-stats.rb b/harness/gc-stats.rb new file mode 100644 index 00000000..f16fac6e --- /dev/null +++ b/harness/gc-stats.rb @@ -0,0 +1,58 @@ +# frozen_string_literal: true + +module GCStats + module_function + + SCALAR_FIELDS = [ + ["gc_count", :count], + ["gc_major_count", :major_gc_count], + ["gc_minor_count", :minor_gc_count], + ["gc_marking_time", :marking_time], + ["gc_sweeping_time", :sweeping_time], + ].map(&:freeze).freeze + + TOTAL_TIME_FIELD = "gc_total_time_ns" + + SCALAR_FIELD_NAMES = (SCALAR_FIELDS.map(&:first) + [TOTAL_TIME_FIELD]).freeze + + def heap_snapshot + return {} unless GC.respond_to?(:stat_heap) + GC.stat_heap + end + + def heap_delta(before, after) + delta = {} + after.each do |heap_idx, after_stats| + before_stats = before[heap_idx] || {} + heap_delta = {} + after_stats.each do |key, val| + next unless val.is_a?(Numeric) && before_stats.key?(key) + heap_delta[key] = val - before_stats[key] + end + delta[heap_idx] = heap_delta unless heap_delta.empty? + end + delta + end + + def snapshot + { + stat: GC.stat(scope: :ractor), + heap: heap_snapshot, + total_time_ns: GC.respond_to?(:total_time) ? GC.total_time : nil, + } + end + + def numeric_delta(before, after) + after - before if before.is_a?(Numeric) && after.is_a?(Numeric) + end + + def delta(before, after) + sample = SCALAR_FIELDS.each_with_object({}) do |(name, key), result| + result[name] = numeric_delta(before[:stat][key], after[:stat][key]) + end + sample[TOTAL_TIME_FIELD] = numeric_delta(before[:total_time_ns], after[:total_time_ns]) + sample["gc_stat_heap_delta"] = heap_delta(before[:heap], after[:heap]) + sample["gc_heap_after"] = after[:heap] + sample + end +end diff --git a/test/gc_stats_test.rb b/test/gc_stats_test.rb new file mode 100644 index 00000000..2e069d5f --- /dev/null +++ b/test/gc_stats_test.rb @@ -0,0 +1,100 @@ +require_relative 'test_helper' +require_relative '../harness/gc-stats' + +describe GCStats do + def snapshot(count:, major:, minor:, marking: nil, sweeping: nil, total_ns: nil, heap: {}) + stat = { count: count, major_gc_count: major, minor_gc_count: minor } + stat[:marking_time] = marking unless marking.nil? + stat[:sweeping_time] = sweeping unless sweeping.nil? + { stat: stat, heap: heap, total_time_ns: total_ns } + end + + describe '.delta' do + it 'subtracts before from after instead of reporting final counter values' do + before = snapshot(count: 100, major: 10, minor: 90, marking: 40, sweeping: 60, total_ns: 5_000_000) + after = snapshot(count: 107, major: 11, minor: 96, marking: 43, sweeping: 62, total_ns: 6_500_000) + + sample = GCStats.delta(before, after) + + assert_equal 7, sample['gc_count'] + assert_equal 1, sample['gc_major_count'] + assert_equal 6, sample['gc_minor_count'] + assert_equal 3, sample['gc_marking_time'] + assert_equal 2, sample['gc_sweeping_time'] + assert_equal 1_500_000, sample['gc_total_time_ns'] + end + + it 'returns nil for a field missing from either snapshot rather than a fake zero' do + before = snapshot(count: 1, major: 1, minor: 0) + after = snapshot(count: 3, major: 1, minor: 2, marking: 5) + + sample = GCStats.delta(before, after) + + assert_equal 2, sample['gc_count'] + assert_nil sample['gc_marking_time'] + assert_nil sample['gc_sweeping_time'] + assert_nil sample['gc_total_time_ns'] + end + + it 'keeps a supported zero delta as numeric zero, distinct from unavailable data' do + before = snapshot(count: 5, major: 1, minor: 4, marking: 9, total_ns: 1_000) + after = snapshot(count: 5, major: 1, minor: 4, marking: 9, total_ns: 1_000) + + sample = GCStats.delta(before, after) + + assert_equal 0, sample['gc_count'] + assert_equal 0, sample['gc_marking_time'] + assert_equal 0, sample['gc_total_time_ns'] + end + + it 'preserves sub-millisecond precision in the nanosecond total' do + before = snapshot(count: 0, major: 0, minor: 0, total_ns: 1_000_000) + after = snapshot(count: 1, major: 0, minor: 1, total_ns: 1_250_000) + + assert_equal 250_000, GCStats.delta(before, after)['gc_total_time_ns'] + end + + it 'keeps heap gauge deltas signed and omits heaps/keys absent from the before snapshot' do + before = snapshot(count: 0, major: 0, minor: 0, heap: { + 0 => { slot_size: 40, heap_eden_slots: 100, heap_live_slots: 50 } + }) + after = snapshot(count: 0, major: 0, minor: 0, heap: { + 0 => { slot_size: 40, heap_eden_slots: 120, heap_live_slots: 40, heap_final_slots: 3 }, + 1 => { slot_size: 80, heap_eden_slots: 10 } + }) + + heap_delta = GCStats.delta(before, after)['gc_stat_heap_delta'] + + assert_equal 20, heap_delta[0][:heap_eden_slots] + assert_equal(-10, heap_delta[0][:heap_live_slots]) + refute heap_delta[0].key?(:heap_final_slots), 'keys missing from before must not appear' + refute heap_delta.key?(1), 'heaps missing from before must not appear' + end + + it 'does not difference non-whitelisted stat keys such as process-global counters' do + before = snapshot(count: 0, major: 0, minor: 0) + before[:stat][:page_pool_total_pages] = 100 + before[:stat][:heap_allocatable_pages] = 7 + after = snapshot(count: 1, major: 0, minor: 1) + after[:stat][:page_pool_total_pages] = 250 + after[:stat][:heap_allocatable_pages] = 9 + + sample = GCStats.delta(before, after) + + refute sample.key?('page_pool_total_pages') + refute sample.key?(:page_pool_total_pages) + refute sample.key?('heap_allocatable_pages') + assert_equal GCStats::SCALAR_FIELD_NAMES + %w[gc_stat_heap_delta gc_heap_after], sample.keys + end + + it 'exposes the after-heap snapshot without copying it into the delta' do + before = snapshot(count: 0, major: 0, minor: 0) + after = snapshot(count: 0, major: 0, minor: 0, heap: { 0 => { slot_size: 40 } }) + + sample = GCStats.delta(before, after) + + assert_same after[:heap], sample['gc_heap_after'] + end + end + +end From 8c729fb7cab889d21a3f782084aab2efdba9ac1c Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:23:16 +0100 Subject: [PATCH 2/6] Record GC total time and global GC count in harness-gc Add gc_total_time_warmup/bench in milliseconds and gc_global_count_warmup/bench (on Ruby versions with support for global_gc_count) to the JSON output, plus a global column in the per-iteration stdout table. Ignore this if the benchmarked Ruby doesn't support global gc counts (introduced during Ruby 4.1 dev cycle). --- harness-gc/harness.rb | 19 +++++++++++++++++++ harness/gc-stats.rb | 6 ++++++ 2 files changed, 25 insertions(+) diff --git a/harness-gc/harness.rb b/harness-gc/harness.rb index 1e98cd36..d26134ab 100644 --- a/harness-gc/harness.rb +++ b/harness-gc/harness.rb @@ -22,12 +22,15 @@ def run_benchmark(_num_itrs_hint, **, &block) gc_counts = [] major_counts = [] minor_counts = [] + global_counts = [] gc_heap_deltas = [] + gc_total_time_ns = [] total_time = 0 num_itrs = 0 has_marking = GC.stat.key?(:marking_time) has_sweeping = GC.stat.key?(:sweeping_time) + has_global_gc = GCStats.stat_available?(:global_gc_count) header = "itr: time" header << " marking" if has_marking @@ -35,11 +38,14 @@ def run_benchmark(_num_itrs_hint, **, &block) header << " gc_count" header << " major" header << " minor" + header << " global*" if has_global_gc header << " maj/min" puts header + puts "(* process/controller-observed; may overlap the other GC counts and is not additive.)" if has_global_gc begin gc_before = GCStats.snapshot + global_gc_before = GC.stat(:global_gc_count) if has_global_gc time = realtime(&block) num_itrs += 1 @@ -52,6 +58,7 @@ def run_benchmark(_num_itrs_hint, **, &block) count_delta = sample["gc_count"] major_delta = sample["gc_major_count"] minor_delta = sample["gc_minor_count"] + global_delta = has_global_gc ? GC.stat(:global_gc_count) - global_gc_before : nil ratio_str = minor_delta > 0 ? "%.2f" % (major_delta.to_f / minor_delta) : "-" itr_str = "%4s %6s" % ["##{num_itrs}:", "#{time_ms}ms"] @@ -60,6 +67,7 @@ def run_benchmark(_num_itrs_hint, **, &block) itr_str << " %9d" % count_delta itr_str << " %9d" % major_delta itr_str << " %9d" % minor_delta + itr_str << " %9d" % global_delta if has_global_gc itr_str << "%9s" % ratio_str puts itr_str @@ -70,7 +78,9 @@ def run_benchmark(_num_itrs_hint, **, &block) gc_counts << count_delta major_counts << major_delta minor_counts << minor_delta + global_counts << global_delta if has_global_gc gc_heap_deltas << sample["gc_stat_heap_delta"] + gc_total_time_ns << sample["gc_total_time_ns"] total_time += time end until num_itrs >= WARMUP_ITRS + MIN_BENCH_ITRS and total_time >= MIN_BENCH_TIME @@ -90,7 +100,16 @@ def run_benchmark(_num_itrs_hint, **, &block) extra["gc_major_count_bench"] = major_counts[bench_range] extra["gc_minor_count_warmup"] = minor_counts[warmup_range] extra["gc_minor_count_bench"] = minor_counts[bench_range] + if has_global_gc + extra["gc_global_count_warmup"] = global_counts[warmup_range] + extra["gc_global_count_bench"] = global_counts[bench_range] + end extra["gc_stat_heap_deltas"] = gc_heap_deltas[bench_range] + if gc_total_time_ns.all?(Numeric) + gc_total_time_ms = gc_total_time_ns.map { |ns| ns / 1_000_000.0 } + extra["gc_total_time_warmup"] = gc_total_time_ms[warmup_range] + extra["gc_total_time_bench"] = gc_total_time_ms[bench_range] + end # Snapshot heap utilisation after benchmark if GC.respond_to?(:stat_heap) diff --git a/harness/gc-stats.rb b/harness/gc-stats.rb index f16fac6e..607fa59a 100644 --- a/harness/gc-stats.rb +++ b/harness/gc-stats.rb @@ -15,6 +15,12 @@ module GCStats SCALAR_FIELD_NAMES = (SCALAR_FIELDS.map(&:first) + [TOTAL_TIME_FIELD]).freeze + def stat_available?(key) + GC.stat(key).is_a?(Numeric) + rescue ArgumentError + false + end + def heap_snapshot return {} unless GC.respond_to?(:stat_heap) GC.stat_heap From 890619eb115ecc8b283ef0d1f632b4093a3bd843 Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:23:53 +0100 Subject: [PATCH 3/6] Add opt-in Ractor-local GC collection to harness-ractor enable with RUBY_BENCH_RACTOR_GC=1 The mode requires Ruby 4.1 or newer (because of the Ractor-local GC.stat and per-Ractor GC.measure_total_time); --- harness-ractor/harness.rb | 174 ++++++++++++++++-- harness/gc-stats.rb | 45 ++++- test/gc_stats_test.rb | 78 +++++++++ test/ractor_gc_harness_test.rb | 310 +++++++++++++++++++++++++++++++++ 4 files changed, 587 insertions(+), 20 deletions(-) create mode 100644 test/ractor_gc_harness_test.rb diff --git a/harness-ractor/harness.rb b/harness-ractor/harness.rb index 0592dc80..0c3111cc 100644 --- a/harness-ractor/harness.rb +++ b/harness-ractor/harness.rb @@ -4,6 +4,20 @@ Warning[:experimental] = false ENV["RUBY_BENCH_RACTOR_HARNESS"] = "1" +RACTOR_GC_ENABLED = ENV["RUBY_BENCH_RACTOR_GC"] == "1" +require_relative '../harness/gc-stats' if RACTOR_GC_ENABLED + +CONTROLLER_GC_SERIES = if RACTOR_GC_ENABLED + { + "gc_global_count_bench" => [:global_gc_count, "global*"], + "gc_controller_compact_count_bench" => [:compact_count, "compacts*"], + }.select { |_name, (stat_key, _label)| GCStats.stat_available?(stat_key) } + .transform_values(&:freeze) + .freeze +else + {}.freeze +end + default_ractors = [ 0, # without ractor 1, 2, 4, 6, 8#, 12, 16, 32 @@ -32,27 +46,41 @@ def join def run_benchmark(num_itrs_hint, ractor_args: [], &block) warmup_itrs = Integer(ENV.fetch('WARMUP_ITRS', 5)) bench_itrs = Integer(ENV.fetch('MIN_BENCH_ITRS', num_itrs_hint)) - if bench_itrs > MAX_ITERS - bench_itrs = MAX_ITERS + bench_itrs = MAX_ITERS if bench_itrs > MAX_ITERS + + if RACTOR_GC_ENABLED + check_ractor_gc_support + gc_config = GC.config.transform_keys(&:to_s) if GC.respond_to?(:config) + GCStats.with_measure_total_time do + run_warmup(warmup_itrs, ractor_args, &block) + run_benchmark_gc(bench_itrs, Ractor.make_shareable(block), ractor_args, gc_config: gc_config) + end + else + puts "r: itr: time" + run_warmup(warmup_itrs, ractor_args, &block) + run_benchmark_timing(bench_itrs, ractor_args, &block) end - # { num_ractors => [itr_in_ms, ...] } - stats = Hash.new { |h,k| h[k] = [] } +end - header = "r: itr: time" - puts header +def check_ractor_gc_support + unless GCStats.ractor_local_gc_supported? + raise NotImplementedError, "Ractor GC metrics require Ruby 4.1 or newer" + end + unless GC.respond_to?(:total_time) && GC.respond_to?(:measure_total_time) && GC.respond_to?(:measure_total_time=) + raise NotImplementedError, "Ractor GC metrics require GC.total_time and GC.measure_total_time=" + end +end - i = 0 - while i < warmup_itrs - args = if ractor_args.empty? - [] - else - ractor_deep_dup(ractor_args) - end - block.call *([0] + args) - i += 1 +def run_warmup(warmup_itrs, ractor_args, &block) + warmup_itrs.times do + args = ractor_args.empty? ? [] : ractor_deep_dup(ractor_args) + block.call(*([0] + args)) end +end + +def run_benchmark_timing(bench_itrs, ractor_args, &block) + stats = Hash.new { |h,k| h[k] = [] } - blk = Ractor.make_shareable(block) RACTORS.each do |rs| num_itrs = 0 while num_itrs < bench_itrs @@ -80,6 +108,120 @@ def run_benchmark(num_itrs_hint, ractor_args: [], &block) return_results([], stats.values.flatten, bench_by_ractors: stats) end +RACTOR_GC_SERIES = { + "gc_count_bench" => "gc_count", + "gc_major_count_bench" => "gc_major_count", + "gc_minor_count_bench" => "gc_minor_count", + "gc_marking_time_bench" => "gc_marking_time", + "gc_sweeping_time_bench" => "gc_sweeping_time", + "gc_total_time_bench" => "gc_total_time_ns", +}.freeze + +def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:) + stats = Hash.new { |h,k| h[k] = [] } + gc_by_ractors = {} + + header = +"r: itr: time gc_total marking sweeping gc_count major minor" + CONTROLLER_GC_SERIES.each_value { |(_stat_key, label)| header << " %9s" % label } + puts header + puts "(* process/controller-observed; may overlap the other GC counts and is not additive.)" if CONTROLLER_GC_SERIES.any? + + RACTORS.each do |rs| + group = { "gc_worker_samples" => [] } + group["gc_controller_samples"] = [] if rs > 0 + series = Hash.new { |h,k| h[k] = [] } + + num_itrs = 0 + while num_itrs < bench_itrs + num_itrs += 1 + elapsed, worker_samples, controller_sample, controller_deltas = run_ractor_gc_iteration(rs, ractor_args, &block) + stats[rs] << elapsed + group["gc_worker_samples"] << worker_samples + group["gc_controller_samples"] << controller_sample if controller_sample + + agg = GCStats.aggregate(worker_samples) + total_ms = agg["gc_total_time_ns"]&.fdiv(1_000_000) + RACTOR_GC_SERIES.each do |series_name, field| + series[series_name] << (field == "gc_total_time_ns" ? total_ms : agg[field]) + end + CONTROLLER_GC_SERIES.each_key { |series_name| series[series_name] << controller_deltas[series_name] } + + itr_str = "%-3s %4s %6s" % [rs, "##{num_itrs}:", "#{(1000 * elapsed).to_i}ms"] + itr_str << " %8s" % (total_ms ? "%.1fms" % total_ms : "N/A") + itr_str << " %8s" % (agg["gc_marking_time"] ? "#{agg["gc_marking_time"]}ms" : "N/A") + itr_str << " %8s" % (agg["gc_sweeping_time"] ? "#{agg["gc_sweeping_time"]}ms" : "N/A") + itr_str << " %9s %9s %9s" % [agg["gc_count"], agg["gc_major_count"], agg["gc_minor_count"]].map { |v| v.nil? ? "N/A" : v.to_s } + CONTROLLER_GC_SERIES.each_key { |series_name| itr_str << " %9s" % (controller_deltas[series_name] || "N/A") } + puts itr_str + end + + series.each do |name, values| + group[name] = values + end + gc_by_ractors[rs] = group + end + + extra = { + bench_by_ractors: stats, + gc_scope: "ractor-local-workload", + gc_stat_scope: "ractor-local", + gc_measure_total_time_scope: "ractor-local", + gc_by_ractors: gc_by_ractors, + } + extra[:gc_config] = gc_config if gc_config + return_results([], stats.values.flatten, **extra) +end + +def controller_gc_snapshot + CONTROLLER_GC_SERIES.transform_values { |(stat_key, _label)| GC.stat(stat_key) } +end + +def controller_gc_deltas(before, after) + before.each_with_object({}) do |(series_name, before_value), deltas| + after_value = after[series_name] + deltas[series_name] = after_value - before_value if before_value.is_a?(Numeric) && after_value.is_a?(Numeric) + end +end + +def run_ractor_gc_iteration(num_ractors, ractor_args, &block) + return run_controller_gc_iteration(ractor_args, &block) if num_ractors.zero? + + controller_before = GCStats.snapshot + counters_before = controller_gc_snapshot + started = Process.clock_gettime(Process::CLOCK_MONOTONIC) + + pending = [] + num_ractors.times do |worker_index| + pending << Ractor.new(worker_index, block, num_ractors, *ractor_args) do |index, workload, count, *args| + sample = GCStats.measure(count, *args, &workload) + sample["worker_index"] = index + sample + end + end + + samples = Array.new(num_ractors) + while pending.any? + ractor, worker_sample = Ractor.select(*pending) + pending.delete(ractor) + samples[worker_sample["worker_index"]] = worker_sample + end + + elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started + counters_after = controller_gc_snapshot + controller_sample = GCStats.delta(controller_before, GCStats.snapshot) + [elapsed, samples, controller_sample, controller_gc_deltas(counters_before, counters_after)] +end + +def run_controller_gc_iteration(ractor_args, &block) + counters_before = controller_gc_snapshot + started = Process.clock_gettime(Process::CLOCK_MONOTONIC) + sample = GCStats.measure(0, *ractor_deep_dup(ractor_args), &block) + elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started + counters_after = controller_gc_snapshot + sample["worker_index"] = 0 + [elapsed, [sample], nil, controller_gc_deltas(counters_before, counters_after)] +end + # NOTE: we use `ractor_deep_dup` instead of `Ractor.make_shareable(copy: true)` for the case of # sending args to the block without a ractor because the arguments passed to `run_benchmark` are # sometimes modified, and we want to allow that because it improves compatibility. We don't want diff --git a/harness/gc-stats.rb b/harness/gc-stats.rb index 607fa59a..aa2772bb 100644 --- a/harness/gc-stats.rb +++ b/harness/gc-stats.rb @@ -21,6 +21,11 @@ def stat_available?(key) false end + def ractor_local_gc_supported?(version = RUBY_VERSION) + major, minor = version.split(".").first(2).map(&:to_i) + major > 4 || (major == 4 && minor >= 1) + end + def heap_snapshot return {} unless GC.respond_to?(:stat_heap) GC.stat_heap @@ -40,6 +45,10 @@ def heap_delta(before, after) delta end + def numeric_delta(before, after) + after - before if before.is_a?(Numeric) && after.is_a?(Numeric) + end + def snapshot { stat: GC.stat(scope: :ractor), @@ -48,10 +57,6 @@ def snapshot } end - def numeric_delta(before, after) - after - before if before.is_a?(Numeric) && after.is_a?(Numeric) - end - def delta(before, after) sample = SCALAR_FIELDS.each_with_object({}) do |(name, key), result| result[name] = numeric_delta(before[:stat][key], after[:stat][key]) @@ -61,4 +66,36 @@ def delta(before, after) sample["gc_heap_after"] = after[:heap] sample end + + def with_measure_total_time + has_measure = GC.respond_to?(:measure_total_time) && GC.respond_to?(:measure_total_time=) + saved = GC.measure_total_time if has_measure + GC.measure_total_time = true if has_measure + begin + yield + ensure + GC.measure_total_time = saved if has_measure + end + end + + def measure(*args) + with_measure_total_time do + before = snapshot + started = Process.clock_gettime(Process::CLOCK_MONOTONIC) + yield(*args) + wall_time = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started + after = snapshot + sample = delta(before, after) + sample["wall_time"] = wall_time + sample + end + end + + def aggregate(samples) + raise ArgumentError, "Cannot aggregate an empty GC sample set" if samples.empty? + SCALAR_FIELD_NAMES.each_with_object({}) do |name, result| + values = samples.map { |s| s[name] } + result[name] = values.all? { |v| v.is_a?(Numeric) } ? values.sum : nil + end + end end diff --git a/test/gc_stats_test.rb b/test/gc_stats_test.rb index 2e069d5f..c900c3dc 100644 --- a/test/gc_stats_test.rb +++ b/test/gc_stats_test.rb @@ -97,4 +97,82 @@ def snapshot(count:, major:, minor:, marking: nil, sweeping: nil, total_ns: nil, end end + describe '.aggregate' do + it 'raises ArgumentError for an empty sample set' do + error = assert_raises(ArgumentError) { GCStats.aggregate([]) } + assert_equal 'Cannot aggregate an empty GC sample set', error.message + end + + it 'sums only the whitelisted scalar fields across worker samples' do + samples = [ + { 'gc_count' => 3, 'gc_major_count' => 1, 'gc_minor_count' => 2, 'gc_marking_time' => 4, 'gc_sweeping_time' => 5, 'gc_total_time_ns' => 900_000, 'wall_time' => 0.01, 'worker_index' => 0, 'gc_stat_heap_delta' => { 0 => { heap_eden_slots: 2 } } }, + { 'gc_count' => 4, 'gc_major_count' => 0, 'gc_minor_count' => 4, 'gc_marking_time' => 6, 'gc_sweeping_time' => 7, 'gc_total_time_ns' => 1_100_000, 'wall_time' => 0.02, 'worker_index' => 1, 'gc_stat_heap_delta' => { 1 => { heap_eden_slots: 5 } } } + ] + + total = GCStats.aggregate(samples) + + assert_equal GCStats::SCALAR_FIELD_NAMES.sort, total.keys.sort + assert_equal 7, total['gc_count'] + assert_equal 1, total['gc_major_count'] + assert_equal 6, total['gc_minor_count'] + assert_equal 10, total['gc_marking_time'] + assert_equal 12, total['gc_sweeping_time'] + assert_equal 2_000_000, total['gc_total_time_ns'] + end + + it 'makes a field nil when any participating sample lacks a numeric value' do + samples = [ + { 'gc_count' => 3, 'gc_major_count' => 1, 'gc_minor_count' => 2, 'gc_marking_time' => 4, 'gc_sweeping_time' => 5, 'gc_total_time_ns' => 900 }, + { 'gc_count' => 4, 'gc_major_count' => 0, 'gc_minor_count' => 4, 'gc_marking_time' => nil, 'gc_sweeping_time' => 7, 'gc_total_time_ns' => 100 } + ] + + total = GCStats.aggregate(samples) + + assert_equal 7, total['gc_count'] + assert_nil total['gc_marking_time'] + end + + it 'keeps a supported all-zero aggregate as numeric zero' do + samples = [ + { 'gc_count' => 0, 'gc_major_count' => 0, 'gc_minor_count' => 0, 'gc_marking_time' => 0, 'gc_sweeping_time' => 0, 'gc_total_time_ns' => 0 }, + { 'gc_count' => 0, 'gc_major_count' => 0, 'gc_minor_count' => 0, 'gc_marking_time' => 0, 'gc_sweeping_time' => 0, 'gc_total_time_ns' => 0 } + ] + + total = GCStats.aggregate(samples) + + assert_equal 0, total['gc_count'] + refute_nil total['gc_total_time_ns'] + end + end + + describe '.ractor_local_gc_supported?' do + it 'rejects Ruby 4.0 builds' do + refute GCStats.ractor_local_gc_supported?('4.0.6') + end + + it 'accepts Ruby 4.1 and newer, comparing version parts as integers' do + assert GCStats.ractor_local_gc_supported?('4.1.0dev') + assert GCStats.ractor_local_gc_supported?('4.10.0') + assert GCStats.ractor_local_gc_supported?('5.0.0') + end + + it 'defaults to the running Ruby version' do + assert_equal GCStats.ractor_local_gc_supported?(RUBY_VERSION), GCStats.ractor_local_gc_supported? + end + end + + describe '.measure' do + it 'restores the GC measurement setting when the workload raises' do + skip 'target lacks GC.measure_total_time=' unless GC.respond_to?(:measure_total_time) && GC.respond_to?(:measure_total_time=) + + saved = GC.measure_total_time + begin + GC.measure_total_time = false + assert_raises(RuntimeError) { GCStats.measure { raise 'boom' } } + assert_equal false, GC.measure_total_time + ensure + GC.measure_total_time = saved + end + end + end end diff --git a/test/ractor_gc_harness_test.rb b/test/ractor_gc_harness_test.rb new file mode 100644 index 00000000..ef28793e --- /dev/null +++ b/test/ractor_gc_harness_test.rb @@ -0,0 +1,310 @@ +require_relative 'test_helper' +require_relative '../harness/gc-stats' +require 'open3' +require 'tmpdir' +require 'json' +require 'rbconfig' + +describe 'Ractor GC harness' do + ROOT = File.expand_path('..', __dir__) + + CLEAN_ENV = { + 'RUBYOPT' => nil, + 'RUBYLIB' => nil, + 'BUNDLE_GEMFILE' => nil, + 'BUNDLER_SETUP' => nil, + }.freeze + + VERSION_PROBE = <<~'RUBY' + puts RUBY_VERSION + RUBY + + WORKLOAD_BODY = <<~'RUBY' + GC.measure_total_time = false + run_benchmark(2, ractor_args: [{ seen: [] }]) do |count, payload| + raise "unexpected worker count #{count.inspect}" unless [0, 1, 2].include?(count) + raise "ractor_args not copied per call: #{payload[:seen].inspect}" unless payload[:seen].empty? + payload[:seen] << count + 20_000.times { Object.new } + GC.start(full_mark: true, immediate_sweep: true) + GC.start(full_mark: true, immediate_sweep: true) + end + puts "restored=#{GC.measure_total_time == false}" + RUBY + + FAILING_WORKLOAD_BODY = <<~'RUBY' + GC.measure_total_time = false + begin + run_benchmark(2) do |count| + raise "worker #{count} boom" if count > 0 + 10_000.times { Object.new } + end + rescue StandardError => e + puts "failed: #{e.class}" + end + puts "restored=#{GC.measure_total_time == false}" + exit(1) + RUBY + + COMPACT_WORKLOAD_BODY = <<~'RUBY' + run_benchmark(2) { |_count| GC.compact } + RUBY + + UNSUPPORTED_API_BODY = <<~'RUBY' + class << GC + undef_method :total_time + end + workload_ran = false + begin + run_benchmark(2) { |_count| workload_ran = true } + rescue NotImplementedError => e + puts "raised: #{e.message}" + end + puts "workload_ran=#{workload_ran}" + RUBY + + FAILING_COUNT_ZERO_BODY = <<~'RUBY' + GC.measure_total_time = false + $count_zero_calls = 0 + begin + run_benchmark(2) do |count| + if count.zero? + $count_zero_calls += 1 + raise "controller boom" if $count_zero_calls > 1 + end + 10_000.times { Object.new } + end + rescue StandardError => e + puts "failed: #{e.class}" + end + puts "restored=#{GC.measure_total_time == false}" + exit(1) + RUBY + + UNSUPPORTED_VERSION_BODY = <<~'RUBY' + def GCStats.ractor_local_gc_supported?(*) = false + workload_ran = false + begin + run_benchmark(2) { |_count| workload_ran = true } + rescue NotImplementedError => e + puts "raised: #{e.message}" + end + puts "workload_ran=#{workload_ran}" + RUBY + + before do + @explicit_target = !ENV['RACTOR_GC_TEST_RUBY'].nil? + @ruby = ENV['RACTOR_GC_TEST_RUBY'] || RbConfig.ruby + begin + out, err, status = Open3.capture3(CLEAN_ENV, @ruby, '--disable-gems', '-e', VERSION_PROBE) + rescue SystemCallError => e + flunk("RACTOR_GC_TEST_RUBY target failed to execute: #{e.message}") if @explicit_target + skip("test ruby #{@ruby} is not executable") + end + + unless status.success? + detail = "target #{@ruby} failed the version probe (exit #{status.exitstatus}): #{err.strip}" + flunk("RACTOR_GC_TEST_RUBY #{detail}") if @explicit_target + skip("test ruby #{detail}") + end + + unless GCStats.ractor_local_gc_supported?(out.strip) + flunk("RACTOR_GC_TEST_RUBY target #{@ruby} is Ruby #{out.strip}; Ractor GC metrics require Ruby 4.1 or newer") if @explicit_target + skip("test ruby #{@ruby} is Ruby #{out.strip} (< 4.1); set RACTOR_GC_TEST_RUBY to a Ruby 4.1 or newer build") + end + end + + def run_workload(body) + Dir.mktmpdir do |dir| + result_path = File.join(dir, 'results.json') + script = File.join(dir, 'workload.rb') + File.write(script, "Warning[:experimental] = false\nrequire #{File.join(ROOT, 'harness', 'loader').inspect}\n" + body) + env = CLEAN_ENV.merge( + 'RUBY_BENCH_RACTOR_GC' => '1', + 'RUBY_BENCH_RACTORS' => '0,1,2', + 'WARMUP_ITRS' => '1', + 'MIN_BENCH_ITRS' => '2', + 'MAX_BENCH_ITRS' => '2', + 'MIN_BENCH_TIME' => '0', + 'RESULT_JSON_PATH' => result_path + ) + stdout, stderr, status = Open3.capture3(env, @ruby, "-I#{File.join(ROOT, 'harness-ractor')}", script, chdir: ROOT) + yield stdout, stderr, status, result_path + end + end + + it 'collects per-worker GC samples for counts 0, 1, and 2' do + run_workload(WORKLOAD_BODY) do |stdout, stderr, status, result_path| + assert status.success?, "workload failed:\n#{stdout}\n#{stderr}" + + assert_includes stdout, 'restored=true' + + data = JSON.parse(File.read(result_path)) + assert_equal 'ractor-local-workload', data['gc_scope'] + assert_equal 'ractor-local', data['gc_stat_scope'] + assert_equal 'ractor-local', data['gc_measure_total_time_scope'] + assert_kind_of Hash, data['gc_config'] + assert_equal %w[0 1 2], data['bench_by_ractors'].keys.sort + assert_equal %w[0 1 2], data['gc_by_ractors'].keys.sort + + controller_series_present = data['gc_by_ractors'].values.any? do |group| + group.key?('gc_global_count_bench') || group.key?('gc_controller_compact_count_bench') + end + if controller_series_present + assert_includes stdout, '(* process/controller-observed; may overlap the other GC counts and is not additive.)', + 'a starred stdout column must always come with the exact legend line' + end + + expected_worker_keys = %w[ + gc_count gc_major_count gc_minor_count gc_marking_time gc_sweeping_time + gc_total_time_ns worker_index wall_time gc_stat_heap_delta gc_heap_after + ].sort + + data['gc_by_ractors'].each do |count, group| + assert_equal 2, data['bench_by_ractors'][count].length, "count #{count} timing samples (warmup excluded)" + assert_equal 2, group['gc_worker_samples'].length, "count #{count} measured iterations" + + %w[ + gc_count_bench gc_major_count_bench gc_minor_count_bench + gc_marking_time_bench gc_sweeping_time_bench gc_total_time_bench + ].each do |series| + assert_equal 2, group[series].length, "count #{count} #{series}" + end + %w[gc_global_count_bench gc_controller_compact_count_bench].each do |series| + next unless group.key?(series) + assert_equal 2, group[series].length, "count #{count} #{series}" + group[series].each do |delta| + assert_kind_of Integer, delta + assert_operator delta, :>=, 0 + end + end + + expected_workers = [count.to_i, 1].max + group['gc_worker_samples'].each_with_index do |workers, i| + assert_equal expected_workers, workers.length, "count #{count} iteration #{i} worker records" + assert_equal (0...expected_workers).to_a, workers.map { |w| w['worker_index'] }, 'spawn-index order' + + workers.each do |w| + assert_equal expected_worker_keys, w.keys.sort + assert_operator w['gc_count'], :>, 0, 'every worker records GC activity' + assert_operator w['gc_total_time_ns'], :>, 0 + assert_kind_of Hash, w['gc_heap_after'], 'worker heaps survive worker termination' + refute_empty w['gc_heap_after'] + end + controller_keys = %w[gc_global_count_bench gc_controller_compact_count_bench global_gc_count compact_count] + workers.each do |w| + controller_keys.each do |key| + refute w.key?(key), "worker sample must not carry controller-observed #{key}" + end + end + + assert_equal workers.sum { |w| w['gc_count'] }, group['gc_count_bench'][i] + assert_equal workers.sum { |w| w['gc_major_count'] }, group['gc_major_count_bench'][i] + assert_equal workers.sum { |w| w['gc_minor_count'] }, group['gc_minor_count_bench'][i] + ns_sum = workers.sum { |w| w['gc_total_time_ns'] } + assert_in_delta ns_sum / 1_000_000.0, group['gc_total_time_bench'][i], 1e-9 + end + + if count == '0' + refute group.key?('gc_controller_samples'), 'count 0 must not double-count the main Ractor' + else + assert_equal 2, group['gc_controller_samples'].length + group['gc_controller_samples'].each do |controller| + refute controller.key?('worker_index') + refute controller.key?('wall_time') + end + end + end + end + end + + it 'excludes collections in another Ractor from a workload sample' do + body = <<~'RUBY' + Warning[:experimental] = false + GC.disable + worker = Ractor.new do + GC.disable + Ractor.receive + before = GC.stat(:count, scope: :ractor) + 10.times { GC.start(full_mark: false, immediate_sweep: true, global: false) } + Ractor.main.send(GC.stat(:count, scope: :ractor) - before) + Ractor.receive + end + + foreign_count = nil + sample = GCStats.measure do + worker.send(:start) + foreign_count = Ractor.receive + end + puts [foreign_count, sample['gc_count'], sample['gc_major_count'], sample['gc_minor_count'], sample['gc_total_time_ns']].inspect + worker.send(:stop) + Ractor.select(worker) + RUBY + stdout, stderr, status = Open3.capture3( + CLEAN_ENV, @ruby, '--disable-gems', '-r', File.join(ROOT, 'harness', 'gc-stats'), '-e', body + ) + + assert status.success?, "scope probe failed:\n#{stdout}\n#{stderr}" + assert_equal [10, 0, 0, 0, 0], JSON.parse(stdout) + end + + it 'records controller-observed compaction deltas for a GC.compact workload' do + run_workload(COMPACT_WORKLOAD_BODY) do |stdout, stderr, status, result_path| + assert status.success?, "workload failed:\n#{stdout}\n#{stderr}" + + data = JSON.parse(File.read(result_path)) + unless data.dig('gc_by_ractors', '0').key?('gc_controller_compact_count_bench') + skip("target #{@ruby} does not expose GC.stat(:compact_count)") + end + + data['gc_by_ractors'].each do |count, group| + deltas = group['gc_controller_compact_count_bench'] + assert_equal 2, deltas.length, "count #{count} compaction deltas" + if count == '0' + assert_equal [1, 1], deltas, 'count 0 compacts exactly once per iteration in the main Ractor' + else + deltas.each do |delta| + assert_operator delta, :>=, 1, "count #{count} must observe at least one compacting cycle" + assert_operator delta, :<=, count.to_i, "count #{count} must not multiply-count compacting cycles" + end + end + end + end + end + + it 'propagates a worker failure without writing partial or dummy results' do + run_workload(FAILING_WORKLOAD_BODY) do |stdout, stderr, status, result_path| + refute status.success?, "expected worker failure to fail the run:\n#{stdout}\n#{stderr}" + assert_includes stdout, 'failed: Ractor::RemoteError' + assert_includes stdout, 'restored=true', 'outer ensure must restore the main-Ractor setting on worker failure' + refute File.exist?(result_path), 'no results file may be written for a failed benchmark' + end + end + + it 'restores the measurement setting when the count-0 workload fails' do + run_workload(FAILING_COUNT_ZERO_BODY) do |stdout, stderr, status, result_path| + refute status.success?, "expected count-0 failure to fail the run:\n#{stdout}\n#{stderr}" + assert_includes stdout, 'failed: RuntimeError' + assert_includes stdout, 'restored=true', 'both ensures must restore the main-Ractor setting on count-0 failure' + refute File.exist?(result_path), 'no results file may be written for a failed benchmark' + end + end + + it 'fails before warmup when the target Ruby is older than 4.1' do + run_workload(UNSUPPORTED_VERSION_BODY) do |stdout, stderr, status, result_path| + assert status.success?, "probe script itself failed:\n#{stdout}\n#{stderr}" + assert_includes stdout, 'raised: Ractor GC metrics require Ruby 4.1 or newer' + assert_includes stdout, 'workload_ran=false' + refute File.exist?(result_path), 'no results file may be written for an unsupported target' + end + end + + it 'fails before warmup when the target lacks the required GC APIs' do + run_workload(UNSUPPORTED_API_BODY) do |stdout, stderr, status, result_path| + assert status.success?, "probe script itself failed:\n#{stdout}\n#{stderr}" + assert_includes stdout, 'raised: Ractor GC metrics require GC.total_time and GC.measure_total_time=' + assert_includes stdout, 'workload_ran=false' + refute File.exist?(result_path), 'no results file may be written for an unsupported target' + end + end +end From 0cc813b501668ef7debc8f5ed42c789b6ef2795e Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:24:46 +0100 Subject: [PATCH 4/6] Merge per-count GC data into ractor breakdown blobs --- lib/ractor_breakdown.rb | 8 +++- test/ractor_breakdown_test.rb | 77 +++++++++++++++++++++++++++++++++++ 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/lib/ractor_breakdown.rb b/lib/ractor_breakdown.rb index ac5306d4..9a849595 100644 --- a/lib/ractor_breakdown.rb +++ b/lib/ractor_breakdown.rb @@ -42,9 +42,15 @@ def expand(bench_data) end def per_count_blob(blob, breakdown, count) - per_count = blob.reject { |k, _| k == 'bench_by_ractors' || k == 'bench' } + per_count = blob.reject { |k, _| k == 'bench_by_ractors' || k == 'gc_by_ractors' || k == 'bench' } per_count['bench'] = breakdown[count.to_s] per_count['warmup'] = [] + gc_by_ractors = blob['gc_by_ractors'] + if gc_by_ractors.is_a?(Hash) && gc_by_ractors.key?(count.to_s) + per_count.merge!(gc_by_ractors[count.to_s]) + else + per_count.delete('gc_scope') + end per_count end end diff --git a/test/ractor_breakdown_test.rb b/test/ractor_breakdown_test.rb index c7f6b397..11e6335d 100644 --- a/test/ractor_breakdown_test.rb +++ b/test/ractor_breakdown_test.rb @@ -87,5 +87,82 @@ # groups computed once, not duplicated per executable assert_equal 1, result.groups.size end + + it 'merges only the matching count\'s gc_by_ractors entry into each synthetic blob' do + blob = { + 'bench' => [3.0], + 'bench_by_ractors' => { '0' => [1.0], '2' => [2.0] }, + 'gc_scope' => 'ractor-local-workload', + 'gc_by_ractors' => { + '0' => { + 'gc_count_bench' => [10], + 'gc_total_time_bench' => [4.0], + 'gc_worker_samples' => [[{ 'gc_count' => 10 }]] + }, + '2' => { + 'gc_count_bench' => [99], + 'gc_total_time_bench' => [12.0], + 'gc_worker_samples' => [[{ 'gc_count' => 50 }, { 'gc_count' => 49 }]], + 'gc_controller_samples' => [{ 'gc_count' => 1 }] + } + }, + 'rss' => 555 + } + + blob['gc_stat_scope'] = 'ractor-local' + blob['gc_measure_total_time_scope'] = 'ractor-local' + blob['gc_config'] = { 'implementation' => 'default' } + + result = RactorBreakdown.expand({ 'ruby' => { 'r' => blob } }) + exe = result.bench_data['ruby'] + key0 = "r\x000" + key2 = "r\x002" + + assert_equal [10], exe[key0]['gc_count_bench'] + assert_equal [4.0], exe[key0]['gc_total_time_bench'] + assert_equal [99], exe[key2]['gc_count_bench'] + assert_equal [12.0], exe[key2]['gc_total_time_bench'] + refute exe[key0].key?('gc_controller_samples') + assert_equal [{ 'gc_count' => 1 }], exe[key2]['gc_controller_samples'] + + refute exe[key0].key?('gc_by_ractors') + refute exe[key2].key?('gc_by_ractors') + refute exe[key0].key?('bench_by_ractors') + assert_equal 'ractor-local-workload', exe[key0]['gc_scope'] + assert_equal 'ractor-local-workload', exe[key2]['gc_scope'] + + [key0, key2].each do |key| + assert_equal 'ractor-local', exe[key]['gc_stat_scope'] + assert_equal 'ractor-local', exe[key]['gc_measure_total_time_scope'] + assert_equal({ 'implementation' => 'default' }, exe[key]['gc_config']) + end + assert_equal 555, exe[key2]['rss'] + end + + it 'produces the old timing-only per-count blob when a count lacks GC data' do + bench_data = { + 'ruby' => { + 'r' => { + 'bench' => [1.0], + 'bench_by_ractors' => { '0' => [1.0] }, + 'gc_scope' => 'ractor-local-workload', + 'gc_stat_scope' => 'ractor-local', + 'gc_measure_total_time_scope' => 'ractor-local', + 'gc_config' => { 'implementation' => 'default' } + } + } + } + + result = RactorBreakdown.expand(bench_data) + per_count = result.bench_data['ruby']["r\x000"] + + assert_equal [1.0], per_count['bench'] + refute per_count.key?('gc_count_bench') + refute per_count.key?('gc_by_ractors') + refute per_count.key?('gc_scope') + assert_equal 'ractor-local', per_count['gc_stat_scope'] + assert_equal 'ractor-local', per_count['gc_measure_total_time_scope'] + assert_equal({ 'implementation' => 'default' }, per_count['gc_config']) + end end end From d5846a8fdcd132dcbb5732968e9d22bd7aa5753d Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:25:04 +0100 Subject: [PATCH 5/6] Render explicit GC scopes in the summary tables --- lib/results_table_builder.rb | 226 ++++++++++++++++----- test/results_table_builder_test.rb | 316 ++++++++++++++++++++++++++++- 2 files changed, 489 insertions(+), 53 deletions(-) diff --git a/lib/results_table_builder.rb b/lib/results_table_builder.rb index eccd9b86..ccead21f 100644 --- a/lib/results_table_builder.rb +++ b/lib/results_table_builder.rb @@ -26,6 +26,12 @@ def include_gc? @include_gc end + def self.ractor_gc_data?(bench_data) + bench_data.values.any? do |benchmarks| + benchmarks.values.any? { |d| d.is_a?(Hash) && d['gc_scope'] == 'ractor-local-workload' } + end + end + def build table = [build_header] format = build_format @@ -105,33 +111,128 @@ def build_format format end + GC_SERIES_KEYS = %w[ + gc_count_bench + gc_major_count_bench + gc_minor_count_bench + gc_marking_time_bench + gc_sweeping_time_bench + gc_total_time_bench + gc_global_count_bench + gc_controller_compact_count_bench + ].freeze + def build_gc_summary_table - return nil unless @include_gc && !@other_names.empty? + return nil unless @include_gc - rows = [["bench", *(["comparison"] if include_gc_comparison_name?), "mark/iter ratio", "sweep/iter ratio", "mark/GC ratio", "sweep/GC ratio", "major/iter", "minor/iter", "minor GC %"]] + label_columns = ["bench", *@row_layout.extra_header_columns] - @bench_names.each do |bench_name| - next unless has_complete_data?(bench_name) + if @other_names.empty? + return build_gc_absolute_table(label_columns) + end - marking_times = extract_gc_times(bench_name, 'gc_marking_time_bench') - sweeping_times = extract_gc_times(bench_name, 'gc_sweeping_time_bench') - major_counts = extract_gc_times(bench_name, 'gc_major_count_bench') - minor_counts = extract_gc_times(bench_name, 'gc_minor_count_bench') - base_mark, *other_marks = marking_times - base_sweep, *other_sweeps = sweeping_times - base_major, *other_majors = major_counts - base_minor, *other_minors = minor_counts + count_suffix = ractor_gc_table? ? " (worker sum)" : "" + + header = label_columns + (include_gc_comparison_name? ? ["comparison"] : []) + header += ["gc/iter ratio", "gc/GC ratio"] if include_gc_total_time? + header += ["mark/iter ratio", "sweep/iter ratio", "mark/GC ratio", "sweep/GC ratio"] + header << "global/iter ratio*" if gc_series_present?('gc_global_count_bench') + header << "GCs/iter#{count_suffix}" << "major/iter#{count_suffix}" << "minor/iter#{count_suffix}" + header << "controller compacts/iter*" if gc_series_present?('gc_controller_compact_count_bench') + header << "minor GC %" + + rows = [header] + gc_entries.each do |entry| + series_by_exe = @executable_names.map do |name| + data = bench_data_for(name, entry.data_key) + { + total: data['gc_total_time_bench'], + mark: data['gc_marking_time_bench'], + sweep: data['gc_sweeping_time_bench'], + major: data['gc_major_count_bench'], + minor: data['gc_minor_count_bench'], + count: gc_count_series(data), + global: data['gc_global_count_bench'], + compact: data['gc_controller_compact_count_bench'], + } + end - @other_names.each_with_index do |name, i| - next unless gc_activity?(base_mark, other_marks[i], base_sweep, other_sweeps[i], base_major, other_majors[i], base_minor, other_minors[i]) + base, *others = series_by_exe + others.each_with_index do |other, i| + next unless gc_activity?(*base.values, *other.values) - rows << gc_summary_row(bench_name, name, base_mark, other_marks[i], base_sweep, other_sweeps[i], base_major, other_majors[i], base_minor, other_minors[i]) + rows << gc_summary_row(gc_label_cells(entry), @other_names[i], base, other) end end rows.size == 1 ? nil : rows end + def build_gc_absolute_table(label_columns) + ractor = ractor_gc_table? + count_suffix = ractor ? " (worker sum)" : "" + + header = label_columns + ["GC ms/iter#{count_suffix}"] + header << "GC ms/worker" if ractor + header += ["mark ms/iter#{count_suffix}", "sweep ms/iter#{count_suffix}", "GCs/iter#{count_suffix}", "major/iter#{count_suffix}", "minor/iter#{count_suffix}"] + header << "global GCs/iter*" if gc_series_present?('gc_global_count_bench') + header << "controller compacts/iter*" if gc_series_present?('gc_controller_compact_count_bench') + + rows = [header] + + gc_entries.each do |entry| + data = bench_data_for(@base_name, entry.data_key) + next unless GC_SERIES_KEYS.any? { |key| data.key?(key) } + + cells = [format_gc_series_mean_precise(data['gc_total_time_bench'])] + cells << gc_ms_per_worker_cell(data['gc_total_time_bench'], data['gc_worker_samples']) if ractor + cells += [ + format_gc_series_mean_precise(data['gc_marking_time_bench']), + format_gc_series_mean_precise(data['gc_sweeping_time_bench']), + format_gc_series_mean(gc_count_series(data)), + format_gc_series_mean(data['gc_major_count_bench']), + format_gc_series_mean(data['gc_minor_count_bench']), + ] + cells << format_gc_series_mean(data['gc_global_count_bench']) if gc_series_present?('gc_global_count_bench') + cells << format_gc_series_mean(data['gc_controller_compact_count_bench']) if gc_series_present?('gc_controller_compact_count_bench') + rows << gc_label_cells(entry) + cells + end + + rows.size == 1 ? nil : rows + end + + def gc_series_present?(key) + @gc_series_present ||= {} + unless @gc_series_present.key?(key) + @gc_series_present[key] = @bench_data.values.any? do |benchmarks| + benchmarks.values.any? { |d| d.is_a?(Hash) && d.key?(key) } + end + end + @gc_series_present[key] + end + + def ractor_gc_table? + return @ractor_gc_table if defined?(@ractor_gc_table) + + @ractor_gc_table = ResultsTableBuilder.ractor_gc_data?(@bench_data) + end + + def gc_entries + @row_layout.entries(@bench_names).select { |entry| has_complete_data?(entry.data_key) } + end + + def gc_label_cells(entry) + [@row_layout.base_name(entry.data_key), *entry.label_cells.drop(1)] + end + + def include_gc_total_time? + return @include_gc_total_time if defined?(@include_gc_total_time) + + @include_gc_total_time = @bench_data.values.any? do |benchmarks| + benchmarks.values.any? { |d| d.is_a?(Hash) && d.key?('gc_total_time_bench') } + end + end + def build_gc_summary_format(gc_table) return nil unless gc_table @@ -215,32 +316,62 @@ def include_gc_comparison_name? @other_names.size > 1 end - def gc_summary_row(bench_name, name, base_mark, other_mark, base_sweep, other_sweep, base_major, other_major, base_minor, other_minor) - row = [bench_name] + def gc_summary_row(label_cells, name, base, other) + row = label_cells row << name if include_gc_comparison_name? - row.concat([ - gc_ratio(base_mark, other_mark), - gc_ratio(base_sweep, other_sweep), - scalar_ratio(gc_time_per_gc(base_mark, base_major, base_minor), gc_time_per_gc(other_mark, other_major, other_minor)), - scalar_ratio(gc_time_per_gc(base_sweep, base_major, base_minor), gc_time_per_gc(other_sweep, other_major, other_minor)), - gc_count_cell(base_major, other_major), - gc_count_cell(base_minor, other_minor), - gc_minor_percent_cell(base_major, base_minor, other_major, other_minor), - ]) + if include_gc_total_time? + row << gc_ratio(base[:total], other[:total]) + row << scalar_ratio(gc_time_per_gc(base[:total], base[:count]), gc_time_per_gc(other[:total], other[:count])) + end + row << gc_ratio(base[:mark], other[:mark]) + row << gc_ratio(base[:sweep], other[:sweep]) + row << scalar_ratio(gc_time_per_gc(base[:mark], base[:count]), gc_time_per_gc(other[:mark], other[:count])) + row << scalar_ratio(gc_time_per_gc(base[:sweep], base[:count]), gc_time_per_gc(other[:sweep], other[:count])) + row << gc_ratio(base[:global], other[:global]) if gc_series_present?('gc_global_count_bench') + row << gc_count_cell(base[:count], other[:count]) + row << gc_count_cell(base[:major], other[:major]) + row << gc_count_cell(base[:minor], other[:minor]) + row << gc_count_cell(base[:compact], other[:compact]) if gc_series_present?('gc_controller_compact_count_bench') + row << gc_minor_percent_cell(base[:major], base[:minor], other[:major], other[:minor]) row end - def gc_time_per_gc(time, major, minor) - return nil if time.nil? || time.empty? || major.nil? || major.empty? || minor.nil? || minor.empty? + def numeric_series?(values) + values.is_a?(Array) && !values.empty? && values.all? { |v| v.is_a?(Numeric) } + end + + def gc_count_series(data) + return data['gc_count_bench'] if data.key?('gc_count_bench') + + major = data['gc_major_count_bench'] + minor = data['gc_minor_count_bench'] + return nil unless numeric_series?(major) && numeric_series?(minor) && major.length == minor.length + + major.zip(minor).map { |a, b| a + b } + end + + def gc_time_per_gc(time, count) + return nil unless numeric_series?(time) && numeric_series?(count) && time.length == count.length - count = mean(major) + mean(minor) - return nil if count == 0.0 + count_mean = mean(count) + return nil if count_mean == 0.0 - mean(time) / count + mean(time) / count_mean end def gc_activity?(*series) - series.any? { |values| values && values.sum > 0.0 } + series.any? do |values| + values.is_a?(Array) && values.any? { |v| v.is_a?(Numeric) && v > 0.0 } + end + end + + def gc_ms_per_worker_cell(totals, worker_samples) + return "N/A" unless totals.is_a?(Array) && worker_samples.is_a?(Array) && totals.length == worker_samples.length + + pairs = totals.zip(worker_samples) + return "N/A" unless pairs.all? { |total, workers| total.is_a?(Numeric) && workers.is_a?(Array) && !workers.empty? } + + "%.3f" % mean(pairs.map { |total, workers| total.fdiv(workers.length) }) end def gc_count_cell(base, other) @@ -252,7 +383,7 @@ def gc_minor_percent_cell(base_major, base_minor, other_major, other_minor) end def gc_minor_percent(major, minor) - return nil if major.nil? || major.empty? || minor.nil? || minor.empty? + return nil unless numeric_series?(major) && numeric_series?(minor) && major.length == minor.length total = major.sum + minor.sum return nil if total == 0.0 @@ -261,21 +392,21 @@ def gc_minor_percent(major, minor) end def format_gc_series_mean(values) - return "N/A" if values.nil? || values.empty? + return "N/A" unless numeric_series?(values) "%.1f" % mean(values) end - def format_gc_scalar(value) + def format_gc_percent(value) return "N/A" if value.nil? - "%.3f" % value + "%.0f%%" % (100.0 * value) end - def format_gc_percent(value) - return "N/A" if value.nil? + def format_gc_series_mean_precise(values) + return "N/A" unless numeric_series?(values) - "%.0f%%" % (100.0 * value) + "%.3f" % mean(values) end def scalar_ratio(base, other) @@ -285,10 +416,9 @@ def scalar_ratio(base, other) end def gc_ratio(base, other) - if base.nil? || base.empty? || other.nil? || other.empty? || - mean(other) == 0.0 - return "N/A" - end + return "N/A" unless numeric_series?(base) && numeric_series?(other) + return "N/A" if mean(other) == 0.0 + pval = @include_pvalue ? Stats.welch_p_value(base, other) : nil format_ratio(mean(base) / mean(other), pval) end @@ -391,14 +521,10 @@ def extract_zjit_stat(bench_name, key) end end - def extract_gc_times(bench_name, key) - @executable_names.map do |name| - bench_data_for(name, bench_name)[key] || [] - end - end - def detect_gc_data(bench_data) - bench_data.values.any? { |benchmarks| benchmarks.values.any? { |d| d.is_a?(Hash) && d.key?('gc_marking_time_bench') } } + bench_data.values.any? do |benchmarks| + benchmarks.values.any? { |d| d.is_a?(Hash) && GC_SERIES_KEYS.any? { |key| d.key?(key) } } + end end def detect_rss_samples(bench_data) diff --git a/test/results_table_builder_test.rb b/test/results_table_builder_test.rb index a70136f0..2201680a 100644 --- a/test/results_table_builder_test.rb +++ b/test/results_table_builder_test.rb @@ -636,11 +636,11 @@ assert_equal ['%s', '%s', '%s', '%.3f', '%s'], format assert_equal [ - 'bench', 'mark/iter ratio', 'sweep/iter ratio', 'mark/GC ratio', 'sweep/GC ratio', 'major/iter', 'minor/iter', 'minor GC %' + 'bench', 'mark/iter ratio', 'sweep/iter ratio', 'mark/GC ratio', 'sweep/GC ratio', 'GCs/iter', 'major/iter', 'minor/iter', 'minor GC %' ], gc_table[0] - assert_equal ['%s', '%s', '%s', '%s', '%s', '%s', '%s', '%s'], gc_format + assert_equal ['%s', '%s', '%s', '%s', '%s', '%s', '%s', '%s', '%s'], gc_format assert_equal [ - 'fib', '1.333', '1.000', '0.667', '0.500', ' 2.0 → 1.0', ' 8.0 → 4.0', ' 80% → 80%' + 'fib', '1.333', '1.000', '0.667', '0.500', '10.0 → 5.0', ' 2.0 → 1.0', ' 8.0 → 4.0', ' 80% → 80%' ], gc_table[1] end @@ -680,6 +680,40 @@ assert_nil gc_table assert_nil gc_format end + + it 'builds a non-Ractor absolute GC table without worker-sum suffixes or GC ms/worker' do + bench_data = { + 'ruby' => { + 'gcbench' => { + 'warmup' => [], + 'bench' => [0.1, 0.1], + 'rss' => 10, + 'gc_total_time_bench' => [4.0, 6.0], + 'gc_marking_time_bench' => [1.0, 3.0], + 'gc_sweeping_time_bench' => [0.5, 1.5], + 'gc_count_bench' => [10, 10], + 'gc_major_count_bench' => [2, 2], + 'gc_minor_count_bench' => [8, 8], + 'gc_global_count_bench' => [1, 1], + 'gc_controller_compact_count_bench' => [0, 0] + } + } + } + + builder = ResultsTableBuilder.new( + executable_names: ['ruby'], + bench_data: bench_data + ) + + _table, _format, gc_table, gc_format = builder.build + + assert_equal [ + 'bench', 'GC ms/iter', 'mark ms/iter', 'sweep ms/iter', 'GCs/iter', 'major/iter', 'minor/iter', + 'global GCs/iter*', 'controller compacts/iter*' + ], gc_table[0] + assert_equal ['%s'] * 9, gc_format + assert_equal ['gcbench', '5.000', '2.000', '1.000', '10.0', '2.0', '8.0', '1.0', '0.0'], gc_table[1] + end end describe 'RSS sampling (rss_samples)' do @@ -802,4 +836,280 @@ assert_equal ['%s', '%s', '%.1f'], format end end + + describe 'Ractor GC data' do + def gc_group(total:, major:, minor:, mark: nil, sweep: nil, count: nil, global: nil, compacts: nil, workers: nil) + count ||= major.zip(minor).map(&:sum) + group = { + 'gc_count_bench' => count, + 'gc_major_count_bench' => major, + 'gc_minor_count_bench' => minor, + 'gc_worker_samples' => workers || count.map { |c| [{ 'gc_count' => c, 'worker_index' => 0 }] } + } + group['gc_total_time_bench'] = total if total + group['gc_marking_time_bench'] = mark if mark + group['gc_sweeping_time_bench'] = sweep if sweep + group['gc_global_count_bench'] = global if global + group['gc_controller_compact_count_bench'] = compacts if compacts + group + end + + def ractor_gc_blob(groups) + { + 'warmup' => [], + 'bench' => groups.values.flat_map { |g| g[:bench] }, + 'rss' => 10 * 1024 * 1024, + 'gc_scope' => 'ractor-local-workload', + 'bench_by_ractors' => groups.transform_values { |g| g[:bench] }, + 'gc_by_ractors' => groups.transform_values { |g| g[:gc] } + } + end + + def build_ractor_gc(bench_data) + expanded = RactorBreakdown.expand(bench_data) + ResultsTableBuilder.new( + executable_names: bench_data.keys, + bench_data: expanded.bench_data, + row_layout: RactorRowLayout.new(groups: expanded.groups) + ).build + end + + it 'renders per-count comparison rows using only each count\'s own series' do + bench_data = { + 'base' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0, 1.0], gc: gc_group(total: [4.0, 4.0], major: [1, 1], minor: [3, 3], mark: [1.0, 1.0], sweep: [1.0, 1.0]) }, + '2' => { bench: [1.0, 1.0], gc: gc_group(total: [12.0, 12.0], major: [2, 2], minor: [6, 6], mark: [2.0, 2.0], sweep: [2.0, 2.0]) } + ) + }, + 'candidate' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0, 1.0], gc: gc_group(total: [2.0, 2.0], major: [1, 1], minor: [1, 1], mark: [1.0, 1.0], sweep: [1.0, 1.0]) }, + '2' => { bench: [1.0, 1.0], gc: gc_group(total: [3.0, 3.0], major: [1, 1], minor: [3, 3], mark: [1.0, 1.0], sweep: [1.0, 1.0]) } + ) + } + } + + table, _format, gc_table, gc_format = build_ractor_gc(bench_data) + + assert_equal ['bench', 'ractors', 'base (ms)', 'candidate (ms)', 'candidate 1st itr', 'base/candidate'], table[0] + assert_equal [ + 'bench', 'ractors', 'gc/iter ratio', 'gc/GC ratio', 'mark/iter ratio', 'sweep/iter ratio', + 'mark/GC ratio', 'sweep/GC ratio', 'GCs/iter (worker sum)', 'major/iter (worker sum)', 'minor/iter (worker sum)', 'minor GC %' + ], gc_table[0] + assert_equal ['%s'] * gc_table[0].size, gc_format + + rows = gc_table[1..].to_h { |row| [row[1], row] } + assert_equal %w[0 2], gc_table[1..].map { |row| row[1] } + + assert_equal ['object-new', 'object-new'], gc_table[1..].map(&:first) + + assert_equal ['object-new', '0', '2.000', '1.000', '1.000', '1.000', '0.500', '0.500', ' 4.0 → 2.0', ' 1.0 → 1.0', ' 3.0 → 1.0', ' 75% → 50%'], rows['0'] + assert_equal ['object-new', '2', '4.000', '2.000', '2.000', '2.000', '1.000', '1.000', ' 8.0 → 4.0', ' 2.0 → 1.0', ' 6.0 → 3.0', ' 75% → 75%'], rows['2'] + + gc_table.flatten.each { |cell| refute_includes cell.to_s, "\x00" } + end + + it 'renders N/A for a count missing optional GC data instead of reusing another count' do + bench_data = { + 'base' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0], gc: gc_group(total: [4.0], major: [1], minor: [3], mark: [1.0], sweep: [1.0]) }, + '2' => { bench: [1.0], gc: gc_group(total: [12.0], major: [2], minor: [6], mark: [2.0], sweep: [2.0]) } + ) + }, + 'candidate' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0], gc: gc_group(total: [2.0], major: [1], minor: [1], mark: [1.0], sweep: [1.0]) }, + '2' => { bench: [1.0], gc: gc_group(total: nil, major: [1], minor: [3], mark: nil, sweep: [1.0]) } + ) + } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + rows = gc_table[1..].to_h { |row| [row[1], row] } + assert_equal '2.000', rows['0'][2], 'count 0 keeps its own gc/iter ratio' + assert_equal '1.000', rows['0'][4] + assert_equal 'N/A', rows['2'][2], 'missing total-time series must not reuse count 0 data or zero' + assert_equal 'N/A', rows['2'][3] + assert_equal 'N/A', rows['2'][4], 'missing marking series must not be fabricated' + assert_equal '2.000', rows['2'][5], 'present sweeping series still renders' + end + + it 'builds an absolute GC table for one executable, keeping supported all-zero rows' do + bench_data = { + 'reference' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0, 1.0], gc: gc_group(total: [0.0, 0.0], major: [0, 0], minor: [0, 0], mark: [0.0, 0.0]) }, + '2' => { bench: [1.0, 1.0], gc: gc_group(total: [3.0, 5.0], major: [1, 1], minor: [3, 5], mark: [1.0, 3.0], sweep: [0.5, 1.5]) } + ) + } + } + + _table, _format, gc_table, gc_format = build_ractor_gc(bench_data) + + assert_equal ['bench', 'ractors', 'GC ms/iter (worker sum)', 'GC ms/worker', 'mark ms/iter (worker sum)', 'sweep ms/iter (worker sum)', 'GCs/iter (worker sum)', 'major/iter (worker sum)', 'minor/iter (worker sum)'], gc_table[0] + assert_equal ['%s'] * 9, gc_format + assert_equal ['object-new', '0', '0.000', '0.000', '0.000', 'N/A', '0.0', '0.0', '0.0'], gc_table[1] + assert_equal ['object-new', '2', '4.000', '4.000', '2.000', '1.000', '5.0', '1.0', '4.0'], gc_table[2] + end + + it 'detects GC data from any recognized series, not just marking time' do + bench_data = { + 'reference' => { + 'object-new' => ractor_gc_blob( + '0' => { bench: [1.0], gc: gc_group(total: nil, major: [2], minor: [4]) } + ) + } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + refute_nil gc_table, 'a blob with only count series is still GC data' + assert_equal ['object-new', '0', 'N/A', 'N/A', 'N/A', 'N/A', '6.0', '2.0', '4.0'], gc_table[1] + end + + it 'derives GC ms/worker by dividing each iteration by its sampled worker count' do + per_worker_group = gc_group( + total: [12.0, 6.0], major: [2, 2], minor: [6, 6], + workers: [ + [{ 'gc_count' => 4, 'worker_index' => 0 }, { 'gc_count' => 4, 'worker_index' => 1 }], + [{ 'gc_count' => 8, 'worker_index' => 0 }] + ] + ) + mismatched_group = gc_group( + total: [12.0, 6.0], major: [2, 2], minor: [6, 6], + workers: [[{ 'gc_count' => 8, 'worker_index' => 0 }]] + ) + empty_workers_group = gc_group(total: [12.0, 6.0], major: [2, 2], minor: [6, 6], workers: [[], []]) + nil_total_group = gc_group(total: [nil, 6.0], major: [2, 2], minor: [6, 6]) + + bench_data = { + 'reference' => { + 'obj-2w' => ractor_gc_blob('2' => { bench: [1.0, 1.0], gc: per_worker_group }), + 'obj-mismatch' => ractor_gc_blob('2' => { bench: [1.0, 1.0], gc: mismatched_group }), + 'obj-empty' => ractor_gc_blob('2' => { bench: [1.0, 1.0], gc: empty_workers_group }), + 'obj-nil-total' => ractor_gc_blob('2' => { bench: [1.0, 1.0], gc: nil_total_group }) + } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + header = gc_table[0] + ms_per_worker_idx = header.index('GC ms/worker') + refute_nil ms_per_worker_idx + rows = gc_table[1..].to_h { |row| [row[0], row] } + assert_equal '6.000', rows['obj-2w'][ms_per_worker_idx] + assert_equal '9.000', rows['obj-2w'][header.index('GC ms/iter (worker sum)')] + assert_equal 'N/A', rows['obj-mismatch'][ms_per_worker_idx], 'mismatched series lengths' + assert_equal 'N/A', rows['obj-empty'][ms_per_worker_idx], 'empty worker list' + assert_equal 'N/A', rows['obj-nil-total'][ms_per_worker_idx], 'unavailable total' + end + + it 'uses the direct count for GCs/iter and per-GC ratios when it differs from major plus minor' do + group = ->(total, count, major, minor) do + gc_group( + total: [total, total], count: [count, count], + major: [major, major], minor: [minor, minor], + mark: [1.0, 1.0], sweep: [1.0, 1.0] + ) + end + bench_data = { + 'base' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: group.call(30.0, 10, 1, 1) }) }, + 'candidate' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: group.call(12.0, 6, 1, 2) }) } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + header = gc_table[0] + row = gc_table[1] + assert_equal '2.500', row[header.index('gc/iter ratio')] + assert_equal '1.500', row[header.index('gc/GC ratio')] + assert_equal '10.0 → 6.0', row[header.index('GCs/iter (worker sum)')] + end + + it 'falls back to major plus minor only when the direct-count key is absent' do + without_direct = gc_group(total: nil, major: [2, 2], minor: [8, 8]).tap { |group| group.delete('gc_count_bench') } + null_direct = gc_group(total: nil, major: [2, 2], minor: [8, 8], count: [nil, nil]) + + bench_data = { + 'reference' => { + 'legacy' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: without_direct }), + 'null-count' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: null_direct }) + } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + header = gc_table[0] + gcs_idx = header.index('GCs/iter (worker sum)') + rows = gc_table[1..].to_h { |row| [row[0], row] } + assert_equal '10.0', rows['legacy'][gcs_idx], 'absent direct-count key sums major and minor' + assert_equal 'N/A', rows['null-count'][gcs_idx], 'present nulls stay unavailable instead of falling back' + assert_equal '2.0', rows['null-count'][header.index('major/iter (worker sum)')], 'neighboring values survive the null series' + end + + it 'renders global GC and controller compaction columns in a Ractor comparison' do + group = ->(global, compacts) do + gc_group( + total: [4.0, 4.0], major: [1, 1], minor: [3, 3], + mark: [1.0, 1.0], sweep: [1.0, 1.0], + global: [global, global], compacts: [compacts, compacts] + ) + end + bench_data = { + 'base' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: group.call(4, 1) }) }, + 'candidate' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: group.call(2, 3) }) } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + header = gc_table[0] + assert_equal [ + 'bench', 'ractors', 'gc/iter ratio', 'gc/GC ratio', 'mark/iter ratio', 'sweep/iter ratio', + 'mark/GC ratio', 'sweep/GC ratio', 'global/iter ratio*', + 'GCs/iter (worker sum)', 'major/iter (worker sum)', 'minor/iter (worker sum)', + 'controller compacts/iter*', 'minor GC %' + ], header + row = gc_table[1] + assert_equal '2.000', row[header.index('global/iter ratio*')], 'base mean divided by comparison mean' + assert_equal ' 1.0 → 3.0', row[header.index('controller compacts/iter*')] + end + + it 'renders N/A for a partially-null series without losing the row or its neighbors' do + partial_group = gc_group(total: [4.0, 4.0], major: [1, 1], minor: [3, 3], mark: [nil, 1.0], sweep: [1.0, 1.0]) + full_group = gc_group(total: [4.0, 4.0], major: [1, 1], minor: [3, 3], mark: [2.0, 2.0], sweep: [1.0, 1.0]) + bench_data = { + 'base' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: partial_group }) }, + 'candidate' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: full_group }) } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + refute_nil gc_table, 'the nil entry must not hide a row whose other metrics have activity' + header = gc_table[0] + row = gc_table[1] + assert_equal 'N/A', row[header.index('mark/iter ratio')], 'a null entry makes the whole series unavailable' + assert_equal 'N/A', row[header.index('mark/GC ratio')] + assert_equal '1.000', row[header.index('sweep/iter ratio')], 'neighboring series keep their values' + assert_equal ' 4.0 → 4.0', row[header.index('GCs/iter (worker sum)')] + end + + it 'shows global and compaction columns in a single-executable Ractor table' do + group = gc_group(total: [4.0, 4.0], major: [1, 1], minor: [3, 3], global: [2, 4], compacts: [1, 1]) + bench_data = { + 'reference' => { 'object-new' => ractor_gc_blob('0' => { bench: [1.0, 1.0], gc: group }) } + } + + _table, _format, gc_table, _gc_format = build_ractor_gc(bench_data) + + assert_equal [ + 'bench', 'ractors', 'GC ms/iter (worker sum)', 'GC ms/worker', 'mark ms/iter (worker sum)', + 'sweep ms/iter (worker sum)', 'GCs/iter (worker sum)', 'major/iter (worker sum)', + 'minor/iter (worker sum)', 'global GCs/iter*', 'controller compacts/iter*' + ], gc_table[0] + assert_equal ['object-new', '0', '4.000', '4.000', 'N/A', 'N/A', '4.0', '1.0', '3.0', '3.0', '1.0'], gc_table[1] + end + end end From fbb557be8e731a912957d63a08cd7af00e8b294c Mon Sep 17 00:00:00 2001 From: Matt Valentine-House Date: Tue, 29 Sep 2026 11:26:27 +0100 Subject: [PATCH 6/6] Document Ractor GC metrics and add --ractor-gc Passing --ractor-gc to run_benchmarks.rb now sets RUBY_BENCH_RACTOR_GC=1 --- README.md | 40 +++++++++++++ lib/argument_parser.rb | 4 ++ lib/benchmark_runner.rb | 32 ++++++++++- lib/benchmark_runner/cli.rb | 4 +- lib/benchmark_suite.rb | 2 +- test/argument_parser_test.rb | 11 +++- test/benchmark_runner_cli_test.rb | 80 +++++++++++++++++++++++++- test/benchmark_runner_test.rb | 94 +++++++++++++++++++++++++++++++ 8 files changed, 260 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index a36d0955..ff3d11cf 100644 --- a/README.md +++ b/README.md @@ -300,6 +300,46 @@ For reference, the JSON output also keeps `rss`, a single snapshot taken after a full GC at the end of the run (the retained set, a lower bound), and `maxrss`, the process's lifetime peak from `getrusage`. +## Measuring Ractor GC activity + +The `--ractor-gc` option of `run_benchmarks.rb` collects Ractor-local GC +metrics for benchmarks that use the Ractor harness (`--category ractor`). +The target must use Ruby 4.1 or newer; older targets fail before warmup. + +```sh +./run_benchmarks.rb --category ractor --chruby=base::ruby-base --ractor-gc +``` + +Each measured iteration samples `GC.stat` and GC total time in every worker +Ractor's own object space. The JSON output records the scope as +`gc_scope: "ractor-local-workload"`, `gc_stat_scope: "ractor-local"`, and +`gc_measure_total_time_scope: "ractor-local"`, plus the target's `gc_config`. + +The summary table adds these columns: + +* `(worker sum)` columns add the Ractor-local counters of the sampled + workers of each iteration. Single-executable reports also show + `GC ms/worker`, which divides each iteration's worker-sum GC time by its + sampled worker count, then averages. +* `global GCs/iter*` shows `GC.stat(:global_gc_count)` deltas observed by the + measuring main Ractor. These count stop-the-world global GC cycles, not + the process-wide total of all Ractors' local and global collections. + Stop-the-world cycles only exist once a second Ractor has been alive, so + count-0 rows read 0.0 unless the workload itself creates Ractors. In + comparison reports the same metric appears as `global/iter ratio*`, the + base executable's mean divided by the comparison executable's mean. +* `controller compacts/iter*` shows the main Ractor's + `GC.stat(:compact_count)` delta. Every global compacting cycle increments + it, but it is not a sum of worker counters. + +The starred columns are controller-observed deltas, not additive with the +worker-sum columns: a global or compacting cycle triggered by a sampled +worker is already included in that worker's counts as a major GC. They can +also reflect cycles from Ractors the harness does not sample. + +Worker records in the JSON output never contain the controller-observed +counters, and controller counters are never summed across workers. + ## Rendering a graph `--graph` option of `run_benchmarks.rb` allows you to render benchmark results as a graph. diff --git a/lib/argument_parser.rb b/lib/argument_parser.rb index 9a83f5e3..6a478945 100644 --- a/lib/argument_parser.rb +++ b/lib/argument_parser.rb @@ -111,6 +111,10 @@ def parse(argv) ENV["WARMUP_ITRS"] = n end + opts.on("--ractor-gc", "collect Ractor-local GC metrics when using the Ractor harness (the target must use Ruby 4.1 or newer)") do + ENV["RUBY_BENCH_RACTOR_GC"] = "1" + end + opts.on("--bench=N", "the number of benchmark iterations for the default harness (default: 10). Also defaults MIN_BENCH_TIME to 0.") do |n| ENV["MIN_BENCH_ITRS"] = n ENV["MIN_BENCH_TIME"] ||= "0" diff --git a/lib/benchmark_runner.rb b/lib/benchmark_runner.rb index 87d27ff9..dc8ef892 100644 --- a/lib/benchmark_runner.rb +++ b/lib/benchmark_runner.rb @@ -82,14 +82,40 @@ def build_output_text(ruby_descriptions, table, format, bench_failures, include_ end if has_gc_summary output_str << "- GC summary compares #{base_name} → comparison. Ratio columns are #{base_name}/comparison; above 1 means the comparison spent less GC time.\n" - output_str << "- mark/iter ratio and sweep/iter ratio compare total GC phase time per benchmark iteration, so they include both per-GC cost and GC frequency changes.\n" - output_str << "- mark/GC ratio and sweep/GC ratio compare average phase time per GC, isolating whether each GC became cheaper or more expensive.\n" - output_str << "- major/iter, minor/iter, and minor GC % show #{base_name} → comparison values, not ratios. Rows with no GC activity are omitted.\n" + output_str << "- gc/iter, mark/iter, and sweep/iter ratio compare total GC (or phase) time per benchmark iteration, so they include both per-GC cost and GC frequency changes.\n" + output_str << "- gc/GC, mark/GC, and sweep/GC ratio divide average GC (or phase) time by the same run's GCs/iter count; they are not complete per-cycle attribution of process-wide GC.\n" + output_str << "- GCs/iter, major/iter, minor/iter, controller compacts/iter*, and minor GC % show #{base_name} → comparison values, not ratios. Rows with no GC activity are omitted.\n" end if include_pvalue output_str << "- ***: p < 0.001, **: p < 0.01, *: p < 0.05 (Welch's t-test)\n" end end + gc_headers = sections.filter_map { |section| section[:gc_table]&.first }.flatten + global_column = gc_headers.any? { |h| h == 'global GCs/iter*' || h == 'global/iter ratio*' } + compact_column = gc_headers.include?('controller compacts/iter*') + if global_column || compact_column + output_str << "GC metric notes:\n" + if global_column + output_str << "- global GCs/iter*: GC.stat(:global_gc_count) deltas per iteration. They count stop-the-world global GC cycles observed by the measuring main Ractor, not the process-wide total of all Ractors' local and global collections. Stop-the-world cycles only exist once a second Ractor has been alive, so count-0 rows read 0.0 unless the workload itself creates Ractors.\n" + end + if gc_headers.include?('global/iter ratio*') + output_str << "- global/iter ratio*: the #{base_name} mean divided by the comparison executable's mean.\n" + end + if compact_column + output_str << "- controller compacts/iter*: the main Ractor's GC.stat(:compact_count) delta per iteration. Every global compacting cycle increments it, but it is not a sum of worker counters.#{other_names.empty? ? '' : " Comparison tables show #{base_name} → comparison values."}\n" + end + output_str << "- * controller-observed deltas can overlap the (worker sum) columns: a global or compacting cycle triggered by a sampled worker is already included in that worker's counts as a major GC. They can also reflect cycles from Ractors the harness does not sample. Do not add the starred columns to the worker sums.\n" + end + + if sections.any? { |section| section[:gc_scope] == 'ractor-local-workload' && section[:gc_table] } + output_str << "Ractor GC scope note:\n" + scope_columns = +"- (worker sum) columns add Ractor-local counters across the sampled workers of each iteration; the main Ractor performs the count-0 workload." + scope_columns << " GC ms/worker divides each iteration's worker-sum GC time by its sampled worker count, then averages." if other_names.empty? + output_str << "#{scope_columns} Controller snapshots and per-worker heap detail are in the JSON output, not this table.\n" + output_str << "- Ruby's Ractor-retirement GC (after a worker's stack is torn down) and Ractors created by the workload itself are not sampled.\n" + output_str << "- GC time is CPU-time accounting, not elapsed pause time; summed across Ractors it can exceed wall time.\n" + output_str << "- Per-GC ratios divide by recorded GC counts, not complete process-wide GC cycles. Phase times are integer milliseconds; total GC time is kept at nanosecond resolution in the raw worker samples.\n" + end output_str end diff --git a/lib/benchmark_runner/cli.rb b/lib/benchmark_runner/cli.rb index f8038c7e..ed59d2d1 100644 --- a/lib/benchmark_runner/cli.rb +++ b/lib/benchmark_runner/cli.rb @@ -195,7 +195,7 @@ def build_output_section(executable_names, bench_data, bench_failures, harness, ) table, format, gc_table, gc_format = builder.build - { + section = { title: harness, table: table, format: format, @@ -204,6 +204,8 @@ def build_output_section(executable_names, bench_data, bench_failures, harness, gc_table: gc_table, gc_format: gc_format, } + section[:gc_scope] = 'ractor-local-workload' if ResultsTableBuilder.ractor_gc_data?(section_data) + section end def sorted_benchmark_names(executable_names, bench_data) diff --git a/lib/benchmark_suite.rb b/lib/benchmark_suite.rb index ce37d2f1..4c22a12a 100644 --- a/lib/benchmark_suite.rb +++ b/lib/benchmark_suite.rb @@ -212,7 +212,7 @@ def compute_benchmark_env(ruby) # Pass benchmark configuration env vars to subprocess. # These may be set after bundler loads, so they'd be lost with with_unbundled_env. - ["WARMUP_ITRS", "MIN_BENCH_ITRS", "MIN_BENCH_TIME", "YJIT_BENCH_STATS", "ZJIT_BENCH_STATS"].each do |var| + ["WARMUP_ITRS", "MIN_BENCH_ITRS", "MIN_BENCH_TIME", "YJIT_BENCH_STATS", "ZJIT_BENCH_STATS", "RUBY_BENCH_RACTOR_GC"].each do |var| env[var] = ENV[var] if ENV.key?(var) end diff --git a/test/argument_parser_test.rb b/test/argument_parser_test.rb index 9ab2db9c..d11f1b8f 100644 --- a/test/argument_parser_test.rb +++ b/test/argument_parser_test.rb @@ -7,7 +7,7 @@ before do @original_env = {} ['WARMUP_ITRS', 'MIN_BENCH_ITRS', 'MIN_BENCH_TIME', 'YJIT_BENCH_STATS', - 'ZJIT_BENCH_STATS', 'RUBIES_DIR', 'HOME'].each do |key| + 'ZJIT_BENCH_STATS', 'RUBY_BENCH_RACTOR_GC', 'RUBIES_DIR', 'HOME'].each do |key| @original_env[key] = ENV[key] end end @@ -364,6 +364,15 @@ def setup_mock_ruby(path) end end + describe '--ractor-gc option' do + it 'sets RUBY_BENCH_RACTOR_GC environment variable' do + parser = ArgumentParser.new + parser.parse(['--ractor-gc']) + + assert_equal '1', ENV['RUBY_BENCH_RACTOR_GC'] + end + end + describe '--bench option' do it 'sets MIN_BENCH_ITRS and MIN_BENCH_TIME environment variables' do parser = ArgumentParser.new diff --git a/test/benchmark_runner_cli_test.rb b/test/benchmark_runner_cli_test.rb index 7b344459..7baf82f3 100644 --- a/test/benchmark_runner_cli_test.rb +++ b/test/benchmark_runner_cli_test.rb @@ -9,7 +9,7 @@ describe BenchmarkRunner::CLI do before do @original_env = {} - ['WARMUP_ITRS', 'MIN_BENCH_ITRS', 'MIN_BENCH_TIME', 'BENCHMARK_QUIET'].each do |key| + ['WARMUP_ITRS', 'MIN_BENCH_ITRS', 'MIN_BENCH_TIME', 'BENCHMARK_QUIET', 'RUBY_BENCH_RACTOR_GC'].each do |key| @original_env[key] = ENV[key] end @@ -117,6 +117,84 @@ def create_args(overrides = {}) assert_equal ['bench', 'ruby (ms)'], sections[1][:table][0] assert_equal ['fib', '100.0 ± 0.0%'], sections[1][:table][1] end + + it 'renders a Ractor GC section whose GC rows identify the same counts as the timing rows' do + args = create_args + cli = BenchmarkRunner::CLI.new(args) + gc_group = ->(total, major, minor) do + { + 'gc_count_bench' => [major + minor], + 'gc_major_count_bench' => [major], + 'gc_minor_count_bench' => [minor], + 'gc_marking_time_bench' => [1.0], + 'gc_sweeping_time_bench' => [1.0], + 'gc_total_time_bench' => [total], + 'gc_worker_samples' => [[{ 'gc_count' => major + minor, 'worker_index' => 0 }]] + } + end + bench_data = { + 'ruby' => { + 'fib' => { + 'warmup' => [], + 'bench' => [0.1], + 'rss' => 10 * 1024 * 1024 + }, + 'object-new' => { + 'warmup' => [], + 'bench' => [1.0, 2.0], + 'rss' => 10 * 1024 * 1024, + 'gc_scope' => 'ractor-local-workload', + 'bench_by_ractors' => { '0' => [1.0], '2' => [2.0] }, + 'gc_by_ractors' => { '0' => gc_group.call(4.0, 1, 3), '2' => gc_group.call(12.0, 2, 6) } + } + } + } + bench_harnesses = { 'fib' => 'harness', 'object-new' => 'harness-ractor' } + + sections = cli.send(:build_output_sections, ['ruby'], bench_data, bench_harnesses, {}) + ractor = sections.find { |section| section[:title] == 'harness-ractor' } + normal = sections.find { |section| section[:title] == 'harness' } + + assert_equal 'ractor-local-workload', ractor[:gc_scope] + refute normal.key?(:gc_scope) + + assert_equal ['object-new', '0'], ractor[:table][1][0..1] + assert_equal ['', '2'], ractor[:table][2][0..1] + assert_equal ['object-new', '0'], ractor[:gc_table][1][0..1] + assert_equal ['object-new', '2'], ractor[:gc_table][2][0..1] + assert_equal ['bench', 'ractors', 'GC ms/iter (worker sum)', 'GC ms/worker', 'mark ms/iter (worker sum)', 'sweep ms/iter (worker sum)', 'GCs/iter (worker sum)', 'major/iter (worker sum)', 'minor/iter (worker sum)'], ractor[:gc_table][0] + ractor[:gc_table].flatten.each { |cell| refute_includes cell.to_s, "\x00" } + + assert_nil normal[:gc_table] + + output = BenchmarkRunner.build_output_text({ 'ruby' => 'ruby 4.1.0dev' }, nil, nil, {}, sections: sections) + assert_match(/Ractor GC scope note:/, output) + assert_match(/CPU-time accounting, not elapsed pause time/, output) + refute_match(/\x00/, output) + end + + it 'tags the section from the raw blob scope even when no count carries GC data' do + args = create_args + cli = BenchmarkRunner::CLI.new(args) + bench_data = { + 'ruby' => { + 'object-new' => { + 'warmup' => [], + 'bench' => [1.0, 2.0], + 'rss' => 10 * 1024 * 1024, + 'gc_scope' => 'ractor-local-workload', + 'bench_by_ractors' => { '0' => [1.0], '2' => [2.0] } + } + } + } + + sections = cli.send(:build_output_sections, ['ruby'], bench_data, { 'object-new' => 'harness-ractor' }, {}) + + assert sections.one?, 'single-harness fixture must produce exactly one section' + section = sections.first + assert_equal 'ractor-local-workload', section[:gc_scope], + 'the raw blob scope must tag the section; expansion drops gc_scope from counts without GC data' + end end describe '#run integration test' do diff --git a/test/benchmark_runner_test.rb b/test/benchmark_runner_test.rb index 04220992..32c34c46 100644 --- a/test/benchmark_runner_test.rb +++ b/test/benchmark_runner_test.rb @@ -437,6 +437,100 @@ refute_includes result, 'mark ruby-base/ruby-exp:' end + it 'explains global GC and controller compaction columns when a GC table shows them' do + ruby_descriptions = { + 'ruby-base' => 'ruby 4.1.0dev', + 'ruby-exp' => 'ruby 4.1.0dev experiment' + } + table = [ + ['bench', 'ruby-base (ms)', 'ruby-exp (ms)', 'ruby-base/ruby-exp'], + ['fib', '100.0', '50.0', '2.000'] + ] + format = ['%s', '%s', '%s', '%s'] + gc_table = [ + ['bench', 'gc/iter ratio', 'global/iter ratio*', 'GCs/iter', 'controller compacts/iter*'], + ['fib', '2.000', '1.500', '10.0 → 5.0', ' 1.0 → 2.0'] + ] + gc_format = ['%s', '%s', '%s', '%s', '%s'] + + result = BenchmarkRunner.build_output_text( + ruby_descriptions, table, format, {}, include_gc: true, gc_table: gc_table, gc_format: gc_format + ) + + assert_includes result, "GC metric notes:\n" + assert_includes result, 'stop-the-world global GC cycles observed by the measuring main Ractor' + assert_includes result, "- global/iter ratio*: the ruby-base mean divided by the comparison executable's mean." + assert_includes result, "- * controller-observed deltas can overlap the (worker sum) columns" + end + + it 'explains the columns for a single-executable GC table without a legend' do + ruby_descriptions = { 'ruby' => 'ruby 4.1.0dev' } + table = [['bench', 'ruby (ms)'], ['fib', '100.0']] + format = ['%s', '%s'] + gc_table = [ + ['bench', 'GC ms/iter', 'GCs/iter', 'global GCs/iter*', 'controller compacts/iter*'], + ['fib', '4.000', '10.0', '1.0', '0.0'] + ] + gc_format = ['%s', '%s', '%s', '%s', '%s'] + + result = BenchmarkRunner.build_output_text( + ruby_descriptions, table, format, {}, include_gc: true, gc_table: gc_table, gc_format: gc_format + ) + + refute_includes result, 'Legend:' + assert_includes result, "GC metric notes:\n" + assert_includes result, '- global GCs/iter*: GC.stat(:global_gc_count) deltas per iteration.' + refute_includes result, 'global/iter ratio*:' + assert_includes result, "- controller compacts/iter*: the main Ractor's GC.stat(:compact_count) delta per iteration." + assert_includes result, 'Do not add the starred columns to the worker sums.' + end + + it 'omits the GC metric notes block when no GC table shows those columns' do + ruby_descriptions = { + 'ruby-base' => 'ruby 3.3.0', + 'ruby-exp' => 'ruby 3.3.0 experiment' + } + table = [ + ['bench', 'ruby-base (ms)', 'ruby-exp (ms)', 'ruby-base/ruby-exp'], + ['fib', '100.0', '50.0', '2.000'] + ] + format = ['%s', '%s', '%s', '%s'] + gc_table = [ + ['bench', 'mark/iter ratio', 'GCs/iter', 'major/iter'], + ['fib', '2.000', '10.0 → 5.0', ' 2.0 → 1.0'] + ] + gc_format = ['%s', '%s', '%s', '%s'] + + result = BenchmarkRunner.build_output_text( + ruby_descriptions, table, format, {}, include_gc: true, gc_table: gc_table, gc_format: gc_format + ) + + refute_includes result, 'GC metric notes:' + assert_includes result, "the same run's GCs/iter count" + assert_includes result, 'show ruby-base → comparison values, not ratios' + end + + it 'omits the Ractor scope note when the section rendered no GC table' do + ruby_descriptions = { 'ruby' => 'ruby 4.1.0dev' } + sections = [ + { + title: 'harness-ractor', + table: [['bench', 'ractors', 'ruby (ms)'], ['object-new', '0', '200.0']], + format: ['%s', '%s', '%.1f'], + failures: {}, + include_gc: true, + gc_table: nil, + gc_scope: 'ractor-local-workload', + } + ] + + result = BenchmarkRunner.build_output_text( + ruby_descriptions, sections.first[:table], sections.first[:format], {}, sections: sections + ) + + refute_includes result, 'Ractor GC scope note:' + end + it 'includes RSS ratio legend when include_rss is true' do ruby_descriptions = { 'ruby-base' => 'ruby 3.3.0',