Skip to content
26 changes: 26 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -165,6 +165,32 @@ intended to be used with any harness except `harness-ractor`.
Note: The `harness-ractor` harness is automatically selected when using these
categories, so there's no need to specify `--harness` manually.

### Ractor counts

The Ractor harness measures each benchmark at 0 (the main Ractor only), 1, 2,
4, 6, and 8 Ractors. Set `RUBY_BENCH_RACTORS` to a comma-separated list to
change the counts, for example `RUBY_BENCH_RACTORS=0,2,8`.

`run_benchmarks.rb` starts a fresh process for each count. Heap pages, the GC
page pool, and JIT state therefore cannot carry over from one count to the
next. Each process runs `WARMUP_ITRS` warmup iterations at its own count
before the measured iterations. As in the default harness, the harness prints
each warmup iteration and records its time.

The JSON output keeps one blob per benchmark. `warmup_by_ractors`,
`bench_by_ractors`, and `gc_by_ractors` hold the measurements for each count.
`warmup_by_ractors` holds wall times only. The `--ractor-gc` mode does not
keep GC samples for warmup iterations. `results_by_ractors` holds the
process-level data of each count: `rss`, `maxrss`, YJIT or ZJIT stats, and
`command_line`. The blob has no top-level `rss`, `maxrss`, or JIT stats,
because no single process ran all counts.

The text summary, the CSV output, and `misc/zjit_diff.rb` show one row for
each count, with the RSS and JIT stats of that count's process.

When you run a benchmark directly with `-Iharness-ractor`, the harness runs
all counts in one process, one count after another.

## Ruby options

By default, ruby-bench benchmarks the Ruby used for `run_benchmarks.rb`.
Expand Down
112 changes: 51 additions & 61 deletions harness-ractor/harness.rb
Original file line number Diff line number Diff line change
Expand Up @@ -17,18 +17,8 @@
{}.freeze
end

default_ractors = [
0, # without ractor
1, 2, 4, 6, 8#, 12, 16, 32
]
if rs = ENV["RUBY_BENCH_RACTORS"]
rs = rs.split(",").map(&:to_i) # If you want to include 0, you have to specify
rs = rs.sort.uniq
if rs.any?
ractors = rs
end
end
RACTORS = (ractors || default_ractors).freeze
require_relative '../lib/ractor_counts'
RACTORS = RactorCounts.from_env

unless Ractor.method_defined?(:join)
class Ractor
Expand All @@ -51,13 +41,11 @@ def run_benchmark(num_itrs_hint, ractor_args: [], &block)
check_ractor_gc_support
gc_config = GC.config.transform_keys(&:to_s) if GC.respond_to?(:config)
GCStats.with_measure_total_time do
run_warmup(warmup_itrs, ractor_args, &block)
run_benchmark_gc(bench_itrs, Ractor.make_shareable(block), ractor_args, gc_config: gc_config)
run_benchmark_gc(warmup_itrs, bench_itrs, Ractor.make_shareable(block), ractor_args, gc_config: gc_config)
end
else
puts "r: itr: time"
run_warmup(warmup_itrs, ractor_args, &block)
run_benchmark_timing(bench_itrs, ractor_args, &block)
run_benchmark_timing(warmup_itrs, bench_itrs, ractor_args, &block)
end
end

Expand All @@ -73,41 +61,36 @@ def check_ractor_gc_support
end
end

def run_warmup(warmup_itrs, ractor_args, &block)
warmup_itrs.times do
args = ractor_args.empty? ? [] : ractor_deep_dup(ractor_args)
block.call(*([0] + args))
def run_benchmark_timing(warmup_itrs, bench_itrs, ractor_args, &block)
warmups = {}
stats = {}

RACTORS.each do |rs|
times = Array.new(warmup_itrs + bench_itrs) do |i|
time = run_timing_iteration(rs, ractor_args, &block)
puts "%-3s %4s %6s" % ["#{rs}", "##{i + 1}:", "#{(1000 * time).to_i}ms"]
time
end
warmups[rs], stats[rs] = times[0...warmup_itrs], times[warmup_itrs..]
end
return_results(warmups.values.flatten, stats.values.flatten, warmup_by_ractors: warmups, bench_by_ractors: stats)
end

def run_benchmark_timing(bench_itrs, ractor_args, &block)
stats = Hash.new { |h,k| h[k] = [] }

RACTORS.each do |rs|
num_itrs = 0
while num_itrs < bench_itrs
before = Process.clock_gettime(Process::CLOCK_MONOTONIC)
if rs.zero?
block.call *([rs] + ractor_deep_dup(ractor_args))
else
rs_list = []
rs.times do
rs_list << Ractor.new(*([rs] + ractor_args), &block) # ractor_args are copied
end
while rs_list.any?
r, _obj = Ractor.select(*rs_list)
rs_list.delete(r)
end
end
num_itrs += 1
time = Process.clock_gettime(Process::CLOCK_MONOTONIC) - before
time_ms = (1000 * time).to_i
itr_str = "%-3s %4s %6s" % ["#{rs}", "##{num_itrs}:", "#{time_ms}ms"]
stats[rs] << time
puts itr_str
def run_timing_iteration(rs, ractor_args, &block)
before = Process.clock_gettime(Process::CLOCK_MONOTONIC)
if rs.zero?
block.call *([rs] + ractor_deep_dup(ractor_args))
else
rs_list = []
rs.times do
rs_list << Ractor.new(*([rs] + ractor_args), &block) # ractor_args are copied
end
while rs_list.any?
r, _obj = Ractor.select(*rs_list)
rs_list.delete(r)
end
end
return_results([], stats.values.flatten, bench_by_ractors: stats)
Process.clock_gettime(Process::CLOCK_MONOTONIC) - before
end

RACTOR_GC_SERIES = {
Expand All @@ -120,8 +103,9 @@ def run_benchmark_timing(bench_itrs, ractor_args, &block)
"gc_total_time_bench" => "gc_total_time_ns",
}.freeze

def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:)
stats = Hash.new { |h,k| h[k] = [] }
def run_benchmark_gc(warmup_itrs, bench_itrs, block, ractor_args, gc_config:)
warmups = {}
stats = {}
gc_by_ractors = {}

header = +"r: itr: time gc_total marking sweeping gc_count major minor global"
Expand All @@ -130,32 +114,37 @@ def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:)
puts "(* controller-observed compacting cycles; may overlap global counts and is not additive.)" if CONTROLLER_GC_SERIES.any?

RACTORS.each do |rs|
warmups[rs] = []
stats[rs] = []
group = { "gc_worker_samples" => [] }
group["gc_controller_samples"] = [] if rs > 0
series = Hash.new { |h,k| h[k] = [] }

num_itrs = 0
while num_itrs < bench_itrs
num_itrs += 1
(warmup_itrs + bench_itrs).times do |i|
elapsed, worker_samples, controller_sample, controller_deltas = run_ractor_gc_iteration(rs, ractor_args, &block)
stats[rs] << elapsed
group["gc_worker_samples"] << worker_samples
group["gc_controller_samples"] << controller_sample if controller_sample

agg = GCStats.aggregate(worker_samples)
total_ms = agg["gc_total_time_ns"]&.fdiv(1_000_000)
RACTOR_GC_SERIES.each do |series_name, field|
series[series_name] << (field == "gc_total_time_ns" ? total_ms : agg[field])
end
CONTROLLER_GC_SERIES.each_key { |series_name| series[series_name] << controller_deltas[series_name] }

itr_str = "%-3s %4s %6s" % [rs, "##{num_itrs}:", "#{(1000 * elapsed).to_i}ms"]
itr_str = "%-3s %4s %6s" % [rs, "##{i + 1}:", "#{(1000 * elapsed).to_i}ms"]
itr_str << " %8s" % (total_ms ? "%.1fms" % total_ms : "N/A")
itr_str << " %8s" % (agg["gc_marking_time"] ? "#{agg["gc_marking_time"]}ms" : "N/A")
itr_str << " %8s" % (agg["gc_sweeping_time"] ? "#{agg["gc_sweeping_time"]}ms" : "N/A")
itr_str << " %9s %9s %9s %9s" % [agg["gc_count"], agg["gc_major_count"], agg["gc_minor_count"], agg["gc_global_count"]].map { |v| v.nil? ? "N/A" : v.to_s }
CONTROLLER_GC_SERIES.each_key { |series_name| itr_str << " %9s" % (controller_deltas[series_name] || "N/A") }
puts itr_str

if i < warmup_itrs
warmups[rs] << elapsed
next
end

stats[rs] << elapsed
group["gc_worker_samples"] << worker_samples
group["gc_controller_samples"] << controller_sample if controller_sample
RACTOR_GC_SERIES.each do |series_name, field|
series[series_name] << (field == "gc_total_time_ns" ? total_ms : agg[field])
end
CONTROLLER_GC_SERIES.each_key { |series_name| series[series_name] << controller_deltas[series_name] }
end

series.each do |name, values|
Expand All @@ -165,14 +154,15 @@ def run_benchmark_gc(bench_itrs, block, ractor_args, gc_config:)
end

extra = {
warmup_by_ractors: warmups,
bench_by_ractors: stats,
gc_scope: "ractor-local-workload",
gc_stat_scope: "ractor-local",
gc_measure_total_time_scope: "ractor-local",
gc_by_ractors: gc_by_ractors,
}
extra[:gc_config] = gc_config if gc_config
return_results([], stats.values.flatten, **extra)
return_results(warmups.values.flatten, stats.values.flatten, **extra)
end

def controller_gc_snapshot
Expand Down
2 changes: 1 addition & 1 deletion lib/benchmark_runner.rb
Original file line number Diff line number Diff line change
Expand Up @@ -74,7 +74,7 @@ def build_output_text(ruby_descriptions, table, format, bench_failures, include_
unless other_names.empty?
output_str << "Legend:\n"
other_names.each do |name|
output_str << "- #{name} 1st itr: ratio of #{base_name}/#{name} time for the first benchmarking iteration.\n"
output_str << "- #{name} 1st itr: ratio of #{base_name}/#{name} time for the first iteration.\n"
output_str << "- #{base_name}/#{name}: ratio of #{base_name}/#{name} time. Higher is better for #{name}. Above 1 represents a speedup.\n"
if include_rss
output_str << "- RSS #{base_name}/#{name}: ratio of #{base_name}/#{name} RSS. Higher is better for #{name}. Above 1 means lower memory usage.\n"
Expand Down
13 changes: 11 additions & 2 deletions lib/benchmark_runner/cli.rb
Original file line number Diff line number Diff line change
Expand Up @@ -110,12 +110,14 @@ def run
puts

# Build the results table
csv_data, csv_layout = csv_view(bench_data)
builder = ResultsTableBuilder.new(
executable_names: ruby_descriptions.keys,
bench_data: bench_data,
bench_data: csv_data,
include_rss: args.rss,
include_pvalue: args.pvalue,
zjit_stats: args.zjit_stats
zjit_stats: args.zjit_stats,
row_layout: csv_layout
)
table, format, gc_table, gc_format = builder.build

Expand Down Expand Up @@ -158,6 +160,13 @@ def run

private

def csv_view(bench_data)
breakdown = RactorBreakdown.expand(bench_data)
return [bench_data, FlatRowLayout.new] if breakdown.groups.empty?

[breakdown.bench_data, RactorRowLayout.new(groups: breakdown.groups)]
end

def build_output_sections(executable_names, bench_data, bench_harnesses, bench_failures)
ordered_names = sorted_benchmark_names(executable_names, bench_data)
failed_names = bench_failures.values.flat_map(&:keys).uniq
Expand Down
54 changes: 43 additions & 11 deletions lib/benchmark_suite.rb
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,8 @@
require_relative 'benchmark_filter'
require_relative 'benchmark_runner'
require_relative 'benchmark_discovery'
require_relative 'ractor_breakdown'
require_relative 'ractor_counts'

# BenchmarkSuite runs a collection of benchmarks and collects their results
class BenchmarkSuite
Expand Down Expand Up @@ -48,22 +50,16 @@ def run_benchmark(entry, ruby:, ruby_description:)
env = benchmark_env(ruby)
caller_json_path = ENV["RESULT_JSON_PATH"]
quiet = ENV['BENCHMARK_QUIET'] == '1'

result_json_path = caller_json_path || File.join(out_path, "temp#{Process.pid}.json")
cmd_prefix = base_cmd(ruby_description, entry.name)

# Clear project-level Bundler environment so benchmarks run in a clean context.
# Benchmarks that need Bundler (e.g., railsbench) set up their own via use_gemfile.
benchmark_harness = benchmark_harness_for(entry.name)

result = if defined?(Bundler)
Bundler.with_unbundled_env do
run_single_benchmark(entry.script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: quiet)
end
else
run_single_benchmark(entry.script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: quiet)
if benchmark_harness == RACTOR_HARNESS
return run_ractor_benchmark(entry, ruby, cmd_prefix, env, benchmark_harness, caller_json_path, quiet: quiet)
end

result_json_path = caller_json_path || File.join(out_path, "temp#{Process.pid}.json")
result = run_benchmark_process(entry.script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: quiet)

if result[:success]
{ name: entry.name, data: process_benchmark_result(result_json_path, result[:command], delete_file: !caller_json_path), harness: benchmark_harness }
else
Expand Down Expand Up @@ -106,6 +102,42 @@ def setup_benchmark_directories
end
end

def run_ractor_benchmark(entry, ruby, cmd_prefix, env, benchmark_harness, caller_json_path, quiet: false)
blobs_by_count = {}

RactorCounts.from_env.each do |count|
result_json_path = File.join(out_path, "temp#{Process.pid}_r#{count}.json")
count_env = env.merge(RactorCounts::ENV_VAR => count.to_s)
result = run_benchmark_process(entry.script_path, result_json_path, ruby, cmd_prefix, count_env, benchmark_harness, quiet: quiet)

unless result[:success]
FileUtils.rm_f(result_json_path)
return { name: entry.name, failure: result[:status].exitstatus, harness: benchmark_harness }
end
command = "#{RactorCounts::ENV_VAR}=#{count} #{result[:command]}"
blobs_by_count[count] = process_benchmark_result(result_json_path, command)
end

data = RactorBreakdown.merge(blobs_by_count)
if caller_json_path
FileUtils.mkdir_p(File.dirname(caller_json_path))
File.write(caller_json_path, JSON.pretty_generate(data))
end
{ name: entry.name, data: data, harness: benchmark_harness }
end

# Clear project-level Bundler environment so benchmarks run in a clean context.
# Benchmarks that need Bundler (e.g., railsbench) set up their own via use_gemfile.
def run_benchmark_process(script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: false)
if defined?(Bundler)
Bundler.with_unbundled_env do
run_single_benchmark(script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: quiet)
end
else
run_single_benchmark(script_path, result_json_path, ruby, cmd_prefix, env, benchmark_harness, quiet: quiet)
end
end

def process_benchmark_result(result_json_path, command, delete_file: true)
JSON.parse(File.read(result_json_path)).tap do |json|
json["command_line"] = command
Expand Down
31 changes: 29 additions & 2 deletions lib/ractor_breakdown.rb
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@

module RactorBreakdown
KEY_SEP = "\x00"
MEASUREMENT_KEYS = %w[warmup bench warmup_by_ractors bench_by_ractors gc_by_ractors].freeze
PROCESS_KEYS = %w[rss maxrss yjit_stats zjit_stats zjit_stats_string command_line].freeze

Result = Struct.new(:bench_data, :groups)

Expand All @@ -15,6 +17,29 @@ def base_name(data_key)
data_key.split(KEY_SEP, 2).first
end

def merge(blobs_by_count)
counts = blobs_by_count.keys.sort
process_data = counts.to_h do |count|
[count.to_s, blobs_by_count[count].reject { |k, _| MEASUREMENT_KEYS.include?(k) }]
end
first, *rest = process_data.values
merged = first.select do |k, v|
!PROCESS_KEYS.include?(k) && rest.all? { |data| data.key?(k) && data[k] == v }
end

merged['warmup'] = counts.flat_map { |count| blobs_by_count[count]['warmup'] }
merged['bench'] = counts.flat_map { |count| blobs_by_count[count]['bench'] }
merged['warmup_by_ractors'] = counts.to_h { |count| [count.to_s, blobs_by_count[count]['warmup']] }
merged['bench_by_ractors'] = counts.to_h { |count| [count.to_s, blobs_by_count[count]['bench']] }
gc_by_ractors = counts.filter_map do |count|
group = blobs_by_count[count].dig('gc_by_ractors', count.to_s)
[count.to_s, group] if group
end.to_h
merged['gc_by_ractors'] = gc_by_ractors unless gc_by_ractors.empty?
merged['results_by_ractors'] = process_data.transform_values { |data| data.reject { |k, _| merged.key?(k) } }
merged
end

def expand(bench_data)
groups = {}
new_data = {}
Expand Down Expand Up @@ -42,9 +67,11 @@ def expand(bench_data)
end

def per_count_blob(blob, breakdown, count)
per_count = blob.reject { |k, _| k == 'bench_by_ractors' || k == 'gc_by_ractors' || k == 'bench' }
per_count = blob.reject { |k, _| %w[warmup_by_ractors bench_by_ractors gc_by_ractors results_by_ractors bench].include?(k) }
process_data = blob['results_by_ractors']
per_count.merge!(process_data[count.to_s]) if process_data.is_a?(Hash) && process_data.key?(count.to_s)
per_count['bench'] = breakdown[count.to_s]
per_count['warmup'] = []
per_count['warmup'] = blob.fetch('warmup_by_ractors').fetch(count.to_s)
gc_by_ractors = blob['gc_by_ractors']
if gc_by_ractors.is_a?(Hash) && gc_by_ractors.key?(count.to_s)
per_count.merge!(gc_by_ractors[count.to_s])
Expand Down
Loading
Loading