Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
93 changes: 93 additions & 0 deletions .github/workflows/benchmarks.yml
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,99 @@ jobs:
path: results/
if-no-files-found: warn

# Builds the results site from every job's results, also when some jobs
# failed, so a PR gets a preview (the site-preview artifact, named outside
# the results-* pattern so a re-run never downloads it). Skipped when the
# benchmark jobs did not run (a lint failure). Only a push to
# main where every job passed publishes it, so a partial run never replaces
# the live site. Not required: a Pages problem never blocks a merge.
site:
needs: [changes, rake]
if: ${{ !cancelled() && needs.changes.outputs.run == 'true' && needs.rake.result != 'skipped' }}
runs-on: ubuntu-latest

steps:
- uses: actions/checkout@v4
- uses: actions/download-artifact@v4
with:
pattern: results-*
path: results/
merge-multiple: true
- name: Build the results site
# A failed benchmark job leaves no result file, like a benchmark that needs a newer Ruby, so the page says missing results may be crashes.
env:
RESULTS_INCOMPLETE: ${{ needs.rake.result != 'success' && '1' || '' }}
run: docker compose run --rm -T -e RESULTS_INCOMPLETE --entrypoint ruby ruby_4.0 script/build_results_site.rb results _site
- name: Upload the site preview
id: preview
uses: actions/upload-artifact@v4
with:
name: site-preview
path: _site/
# A re-run replaces the earlier attempt's preview.
overwrite: true
# A one-click link on the run's Summary page; it needs no extra permissions, so it works for PRs from forks too.
- name: Link the site preview in the run summary
env:
PREVIEW_URL: ${{ steps.preview.outputs.artifact-url }}
run: |
{
echo "### Results site preview"
echo ""
echo "[Download the site preview]($PREVIEW_URL) (a zip, needs a GitHub login), unzip it and open \`index.html\` in a browser."
} >> "$GITHUB_STEP_SUMMARY"
- name: Upload the site for GitHub Pages
if: github.event_name == 'push' && github.ref_name == 'main' && needs.rake.result == 'success'
uses: actions/upload-pages-artifact@v5
with:
path: _site/

deploy:
needs: [rake, site]
if: github.event_name == 'push' && github.ref_name == 'main' && needs.rake.result == 'success' && needs.site.result == 'success'
runs-on: ubuntu-latest
# Only this job can publish; the rest of the workflow keeps the default
# token permissions.
permissions:
pages: write
id-token: write
environment:
name: github-pages
url: ${{ steps.deployment.outputs.page_url }}
# One deploy at a time; a newer merge waits instead of cancelling one that
# is halfway through.
concurrency:
group: pages
cancel-in-progress: false

steps:
# Runs can finish out of order, so an older run must not put its results back over a newer one.
# When main moved on, ask pick-benchmarks.sh (the same rule CI uses) whether the newer commits run benchmarks.
# If they do, their own run deploys newer results; if not (a README-only merge), this run's results are still the newest.
- uses: actions/checkout@v4
with:
ref: main
fetch-depth: 0
- name: Check no newer run on main will deploy
id: newest
run: |
if [ "$(git rev-parse HEAD)" = "$GITHUB_SHA" ]; then
echo "deploy=true" >> "$GITHUB_OUTPUT"
exit 0
fi
GITHUB_EVENT_NAME=push GITHUB_OUTPUT=newer.txt .github/scripts/pick-benchmarks.sh "$GITHUB_SHA"
if grep -q '^run=true' newer.txt; then
echo "main moved on and its newer commits run benchmarks, so their run deploys; skipping."
echo "deploy=false" >> "$GITHUB_OUTPUT"
else
echo "main moved on, but nothing since this run affects benchmarks; deploying."
echo "deploy=true" >> "$GITHUB_OUTPUT"
fi
- name: Deploy to GitHub Pages
id: deployment
if: steps.newest.outputs.deploy == 'true'
uses: actions/deploy-pages@v5

# The check to require on main. Passes when every benchmark job passed, or
# when there was nothing to benchmark.
benchmarks-ok:
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -2,3 +2,4 @@
*.bundle
/Gemfile.lock
/results/
/_site/
10 changes: 10 additions & 0 deletions CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -112,6 +112,16 @@ it ran on:
RESULTS_DIR=results RESULTS_LABEL=ruby_3.4 docker compose run --rm ruby_3.4 code/your-new/entry.rb
```

To see those results the way the results site shows them, build the site
into `_site/`, then open `_site/index.html` in a browser:

```
docker compose run --rm -T --entrypoint ruby ruby_4.0 script/build_results_site.rb results _site
```

CI does the same after every run: the site is attached to the run as the
`site-preview` artifact, and published to GitHub Pages from `main`.

## Benchmarks that need a newer Ruby

CI runs every benchmark on every Ruby in `compose.yaml`, back to Ruby 2.1, and
Expand Down
179 changes: 179 additions & 0 deletions script/build_results_site.rb
Original file line number Diff line number Diff line change
@@ -0,0 +1,179 @@
# Builds the results site from the JSON written by docker/collect_results.rb:
#
# ruby script/build_results_site.rb [results_dir] [output_dir]
#
# results_dir defaults to results/ and output_dir to _site/. Result files are
# found at any depth, so both a local results/<label>/ folder and CI's merged
# artifacts work. The site answers one question per benchmark: does the
# report the file claims is fastest win on this Ruby and build? Every verdict
# compares reports inside one run, never one Ruby or build against another:
# each build ran as its own CI job, often on a different CPU model.
require "fileutils"
require "json"

module ResultsSite
REPO_URL = "https://github.com/fastruby/fast-ruby".freeze
ENGINES = %w[ruby jruby truffleruby].freeze
TEMPLATE = File.expand_path("results_site.html", __dir__)

# Over a billion iterations per second is under 1 ns per iteration: the JIT
# most likely removed the work (TruffleRuby reports hundreds of billions).
SUSPECT_IPS = 1e9

# Sections come from the folder under code/, in this order.
# A folder not listed here is shown after these, with its name capitalized.
SECTIONS = {
"general" => "General",
"method" => "Method Invocation",
"array" => "Array",
"enumerable" => "Enumerable",
"date" => "Date",
"hash" => "Hash",
"proc-and-block" => "Proc & Block",
"string" => "String",
"time" => "Time",
"range" => "Range",
"bigdecimal" => "BigDecimal"
}.freeze

module_function

def build(results_dir, output_dir)
# Only result files: anything else under results_dir
# (a site built there earlier, another artifact) has no "entries".
reports = Dir[File.join(results_dir, "**", "*.json")].sort.map { |path| JSON.parse(File.read(path)) }
reports.select! { |r| r.is_a?(Hash) && r["entries"].is_a?(Array) && r["label"] }
abort "No results found in #{results_dir}" if reports.empty?

data = site_data(reports)
FileUtils.rm_rf(output_dir)
FileUtils.mkdir_p(output_dir)
File.write(File.join(output_dir, "index.html"), render(data))
File.write(File.join(output_dir, "results.json"), JSON.generate(data))
puts "Built #{data[:benchmarks].size} benchmarks on #{data[:meta][:labels].size} builds into #{output_dir}"
end

# The claim is the first report: the lint guarantees the claimed one (the
# report calling `fastest`, `faster` or `fast`) is listed first. A tie is
# benchmark-ips' "same-ish": the i/s +- error ranges overlap, with the same
# comparison as Benchmark::IPS::Stats::StatsMetric#overlaps?.
def verdict(entries, claim)
sorted = entries.sort_by { |e| -e["ips"] }
top = sorted.first
claimed = entries.find { |e| e["name"] == claim }
suspect = entries.any? { |e| e["ips"] > SUSPECT_IPS }
return { state: "na", suspect: suspect } unless claimed

if claimed.equal?(top)
second = sorted[1]
if second.nil? || overlaps?(top, second)
{ state: "eq", suspect: suspect }
else
{ state: "ok", ratio: top["ips"] / second["ips"], suspect: suspect }
end
elsif overlaps?(claimed, top)
{ state: "eq", suspect: suspect }
else
{ state: "no", ratio: top["ips"] / claimed["ips"], winner: top["name"], suspect: suspect }
end
end

def overlaps?(a, b)
a["ips"] + a["error"] > b["ips"] - b["error"] && a["ips"] - a["error"] < b["ips"] + b["error"]
end

def site_data(reports)
labels = reports.map { |r| r["label"] }.uniq.sort_by { |l| label_sort_key(l) }
grouped = reports.group_by { |r| [r["benchmark"], r["part"]] }
multi_part = grouped.keys.group_by(&:first).select { |_, parts| parts.size > 1 }.keys

benchmarks = grouped.map do |(file, part), runs|
# The run with the most reports has them all; older Rubies can skip a
# report guarded by RUBY_VERSION.
names = runs.max_by { |r| r["entries"].size }["entries"].map { |e| e["name"] }
title = title_for(names)
title += " (part #{part})" if multi_part.include?(file)
{
file: file,
part: part,
title: title,
section: section_for(file),
names: names,
runs: runs.to_h { |r| [r["label"], run_data(r, names)] }
}
end

order = SECTIONS.values
benchmarks.sort_by! { |b| [order.index(b[:section]) || order.size, b[:section], b[:file], b[:part]] }
# Two files with the same report labels get their file name added.
benchmarks.group_by { |b| b[:title] }.each_value do |same|
same.each { |b| b[:title] += " (#{File.basename(b[:file])})" } if same.size > 1
end

{ meta: meta(reports, labels), benchmarks: benchmarks }
end

# Per build: the verdict, and ips/error per report in the order of `names`
# (nil where that report did not run).
def run_data(report, names)
entries = report["entries"]
v = verdict(entries, names.first)
by_name = entries.to_h { |e| [e["name"], e] }
v[:ratio] = v[:ratio].round(4) if v[:ratio]
v[:winner] = names.index(v[:winner]) if v[:winner]
v.merge(entries: names.map { |n| (e = by_name[n]) && [e["ips"].round(3), e["error"].to_f.round(1)] })
end

def meta(reports, labels)
first = reports.first
by_label = reports.group_by { |r| r["label"] }.transform_values(&:first)
{
commit: first["commit"],
run_id: first["run_id"],
pr: first["pr"],
# Set by CI when a benchmark job failed: its results are missing, which looks like a skip.
incomplete: ENV["RESULTS_INCOMPLETE"] == "1",
built_at: Time.now.utc.strftime("%Y-%m-%d %H:%M UTC"),
repo: REPO_URL,
labels: labels,
default: default_label(labels),
descriptions: by_label.transform_values { |r| r["ruby_description"] },
cpus: by_label.transform_values { |r| r.dig("environment", "cpu") }
}
end

# The newest released MRI: head builds and JIT variants are left out.
def default_label(labels)
labels.select { |l| l.match?(/\Aruby_\d[\d.]*\z/) }.max_by { |l| Gem::Version.new(l.delete_prefix("ruby_")) } || labels.first
end

# ruby, then jruby, then truffleruby; oldest to newest, head last; the plain
# build before its JIT variants.
def label_sort_key(label)
base, variant = label.split("+", 2)
engine, version = base.match(/\A([a-z]+)_(.+)\z/)&.captures || [base, ""]
head = version == "head" ? 1 : 0
[ENGINES.index(engine) || ENGINES.size, head, version.scan(/\d+/).map(&:to_i), variant.to_s]
end

# The title is the report labels, claimed report first: "`String#tr` vs `String#gsub`".
# A label with a backtick in it is left as plain text, so it cannot break the code style.
def title_for(names)
names.map { |n| n.include?("`") ? n : "`#{n}`" }.join(" vs ")
end

# code/proc-and-block/proc-call-vs-yield.rb is in "Proc & Block".
def section_for(file)
folder = file.split("/")[1].to_s
SECTIONS.fetch(folder) { folder.split("-").map(&:capitalize).join(" ") }
end

# The data goes inline so the page also works opened from disk. "</" is
# escaped so a report name can never close the script tag.
def render(data)
json = JSON.generate(data).gsub("</", "<\\/").gsub("\u2028", "\\u2028").gsub("\u2029", "\\u2029")
File.read(TEMPLATE).sub("/*RESULTS_DATA*/") { "window.RESULTS = #{json};" }
end
end

ResultsSite.build(ARGV[0] || "results", ARGV[1] || "_site") if $PROGRAM_NAME == __FILE__
Loading
Loading