From 765cd3823aea1052f1c5dfc9b83f254c30bf4843 Mon Sep 17 00:00:00 2001 From: gilice Date: Sat, 8 Mar 2025 16:09:45 +0100 Subject: [PATCH] bench: rewrite in python, add instruction count profiling This is to some extent a simple transliteration of bench.sh, but also adds some additional features: - the ability to select which benchmarks to run, instead of having to run all at once (via --cases option) - help text (--help), and argument parsing - ability to profile instruction counts (via --mode icount). Instruction count is a stable metric, that correlates to the execution time, but is less affected by the environment (like other heavy programs running, or the system hardware). Other projects (most notably rustc) use instruction count as a primary target metric. Change-Id: I791cac499e64983efb8bcd4ea1faef13ded09ee6 --- bench/.gitignore | 1 + bench/bench.py | 116 +++++++++++++++++++++++++++++++++++++++++++++ bench/bench.sh | 68 -------------------------- bench/summarize.jq | 22 --------- 4 files changed, 117 insertions(+), 90 deletions(-) create mode 100755 bench/bench.py delete mode 100755 bench/bench.sh delete mode 100755 bench/summarize.jq diff --git a/bench/.gitignore b/bench/.gitignore index 8115aa6f2..4cf9c5d25 100644 --- a/bench/.gitignore +++ b/bench/.gitignore @@ -1,3 +1,4 @@ bench-*.json bench-*.md +perf-*.json nixpkgs diff --git a/bench/bench.py b/bench/bench.py new file mode 100755 index 000000000..8dfcf24a2 --- /dev/null +++ b/bench/bench.py @@ -0,0 +1,116 @@ +#!/usr/bin/env nix-shell +#!nix-shell -i python3 -p python3 -p hyperfine -p "if stdenv.isLinux then linuxPackages.perf else null" + +import argparse +import subprocess +import os +import json +import tempfile +import platform + +flake_args = ["--extra-experimental-features","'nix-command flakes'"] +# hyperfine has its own variable substitution, so we use that and pass build="{BUILD}" here. +# perf doesn't have variable substitution, so we call these with build being the actual build directory. +cases = { + "search": lambda build: [f"{build}/bin/nix", *flake_args, "search", "--no-eval-cache", "github:nixos/nixpkgs/e1fa12d4f6c6fe19ccb59cac54b5b3f25e160870", "hello"], + "rebuild": lambda build: [f"{build}/bin/nix", *flake_args, "eval", "--raw", "--impure", "--expr", "'with import {}; system'"], + "rebuild_lh": lambda build: ["GC_INITIAL_HEAP_SIZE=10g", f"{build}/bin/nix", *flake_args, "eval", "--raw", "--impure", "--expr", "'with import {}; system'"], + "parse": lambda build: [f"{build}/bin/nix", *flake_args, "eval", "-f", "bench/nixpkgs/pkgs/development/haskell-modules/hackage-packages.nix"], +} + +arg_parser = argparse.ArgumentParser() +# FIXME(jade, gilice): it is a reasonable use case to want to run a benchmark run +# on just one build. However, since we are using hyperfine in comparison +# mode, we would have to combine the JSON ourselves to support that, which +# would probably be better done by writing a benchmarking script in +# not-bash. +arg_parser.add_argument('builds', nargs='+', help="At least two build directories to compare, containing bin/nix") +arg_parser.add_argument('--cases', type=str, help="A comma-separated list of cases you want to run. Defaults to running all") +available_modes = [ "walltime" ] + [ "icount" ] if platform.system() == 'Linux' else [] # perf doesn't run on Darwin +arg_parser.add_argument('--mode', choices=available_modes, default="walltime") +args = arg_parser.parse_args() +if len(args.builds) < 2: + raise ValueError("need at least two build directories to compare") + +benchmarks: list[str] = [] +if args.cases is None: + benchmarks = list(cases.keys()) +else: + for case in args.cases.split(","): + if case not in cases: raise ValueError(f"no such case: {case}") + benchmarks.append(case) + +def bench_walltime(env): + hyperfine_args = ["--parameter-list", "BUILD", ','.join(args.builds), "--warmup", "2", "--runs", "10"] + for case in benchmarks: + case_command = cases[case]("{BUILD}") # see the comment on cases + subprocess.run([ + "taskset", "-c", "2,3", + "chrt", "-f","50", + "hyperfine", *hyperfine_args, "--export-json", f"bench/bench-{case}.json", "--export-markdown", f"bench/bench-{case}.md", "--", " ".join(case_command) + ], env=env, check=True) + + print("Benchmarks summary\n---\n") + for case in benchmarks: + fd = open(f"bench/bench-{case}.json") + result_json = json.load(fd) + fd.close() + for result in result_json["results"]: + print(result["command"]) + print("-" * min(80,len(result["command"]))) + attr_rounded = lambda attr: f"{result[attr]:.3f}" + print(" mean: ", attr_rounded("mean"), "±", attr_rounded("stddev")) + print(" user:", attr_rounded("user"), "| system", attr_rounded("system")) + print(" median: ", attr_rounded("median")) + print(" range: ", attr_rounded("min") + "s.." + attr_rounded("max")+"s") + print(" relative:", f"{result["mean"]/result_json["results"][0]["mean"]:.3f}") + print("\n") + + +def bench_icount(env): + perf_results_for: dict[str, list[tuple[str, float]]] = {} + for case in benchmarks: + for build in args.builds: + case_command = cases[case](build) + # the perf stat -j output (incorrectly) localizes numbers, which will trip up the json parser. + env["LC_ALL"]="C" + commandline = [ + "perf", "stat", "-o", f"bench/perf-{case}.json", "-j", "sh", "-c", " ".join(case_command) + ] + subprocess.run(commandline, env=env, check=True, stdout=subprocess.DEVNULL) # warmup run + subprocess.run(commandline, env=env, check=True, stdout=subprocess.DEVNULL) + perf_fd = open(f"bench/perf-{case}.json") + perf_data = [json.loads(x) for x in perf_fd.readlines()] + perf_fd.close() + + instr = next(x for x in perf_data if x["event"] in ["instructions", "instructions:u"]) # an implementation of a find_first iterator + if case not in perf_results_for: perf_results_for[case] = [] + perf_results_for[case].append((" ".join(case_command), float(instr["counter-value"]))) + + print("Benchmarks summary\n---\n") + for (case, entries) in perf_results_for.items(): + for entry in entries: + cmd,instr = entry + print(cmd) + print("-" * min(80,len(cmd))) + print(" instructions: ", int(instr)) + print(" relative instructions:", int(instr)/perf_results_for[case][0][1]) + print("\n") + + +with tempfile.TemporaryDirectory() as tmp_dir: + subprocess.run([ + "nix", "build", + "--extra-experimental-features", "nix-command flakes", + "--impure", "--expr",'(builtins.getFlake "git+file:.").inputs.nixpkgs.outPath', + "-o","bench/nixpkgs" + ], check=True) + subenv = os.environ.copy() + subenv["NIX_CONF_DIR"] = "/var/empty" + subenv["NIX_REMOTE"] = tmp_dir + subenv["NIX_PATH"] = "nixpkgs=bench/nixpkgs:nixos-config=bench/configuration.nix" + + if args.mode == "walltime": + bench_walltime(subenv) + else: + bench_icount(subenv) diff --git a/bench/bench.sh b/bench/bench.sh deleted file mode 100755 index 15d8af05a..000000000 --- a/bench/bench.sh +++ /dev/null @@ -1,68 +0,0 @@ -#!/usr/bin/env nix-shell -#!nix-shell -i bash -p bash -p hyperfine - -set -euo pipefail -shopt -s inherit_errexit - -scriptdir=$(cd "$(dirname -- "$0")" ; pwd -P) -cd "$scriptdir/.." - -if [[ $# -lt 2 ]]; then - # FIXME(jade): it is a reasonable use case to want to run a benchmark run - # on just one build. However, since we are using hyperfine in comparison - # mode, we would have to combine the JSON ourselves to support that, which - # would probably be better done by writing a benchmarking script in - # not-bash. - echo "Fewer than two result dirs given, nothing to compare!" >&2 - echo "Pass some directories (with names indicating which alternative they are) with bin/nix in them" >&2 - echo "Usage: ./bench/bench.sh result-1 result-2 [result-3...]" >&2 - exit 1 -fi - -_exit="" -trap "$_exit" EXIT - -flake_args=("--extra-experimental-features" "nix-command flakes") - -# XXX: yes this is very silly. flakes~!! -nix build "${flake_args[@]}" --impure --expr '(builtins.getFlake "git+file:.").inputs.nixpkgs.outPath' -o bench/nixpkgs - -# We must ignore the global config, or else NIX_PATH won't work reliably. -# See https://github.com/NixOS/nix/issues/9574 -export NIX_CONF_DIR='/var/empty' -export NIX_REMOTE="$(mktemp -d)" -_exit='rm -rfv "$NIX_REMOTE"; $_exit' -export NIX_PATH="nixpkgs=bench/nixpkgs:nixos-config=bench/configuration.nix" - -builds=("$@") - -flake_args="${flake_args[*]@Q}" - -hyperfineArgs=( - --parameter-list BUILD "$(IFS=,; echo "${builds[*]}")" - --warmup 2 --runs 10 -) - -declare -A cases -cases=( - [search]="{BUILD}/bin/nix $flake_args search --no-eval-cache github:nixos/nixpkgs/e1fa12d4f6c6fe19ccb59cac54b5b3f25e160870 hello" - [rebuild]="{BUILD}/bin/nix $flake_args eval --raw --impure --expr 'with import {}; system'" - [rebuild-lh]="GC_INITIAL_HEAP_SIZE=10g {BUILD}/bin/nix eval $flake_args --raw --impure --expr 'with import {}; system'" - [parse]="{BUILD}/bin/nix $flake_args eval -f bench/nixpkgs/pkgs/development/haskell-modules/hackage-packages.nix" -) - -benches=( - rebuild - rebuild-lh - search - parse -) - -for k in "${benches[@]}"; do - taskset -c 2,3 \ - chrt -f 50 \ - hyperfine "${hyperfineArgs[@]}" --export-json="bench/bench-${k}.json" --export-markdown="bench/bench-${k}.md" "${cases[$k]}" -done - -echo "Benchmarks summary (from ./bench/summarize.jq bench/bench-*.json)" -bench/summarize.jq bench/*.json diff --git a/bench/summarize.jq b/bench/summarize.jq deleted file mode 100755 index 5d1449108..000000000 --- a/bench/summarize.jq +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env -S jq -Mrf - -def round3: - . * 1000 | round | . / 1000 - ; - -def stats($first): - [ - " mean: \(.mean | round3)s ± \(.stddev | round3)s", - " user: \(.user | round3)s | system: \(.system | round3)s", - " median: \(.median | round3)s", - " range: \(.min | round3)s ... \(.max | round3)s", - " relative: \(.mean / $first.mean | round3)" - ] - | join("\n") - ; - -def fmt($first): - "\(.command)\n" + (. | stats($first)) - ; - -[.results | .[0] as $first | .[] | fmt($first)] | join("\n\n") | (. + "\n\n---\n")