Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -406,6 +406,12 @@ set (Seastar_DEBUG_SHARED_PTR
STRING
"Enable shared_ptr debugging. Can be ON, OFF or DEFAULT (which enables it for Debug and Sanitize)")

set (Seastar_PERF_TESTS_HONOR_DECLARED_RATE
"DEFAULT"
CACHE
STRING
"Honor an iteration rate declared by a perf test. Can be ON, OFF or DEFAULT (which enables it for release builds)")

#
# Useful (non-cache) variables.
#
Expand Down Expand Up @@ -1364,6 +1370,16 @@ if (Seastar_INSTALL OR Seastar_TESTING)
add_library (Seastar::seastar_perf_testing ALIAS seastar_perf_testing)
target_compile_definitions (seastar_perf_testing
PRIVATE ${Seastar_PRIVATE_COMPILE_DEFINITIONS})
# A declared iteration rate only holds for an optimized build, so by default
# only those honor one. configure.py's release mode maps to RelWithDebInfo.
tri_state_option (${Seastar_PERF_TESTS_HONOR_DECLARED_RATE}
DEFAULT_BUILD_TYPES "Release" "RelWithDebInfo"
CONDITION condition)
if (condition)
target_compile_definitions (seastar_perf_testing
PUBLIC
$<${condition}:SEASTAR_PERF_TESTS_HONOR_DECLARED_RATE=1>)
endif ()
target_compile_options (seastar_perf_testing
PRIVATE ${Seastar_PRIVATE_CXX_FLAGS})
target_link_libraries (seastar_perf_testing
Expand Down
7 changes: 7 additions & 0 deletions configure.py
Original file line number Diff line number Diff line change
Expand Up @@ -212,6 +212,11 @@ def resolve_compilers_for_compiler_cache(args, compiler_cache):
name='debug-shared-ptr',
dest="debug_shared_ptr",
help='Debug shared_ptr')
add_tristate(
arg_parser,
name='perf-test-declared-rates',
dest='perf_test_declared_rates',
help='honoring the iteration rates declared by perf tests')
add_tristate(
arg_parser,
name='io_uring',
Expand Down Expand Up @@ -318,6 +323,8 @@ def configure_mode(mode):
tr(args.deferred_action_require_noexcept, 'DEFERRED_ACTION_REQUIRE_NOEXCEPT'),
tr(args.unused_result_error, 'UNUSED_RESULT_ERROR'),
tr(args.debug_shared_ptr, 'DEBUG_SHARED_PTR', value_when_none='default'),
tr(args.perf_test_declared_rates, 'PERF_TESTS_HONOR_DECLARED_RATE',
value_when_none='DEFAULT'),
]

if not which('ninja-build') and which('ninja'):
Expand Down
71 changes: 60 additions & 11 deletions include/seastar/testing/perf_tests.hh
Original file line number Diff line number Diff line change
Expand Up @@ -41,11 +41,45 @@ using seastar::future;
using seastar::noncopyable_function;
using seastar::is_future;

// Whether a rate declared with perf_tests::test_options::iters_per_sec is
// honored. Such a rate is measured on an optimized build, and a build with
// sanitizers, debug checks or no inlining runs an order of magnitude slower, so
// holding the iteration count fixed there would stretch every run by that
// factor to produce numbers that are not comparable to an optimized build's
// anyway. The build system therefore opts in: seastar's CMake build defines
// this for the release build types and its Bazel build for optimized builds.
// Everywhere else a declared rate is ignored and the count is calibrated by the
// dry run as usual.
#ifndef SEASTAR_PERF_TESTS_HONOR_DECLARED_RATE
#define SEASTAR_PERF_TESTS_HONOR_DECLARED_RATE 0
#endif

namespace perf_tests {

// The type of the pre-run hook. See PERF_PRE_RUN_HOOK.
using pre_run_hook = noncopyable_function<void(const sstring& test_group, const sstring& test_case)>;

// Optional declarations about a test, passed as a trailing argument to the
// PERF_TEST macros with designated initializers:
//
// PERF_TEST(my_group, my_case, .iters_per_sec = 10'000'000) { ... }
//
// Default initializing this struct results in the default options.
struct test_options {
// The number of iterations the test completes in one second. When non-zero,
// the iteration count of a run is fixed at iters_per_sec * --duration rather
// than being calibrated by a timed dry run, so every run executes the same
// number of iterations no matter what machine it runs on, and the duration
// of a run becomes the approximate quantity instead. An explicit
// --iterations takes precedence over this value, which is in turn only
// honored in a build that sets SEASTAR_PERF_TESTS_HONOR_DECLARED_RATE.
//
// A double so that a rate can be written in scientific notation - .iters_per_sec
// = 1.2e6 as well as 1'200'000. A value that is not finite and positive is
// ignored, leaving the count to be calibrated.
double iters_per_sec = 0;
};

namespace internal {

struct config;
Expand Down Expand Up @@ -127,6 +161,7 @@ inline perf_stats& perf_stats::operator-=(perf_stats b) {
class performance_test {
std::string _test_case;
std::string _test_group;
test_options _options;

uint64_t _single_run_iterations = 0;
std::atomic<uint64_t> _max_single_run_iterations;
Expand Down Expand Up @@ -160,15 +195,18 @@ protected:
void start_run();
run_result stop_run();
public:
performance_test(const std::string& test_case, const std::string& test_group)
performance_test(const std::string& test_case, const std::string& test_group,
test_options options = {})
: _test_case(test_case)
, _test_group(test_group)
, _options(options)
{ }

virtual ~performance_test() = default;

const std::string& test_case() const { return _test_case; }
const std::string& test_group() const { return _test_group; }
const test_options& options() const { return _options; }
std::string name() const { return fmt::format("{}.{}", test_group(), test_case()); }

void run(const config&);
Expand Down Expand Up @@ -236,8 +274,9 @@ public:

template<typename Test>
struct test_registrar {
test_registrar(const std::string& test_group, const std::string& test_case) {
auto test = std::make_unique<concrete_performance_test<Test>>(test_case, test_group);
test_registrar(const std::string& test_group, const std::string& test_case,
test_options options = {}) {
auto test = std::make_unique<concrete_performance_test<Test>>(test_case, test_group, options);
performance_test::register_test(std::move(test));
}
};
Expand Down Expand Up @@ -284,38 +323,48 @@ void do_not_optimize(const T& v)
// the test function shall return either size_t or future<size_t> for synchronous and
// asynchronous cases respectively. The returned value shall be the number of iterations
// done in a single test run.
//
// All four macros accept an optional trailing argument declaring perf_tests::test_options
// for the test, written with designated initializers:
//
// PERF_TEST(my_group, my_case, .iters_per_sec = 10'000'000) { ... }
//

#define PERF_TEST_F(test_group, test_case) \
#define PERF_TEST_F(test_group, test_case, ...) \
struct test_##test_group##_##test_case : test_group { \
[[gnu::always_inline]] inline auto run(); \
}; \
static ::perf_tests::internal::test_registrar<test_##test_group##_##test_case> \
test_##test_group##_##test_case##_registrar(#test_group, #test_case); \
test_##test_group##_##test_case##_registrar(#test_group, #test_case \
__VA_OPT__(, ::perf_tests::test_options{__VA_ARGS__})); \
[[gnu::always_inline]] auto test_##test_group##_##test_case::run()

#define PERF_TEST(test_group, test_case) \
#define PERF_TEST(test_group, test_case, ...) \
struct test_##test_group##_##test_case { \
[[gnu::always_inline]] inline auto run(); \
}; \
static ::perf_tests::internal::test_registrar<test_##test_group##_##test_case> \
test_##test_group##_##test_case##_registrar(#test_group, #test_case); \
test_##test_group##_##test_case##_registrar(#test_group, #test_case \
__VA_OPT__(, ::perf_tests::test_options{__VA_ARGS__})); \
[[gnu::always_inline]] auto test_##test_group##_##test_case::run()


#define PERF_TEST_C(test_group, test_case) \
#define PERF_TEST_C(test_group, test_case, ...) \
struct test_##test_group##_##test_case : test_group { \
inline future<> run(); \
}; \
static ::perf_tests::internal::test_registrar<test_##test_group##_##test_case> \
test_##test_group##_##test_case##_registrar(#test_group, #test_case); \
test_##test_group##_##test_case##_registrar(#test_group, #test_case \
__VA_OPT__(, ::perf_tests::test_options{__VA_ARGS__})); \
future<> test_##test_group##_##test_case::run()

#define PERF_TEST_CN(test_group, test_case) \
#define PERF_TEST_CN(test_group, test_case, ...) \
struct test_##test_group##_##test_case : test_group { \
inline future<size_t> run(); \
}; \
static ::perf_tests::internal::test_registrar<test_##test_group##_##test_case> \
test_##test_group##_##test_case##_registrar(#test_group, #test_case); \
test_##test_group##_##test_case##_registrar(#test_group, #test_case \
__VA_OPT__(, ::perf_tests::test_options{__VA_ARGS__})); \
future<size_t> test_##test_group##_##test_case::run()


Expand Down
54 changes: 51 additions & 3 deletions tests/perf/perf-tests.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,15 +16,20 @@ combined.one_row 745336 691.218ns 0.175ns 689.073ns
combined.single_active 7871 85.271us 76.185ns 85.145us 108.316us
```

`perf-tests` allows limiting the number of iterations or the duration of each run. In the latter case there is an additional dry run used to estimate how many iterations can be run in the specified time. The measured runs are limited by that number of iterations. This means that there is no overhead caused by timers and that each run consists of the same number of iterations.
`perf-tests` allows limiting the number of iterations or the duration of each
run. If duration is used, the implementation depends on whether a test declares
iters_per_sec. If it does the total iteration count is calculated directly
based on that value, and if not an additional dry run is used to estimate how
many iterations can be run in the specified time.

### Flags

* `-i <n>` or `--iterations <n>` – limits the number of iterations in each run to no more than `n` (0 for unlimited)
* `-d <t>` or `--duration <t>` – limits the duration of each run to no more than `t` seconds (0 for unlimited)
* `-i <n>` or `--iterations <n>` – fixes the number of iterations in each run at `n`, however long that takes (0 to let `--duration` decide)
* `-d <t>` or `--duration <t>` – limits the duration of each run to no more than `t` seconds, unless `--iterations` has already fixed the count (0 for unlimited)
* `-r <n>` or `--runs <n>` – the number of runs of each test to execute
* `-t <regexs>` or `--tests <regexs>` – executes only tests which names match any regular expression in a comma-separated list `regexs`
* `--list` – lists all available tests
* `--suggest-rates` – after the results, print the iteration rate measured for each test as a `.iters_per_sec` declaration to paste into its `PERF_TEST` macro
* `--overhead-threshold <percent>` – warn if measurement overhead exceeds this percentage (default: 10)
* `--fail-on-high-overhead` – fail the test run if any test exceeds the overhead threshold

Expand Down Expand Up @@ -138,3 +143,46 @@ WARNING: test 'example.my_test' has high measurement overhead: 15.2% (threshold:
```

You can adjust the threshold with `--overhead-threshold <percent>`, or fail the test run entirely when overhead is too high with `--fail-on-high-overhead`.

### Fixed iteration counts

Calibrating the iteration count with a dry run keeps the duration of a run stable, but makes the number of iterations depend on how fast the machine is and on whatever else it was doing during the dry run. A test which declares how many iterations it completes in one second gets the opposite trade-off: the count is fixed at the declared rate times `--duration`, no dry run is used to pick it, and the duration of a run becomes the approximate quantity.

```c++
PERF_TEST(example, declared_rate, .iters_per_sec = 10'000'000)
{
auto v = compute_value();
perf_tests::do_not_optimize(v);
}
```

The rate is a `double`, so it can be written in scientific notation - `.iters_per_sec = 1.2e6` as well as `1'200'000`.

A declared rate is only honored in a release build. The rate is a measurement of one, and a build with sanitizers or debug checks runs an order of magnitude slower, so holding the count fixed there would stretch every run by that factor to produce numbers that are not comparable to a release build's anyway. The opt-in is the `SEASTAR_PERF_TESTS_HONOR_DECLARED_RATE` macro, which the CMake build defines for the release build types and the Bazel build for `--config=release`. Either can be told otherwise, to hold the count fixed in a slow build or to calibrate it in a fast one: pass `--enable-perf-test-declared-rates` or `--disable-perf-test-declared-rates` to `configure.py`, or set `--@seastar//:perf_test_declared_rates` under Bazel. In a build which does not honor them the count is calibrated by the dry run as usual, and the configuration header says so:

```
declared rates: ignored (not a release build)
```

With the default `--duration 1` that test runs exactly 10,000,000 iterations per run; with `--duration 5`, exactly 50,000,000. The rate is expressed in the same iterations that `--iterations` limits and the `iters` column reports, so a test which returns an iteration count from its body declares its rate in those inner iterations.

An explicit `--iterations` overrides the declaration, and `--duration 0` (no duration limit) disables it, since there is then no duration to scale by.

The rate is a measurement, so the framework will take it for you: `--suggest-rates` prints one pasteable declaration per test, measured on the wall clock (what `--duration` actually limits) and rounded to two significant digits.

```
measured iteration rates, to declare in the PERF_TEST macro so that a
run's iteration count no longer depends on the speed of the machine:

example.simple1 .iters_per_sec = 2'700'000'000
example.declared_rate .iters_per_sec = 52'000'000
example.big_inner_loop .iters_per_sec = 540
```

It works whether or not the test already declares a rate, so the same command both writes the declarations and refreshes them.

A declared rate is a measurement of one machine, so it only approximates the duration on another. It does not affect the reported results, which are always per-iteration, so a rate that is out of date costs nothing but a run that is shorter or longer than asked for. Once a run strays more than a factor of two from the requested duration, the achieved rate is reported so the declaration can be refreshed:

```
WARNING: test 'example.declared_rate' declares 100000 iterations/s but achieved 2.87e+07/s, so each run took 3.484ms rather than the requested 1.000s
```
Loading
Loading