From fef5f4478ad0d5ed723e894be1331d9355352482 Mon Sep 17 00:00:00 2001 From: ddidderr Date: Sat, 11 Jul 2026 14:24:05 +0200 Subject: [PATCH] feat(rust): port benchmark loop and CLI bench mode Move the implementation of programs/benchfn.c into rust/src/benchfn.rs and wire benchmark mode (-b/-e/-i) into the Rust CLI frontend, which previously rejected those options as not yet implemented. `zstd -b1 -i0 FILE` and range runs like `zstd -b5e6 -i0 FILE` work again, including the synthetic-sample benchmark when no file is given. benchfn.rs is a faithful port of the run/timing state machine: BMK_benchFunction keeps the exact loop accounting (first-loop blockResults and errorFn checks, dstSize summed on the first loop only, 0xE5 warm-up of result buffers, nbLoops minimum of 1) and BMK_benchTimedFn keeps the same convergence behavior (x10 workload growth for short runs, budget-based nbLoops estimation, runs below half the run budget re-tried rather than reported, best qualifying run returned). Arithmetic that C leaves to unsigned wrap-around uses wrapping operations so debug builds cannot panic where release C would wrap. ABI notes: BMK_runTime_t and BMK_runOutcome_t are returned by value across the C boundary and BMK_benchParams_t is passed by value, so all three are repr(C) mirrors of benchfn.h; their field offsets are pinned by const asserts in Rust and matching C static asserts in the benchfn.c shim, which is now declaration-only. BMK_timedFnState_t stays opaque, fits the 64-byte BMK_timedFnState_shell (compile-time checked), and is malloc/free-managed so creation and destruction remain interchangeable with C callers. The CLI parses -b (bench mode), -e (range end, digits attach directly, defaulting to 0 like readU32FromChar) and -i (duration in seconds), then dispatches through a new ZSTD_rust_cli_bench bridge in the zstdcli.c shim. The bridge exists because benchmark availability is a C preprocessor property (ZSTD_NOBENCH): orchestration and reporting stay in C benchzstd.c, stripped variants (zstd-small, zstd-compress, zstd-decompress) compile the stub branch and report "benchmark mode is not available in this build", and the Rust side never references benchmark symbols directly. Level clamping against ZSTD_maxCLevel() happens in the bridge, where the symbol is guaranteed to exist whenever benchmarking is compiled in. -T selects the worker count, defaulting to single-threaded like the C bench path; -S (separate files) and --priority=rt remain unimplemented. Makefile updates only extend the Rust source prerequisite lists with benchfn.rs; the helpers-archive plumbing from the timefn commit already links fullbench(-lib/-dll/32) and paramgrill, the benchfn consumers among the C tests. Original C test sources are untouched. Known pre-existing issues, unchanged by this commit: tests/fullbench-lib fails to link at the base commit too (libzstd.a precedes fullbench.c in its link line), and the cli-tests basic/help.sh, compression/levels.sh, compression/golden.sh, and decompression/pass-through.sh scripts fail identically with a base-commit binary because the Rust CLI frontend is still a partial reimplementation. Test Plan: - cd rust && cargo fmt --check && cargo clippy --all-targets -- -D warnings && cargo test --all-targets && cargo build --release - cd rust/cli && cargo fmt --check && cargo clippy --all-targets -- -D warnings && cargo test --all-targets; repeat tests with --no-default-features plus features compression / decompression / (none) - make -C programs zstd; ./programs/zstd -b1 -i0 lib/common/xxhash.c; ./programs/zstd -b5e6 -i0 programs/fileio.c; ./programs/zstd -b1 -i0 (synthetic); echo roundtrip via zstd | zstd -d - make -C programs zstd-small zstd-compress zstd-decompress zstd-nolegacy zstd-dictBuilder; zstd-small -b reports benchmark unavailable; compress/ decompress roundtrip across the split binaries - make -C tests fullbench fuzzer zstreamtest paramgrill decodecorpus poolTests fullbench32 fuzzer32; ./tests/fullbench -i0 (exercises Rust BMK_benchTimedFn from C); ./tests/fullbench32 -i0; ./tests/fuzzer -i1 --no-big-tests; ./tests/poolTests; make -C tests test-rust-lib-smoke - cli-tests subset: basic/version.sh, compression/basic.sh, compression/multiple-files.sh pass; failing scripts match the base commit Refs: rust/README.md --- programs/Makefile | 2 +- programs/benchfn.c | 260 ++----------------- programs/zstdcli.c | 51 ++++ rust/README.md | 17 +- rust/cli/src/lib.rs | 2 + rust/src/benchfn.rs | 575 +++++++++++++++++++++++++++++++++++++++++++ rust/src/zstd_cli.rs | 152 +++++++++++- tests/Makefile | 2 +- 8 files changed, 805 insertions(+), 256 deletions(-) create mode 100644 rust/src/benchfn.rs diff --git a/programs/Makefile b/programs/Makefile index bcd268e99..689e53893 100644 --- a/programs/Makefile +++ b/programs/Makefile @@ -33,7 +33,7 @@ RUST_SOURCES := $(RUST_MANIFEST) $(RUST_DIR)/Cargo.lock \ $(shell find $(RUST_DIR)/src -type f -name '*.rs' -print) RUST_CLI_SOURCES := $(RUST_CLI_MANIFEST) $(RUST_CLI_DIR)/Cargo.lock \ $(RUST_CLI_DIR)/src/lib.rs $(RUST_DIR)/src/zstd_cli.rs \ - $(RUST_DIR)/src/timefn.rs + $(RUST_DIR)/src/timefn.rs $(RUST_DIR)/src/benchfn.rs # Keep Rust's HUF implementation in lockstep with libzstd.mk's C selection. # Forced modes may arrive as libzstd.mk variables or as direct -D flags in diff --git a/programs/benchfn.c b/programs/benchfn.c index 3e042cf38..17d7cd1e1 100644 --- a/programs/benchfn.c +++ b/programs/benchfn.c @@ -8,249 +8,23 @@ * You may select, at your option, one of the above-listed licenses. */ +/* The implementation lives in rust/src/benchfn.rs, built into the Rust CLI + * static archive. This translation unit stays in the original source lists so + * build configuration keeps working while the implementation is in Rust. */ - -/* ************************************* -* Includes -***************************************/ -#include /* malloc, free */ -#include /* memset */ -#include /* assert */ - -#include "timefn.h" /* UTIL_time_t, UTIL_getTime */ +#include /* size_t, offsetof */ #include "benchfn.h" - -/* ************************************* -* Constants -***************************************/ -#define TIMELOOP_MICROSEC SEC_TO_MICRO /* 1 second */ -#define TIMELOOP_NANOSEC (1*1000000000ULL) /* 1 second */ - -#define KB *(1 <<10) -#define MB *(1 <<20) -#define GB *(1U<<30) - - -/* ************************************* -* Debug errors -***************************************/ -#if defined(DEBUG) && (DEBUG >= 1) -# include /* fprintf */ -# define DISPLAY(...) fprintf(stderr, __VA_ARGS__) -# define DEBUGOUTPUT(...) { if (DEBUG) DISPLAY(__VA_ARGS__); } -#else -# define DEBUGOUTPUT(...) -#endif - - -/* error without displaying */ -#define RETURN_QUIET_ERROR(retValue, ...) { \ - DEBUGOUTPUT("%s: %i: \n", __FILE__, __LINE__); \ - DEBUGOUTPUT("Error : "); \ - DEBUGOUTPUT(__VA_ARGS__); \ - DEBUGOUTPUT(" \n"); \ - return retValue; \ -} - -/* Abort execution if a condition is not met */ -#define CONTROL(c) { if (!(c)) { DEBUGOUTPUT("error: %s \n", #c); abort(); } } - - -/* ************************************* -* Benchmarking an arbitrary function -***************************************/ - -int BMK_isSuccessful_runOutcome(BMK_runOutcome_t outcome) -{ - return outcome.error_tag_never_ever_use_directly == 0; -} - -/* warning : this function will stop program execution if outcome is invalid ! - * check outcome validity first, using BMK_isValid_runResult() */ -BMK_runTime_t BMK_extract_runTime(BMK_runOutcome_t outcome) -{ - CONTROL(outcome.error_tag_never_ever_use_directly == 0); - return outcome.internal_never_ever_use_directly; -} - -size_t BMK_extract_errorResult(BMK_runOutcome_t outcome) -{ - CONTROL(outcome.error_tag_never_ever_use_directly != 0); - return outcome.error_result_never_ever_use_directly; -} - -static BMK_runOutcome_t BMK_runOutcome_error(size_t errorResult) -{ - BMK_runOutcome_t b; - memset(&b, 0, sizeof(b)); - b.error_tag_never_ever_use_directly = 1; - b.error_result_never_ever_use_directly = errorResult; - return b; -} - -static BMK_runOutcome_t BMK_setValid_runTime(BMK_runTime_t runTime) -{ - BMK_runOutcome_t outcome; - outcome.error_tag_never_ever_use_directly = 0; - outcome.internal_never_ever_use_directly = runTime; - return outcome; -} - - -/* initFn will be measured once, benchFn will be measured `nbLoops` times */ -/* initFn is optional, provide NULL if none */ -/* benchFn must return a size_t value that errorFn can interpret */ -/* takes # of blocks and list of size & stuff for each. */ -/* can report result of benchFn for each block into blockResult. */ -/* blockResult is optional, provide NULL if this information is not required */ -/* note : time per loop can be reported as zero if run time < timer resolution */ -BMK_runOutcome_t BMK_benchFunction(BMK_benchParams_t p, - unsigned nbLoops) -{ - nbLoops += !nbLoops; /* minimum nbLoops is 1 */ - - /* init */ - { size_t i; - for(i = 0; i < p.blockCount; i++) { - memset(p.dstBuffers[i], 0xE5, p.dstCapacities[i]); /* warm up and erase result buffer */ - } } - - /* benchmark */ - { size_t dstSize = 0; - UTIL_time_t const clockStart = UTIL_getTime(); - unsigned loopNb, blockNb; - if (p.initFn != NULL) p.initFn(p.initPayload); - for (loopNb = 0; loopNb < nbLoops; loopNb++) { - for (blockNb = 0; blockNb < p.blockCount; blockNb++) { - size_t const res = p.benchFn(p.srcBuffers[blockNb], p.srcSizes[blockNb], - p.dstBuffers[blockNb], p.dstCapacities[blockNb], - p.benchPayload); - if (loopNb == 0) { - if (p.blockResults != NULL) p.blockResults[blockNb] = res; - if ((p.errorFn != NULL) && (p.errorFn(res))) { - RETURN_QUIET_ERROR(BMK_runOutcome_error(res), - "Function benchmark failed on block %u (of size %u) with error %i", - blockNb, (unsigned)p.srcSizes[blockNb], (int)res); - } - dstSize += res; - } } - } /* for (loopNb = 0; loopNb < nbLoops; loopNb++) */ - - { PTime const totalTime = UTIL_clockSpanNano(clockStart); - BMK_runTime_t rt; - rt.nanoSecPerRun = (double)totalTime / nbLoops; - rt.sumOfReturn = dstSize; - return BMK_setValid_runTime(rt); - } } -} - - -/* ==== Benchmarking any function, providing intermediate results ==== */ - -struct BMK_timedFnState_s { - PTime timeSpent_ns; - PTime timeBudget_ns; - PTime runBudget_ns; - BMK_runTime_t fastestRun; - unsigned nbLoops; - UTIL_time_t coolTime; -}; /* typedef'd to BMK_timedFnState_t within bench.h */ - -BMK_timedFnState_t* BMK_createTimedFnState(unsigned total_ms, unsigned run_ms) -{ - BMK_timedFnState_t* const r = (BMK_timedFnState_t*)malloc(sizeof(*r)); - if (r == NULL) return NULL; /* malloc() error */ - BMK_resetTimedFnState(r, total_ms, run_ms); - return r; -} - -void BMK_freeTimedFnState(BMK_timedFnState_t* state) { free(state); } - -BMK_timedFnState_t* -BMK_initStatic_timedFnState(void* buffer, size_t size, unsigned total_ms, unsigned run_ms) -{ - typedef char check_size[ 2 * (sizeof(BMK_timedFnState_shell) >= sizeof(struct BMK_timedFnState_s)) - 1]; /* static assert : a compilation failure indicates that BMK_timedFnState_shell is not large enough */ - typedef struct { check_size c; BMK_timedFnState_t tfs; } tfs_align; /* force tfs to be aligned at its next best position */ - size_t const tfs_alignment = offsetof(tfs_align, tfs); /* provides the minimal alignment restriction for BMK_timedFnState_t */ - BMK_timedFnState_t* const r = (BMK_timedFnState_t*)buffer; - if (buffer == NULL) return NULL; - if (size < sizeof(struct BMK_timedFnState_s)) return NULL; - if ((size_t)buffer % tfs_alignment) return NULL; /* buffer must be properly aligned */ - BMK_resetTimedFnState(r, total_ms, run_ms); - return r; -} - -void BMK_resetTimedFnState(BMK_timedFnState_t* timedFnState, unsigned total_ms, unsigned run_ms) -{ - if (!total_ms) total_ms = 1 ; - if (!run_ms) run_ms = 1; - if (run_ms > total_ms) run_ms = total_ms; - timedFnState->timeSpent_ns = 0; - timedFnState->timeBudget_ns = (PTime)total_ms * TIMELOOP_NANOSEC / 1000; - timedFnState->runBudget_ns = (PTime)run_ms * TIMELOOP_NANOSEC / 1000; - timedFnState->fastestRun.nanoSecPerRun = (double)TIMELOOP_NANOSEC * 2000000000; /* hopefully large enough : must be larger than any potential measurement */ - timedFnState->fastestRun.sumOfReturn = (size_t)(-1LL); - timedFnState->nbLoops = 1; - timedFnState->coolTime = UTIL_getTime(); -} - -/* Tells if nb of seconds set in timedFnState for all runs is spent. - * note : this function will return 1 if BMK_benchFunctionTimed() has actually errored. */ -int BMK_isCompleted_TimedFn(const BMK_timedFnState_t* timedFnState) -{ - return (timedFnState->timeSpent_ns >= timedFnState->timeBudget_ns); -} - - -#undef MIN -#define MIN(a,b) ( (a) < (b) ? (a) : (b) ) - -#define MINUSABLETIME (TIMELOOP_NANOSEC / 2) /* 0.5 seconds */ - -BMK_runOutcome_t BMK_benchTimedFn(BMK_timedFnState_t* cont, - BMK_benchParams_t p) -{ - PTime const runBudget_ns = cont->runBudget_ns; - PTime const runTimeMin_ns = runBudget_ns / 2; - int completed = 0; - BMK_runTime_t bestRunTime = cont->fastestRun; - - while (!completed) { - BMK_runOutcome_t const runResult = BMK_benchFunction(p, cont->nbLoops); - - if(!BMK_isSuccessful_runOutcome(runResult)) { /* error : move out */ - return runResult; - } - - { BMK_runTime_t const newRunTime = BMK_extract_runTime(runResult); - double const loopDuration_ns = newRunTime.nanoSecPerRun * cont->nbLoops; - - cont->timeSpent_ns += (unsigned long long)loopDuration_ns; - - /* estimate nbLoops for next run to last approximately 1 second */ - if (loopDuration_ns > ((double)runBudget_ns / 50)) { - double const fastestRun_ns = MIN(bestRunTime.nanoSecPerRun, newRunTime.nanoSecPerRun); - cont->nbLoops = (unsigned)((double)runBudget_ns / fastestRun_ns) + 1; - } else { - /* previous run was too short : blindly increase workload by x multiplier */ - const unsigned multiplier = 10; - assert(cont->nbLoops < ((unsigned)-1) / multiplier); /* avoid overflow */ - cont->nbLoops *= multiplier; - } - - if(loopDuration_ns < (double)runTimeMin_ns) { - /* don't report results for which benchmark run time was too small : increased risks of rounding errors */ - assert(completed == 0); - continue; - } else { - if(newRunTime.nanoSecPerRun < bestRunTime.nanoSecPerRun) { - bestRunTime = newRunTime; - } - completed = 1; - } - } - } /* while (!completed) */ - - return BMK_setValid_runTime(bestRunTime); -} +/* BMK_runTime_t and BMK_runOutcome_t are returned by value across the C/Rust + * boundary, and BMK_benchParams_t is passed by value. The Rust #[repr(C)] + * definitions mirror the offsets pinned here. */ +typedef char BMK_staticAssert_runTimeSumOffset[ + (offsetof(BMK_runTime_t, sumOfReturn) == sizeof(double)) ? 1 : -1]; +typedef char BMK_staticAssert_outcomeResultOffset[ + (offsetof(BMK_runOutcome_t, error_result_never_ever_use_directly) + == sizeof(BMK_runTime_t)) ? 1 : -1]; +typedef char BMK_staticAssert_outcomeTagOffset[ + (offsetof(BMK_runOutcome_t, error_tag_never_ever_use_directly) + == sizeof(BMK_runTime_t) + sizeof(size_t)) ? 1 : -1]; +typedef char BMK_staticAssert_shellAlignment[ + (sizeof(BMK_timedFnState_shell) == BMK_TIMEDFNSTATE_SIZE) ? 1 : -1]; diff --git a/programs/zstdcli.c b/programs/zstdcli.c index 97113d884..65a0e2fac 100644 --- a/programs/zstdcli.c +++ b/programs/zstdcli.c @@ -10,16 +10,67 @@ /* The CLI parser and control flow live in rust/src/zstd_cli.rs. Keep this * translation unit as the stable C entry point used by program launchers. */ +#include /* size_t */ +#define ZSTD_STATIC_LINKING_ONLY /* ZSTD_compressionParameters */ #include "../lib/zstd.h" +#ifndef ZSTD_NOBENCH +# include "benchzstd.h" /* BMK_benchFilesAdvanced, BMK_syntheticTest */ +#endif int ZSTD_rust_cli_main(int argCount, const char* const argv[]); const char* ZSTD_rust_cli_expected_version(void); +int ZSTD_rust_cli_bench(const char* const* fileNames, unsigned nbFiles, + const char* dictFileName, + int startCLevel, int endCLevel, + const ZSTD_compressionParameters* compressionParams, + int displayLevel, unsigned nbSeconds, + size_t blockSize, int nbWorkers); const char* ZSTD_rust_cli_expected_version(void) { return ZSTD_VERSION_STRING; } +/* Benchmark bridge for the Rust CLI. Whether benchmarking exists is a C + * preprocessor property (ZSTD_NOBENCH), so the decision stays in this shim: + * the Rust frontend calls in unconditionally, and stripped program variants + * never reference benchmark symbols. + * @return the benchmark result code (>= 0), or -1 when unavailable. */ +int ZSTD_rust_cli_bench(const char* const* fileNames, unsigned nbFiles, + const char* dictFileName, + int startCLevel, int endCLevel, + const ZSTD_compressionParameters* compressionParams, + int displayLevel, unsigned nbSeconds, + size_t blockSize, int nbWorkers) +{ +#ifndef ZSTD_NOBENCH + BMK_advancedParams_t advancedParams = BMK_initAdvancedParams(); + int startLevel = startCLevel; + int endLevel = endCLevel; + advancedParams.nbSeconds = nbSeconds; + advancedParams.blockSize = blockSize; + advancedParams.nbWorkers = nbWorkers; + if (startLevel > ZSTD_maxCLevel()) startLevel = ZSTD_maxCLevel(); + if (endLevel > ZSTD_maxCLevel()) endLevel = ZSTD_maxCLevel(); + if (endLevel < startLevel) endLevel = startLevel; + if (nbFiles == 0) { + /* No input file: benchmark a synthetic sample (lorem generator). */ + return BMK_syntheticTest(-1.0, startLevel, endLevel, + compressionParams, displayLevel, + &advancedParams); + } + return BMK_benchFilesAdvanced(fileNames, nbFiles, dictFileName, + startLevel, endLevel, + compressionParams, displayLevel, + &advancedParams); +#else + (void)fileNames; (void)nbFiles; (void)dictFileName; + (void)startCLevel; (void)endCLevel; (void)compressionParams; + (void)displayLevel; (void)nbSeconds; (void)blockSize; (void)nbWorkers; + return -1; +#endif +} + int main(int argCount, const char* argv[]) { return ZSTD_rust_cli_main(argCount, argv); diff --git a/rust/README.md b/rust/README.md index ff4502bcc..07653f653 100644 --- a/rust/README.md +++ b/rust/README.md @@ -69,14 +69,19 @@ zstd ABI: so library builds do not acquire program-only dependencies. The C `fileio` backend still owns file opening, safe replacement, sparse writes, metadata, and streaming I/O. - - `timefn` provides the monotonic nanosecond clock behind `UTIL_time_t`. - It also lives in the `cli/` package, but C test binaries (fullbench, - fuzzer, zstreamtest, paramgrill, ...) link a helpers-only build of that - archive, produced without the package's `cli` feature, because the parser - layer requires the C `fileio` backend that tests do not compile. + - `timefn` provides the monotonic nanosecond clock behind `UTIL_time_t`, + and `benchfn` owns the benchmark run/timing loop (`BMK_benchFunction`, + `BMK_benchTimedFn`) used by the CLI benchmark mode and by C test tools. + Both live in the `cli/` package, but C test binaries (fullbench, fuzzer, + zstreamtest, paramgrill, ...) link a helpers-only build of that archive, + produced without the package's `cli` feature, because the parser layer + requires the C `fileio` backend that tests do not compile. Benchmark + orchestration and reporting (`benchzstd.c`) remain C, reached from the + Rust parser through the `ZSTD_NOBENCH`-gated bridge in `zstdcli.c`. The optimal block matcher, high-level frame compression, dictionary-building -except suffix-array construction, legacy decoding callbacks, and the CLI +except suffix-array construction, legacy decoding callbacks, benchmark +orchestration (`benchzstd`), and the CLI file-I/O backend are still C. They must move before the rewrite is complete. Keeping that boundary explicit prevents a passing hybrid build from being mistaken for the final all-Rust result. diff --git a/rust/cli/src/lib.rs b/rust/cli/src/lib.rs index e33acb81b..3b800f5c7 100644 --- a/rust/cli/src/lib.rs +++ b/rust/cli/src/lib.rs @@ -1,3 +1,5 @@ +#[path = "../../src/benchfn.rs"] +mod benchfn; #[path = "../../src/timefn.rs"] mod timefn; #[cfg(feature = "cli")] diff --git a/rust/src/benchfn.rs b/rust/src/benchfn.rs new file mode 100644 index 000000000..25b2f01aa --- /dev/null +++ b/rust/src/benchfn.rs @@ -0,0 +1,575 @@ +#![allow(non_camel_case_types)] +#![allow(non_snake_case)] +#![allow(clippy::missing_safety_doc)] + +//! Benchmark loop for arbitrary functions over a set of blocks. +//! +//! Port of `programs/benchfn.c`. `BMK_benchFunction` measures one batch of +//! runs; `BMK_benchTimedFn` repeats batches, growing the loop count until a +//! run lasts long enough to be reported reliably against `run_ms`, within a +//! `total_ms` budget tracked by `BMK_timedFnState_t`. +//! +//! ABI notes: `BMK_runOutcome_t` and `BMK_runTime_t` are returned by value +//! across the C boundary and `BMK_benchParams_t` is passed by value, so all +//! three are `repr(C)` mirrors of the `benchfn.h` layout, pinned by asserts +//! here and in the C shim. `BMK_timedFnState_t` is opaque to C, but +//! `BMK_initStatic_timedFnState` guarantees it fits the 64-byte +//! `BMK_timedFnState_shell`, and `BMK_createTimedFnState` uses `malloc` so +//! creation and destruction stay interchangeable with C callers. + +use std::os::raw::{c_int, c_uint, c_void}; +use std::ptr; + +use crate::timefn::{PTime, UTIL_clockSpanNano, UTIL_getTime, UTIL_time_t}; + +const TIMELOOP_NANOSEC: PTime = 1_000_000_000; + +/// Valid benchmark result (`BMK_runTime_t` in benchfn.h). +#[repr(C)] +#[derive(Clone, Copy, Debug)] +pub struct BMK_runTime_t { + /// Time per iteration, over all blocks. + pub nanoSecPerRun: f64, + /// Sum of the benchmarked function's return values, first loop only. + pub sumOfReturn: usize, +} + +/// Outcome variant of a benchmark run (`BMK_runOutcome_t` in benchfn.h): +/// either a valid `BMK_runTime_t` or an error result. C callers treat it as +/// opaque and use the accessor functions below. +#[repr(C)] +#[derive(Clone, Copy, Debug)] +pub struct BMK_runOutcome_t { + pub internal_never_ever_use_directly: BMK_runTime_t, + pub error_result_never_ever_use_directly: usize, + pub error_tag_never_ever_use_directly: c_int, +} + +// These mirror the static asserts in the programs/benchfn.c shim: the structs +// cross the ABI by value, so field offsets must match the C header exactly. +const _: () = + assert!(std::mem::offset_of!(BMK_runTime_t, sumOfReturn) == std::mem::size_of::()); +const _: () = assert!( + std::mem::offset_of!(BMK_runOutcome_t, error_result_never_ever_use_directly) + == std::mem::size_of::() +); +const _: () = assert!( + std::mem::offset_of!(BMK_runOutcome_t, error_tag_never_ever_use_directly) + == std::mem::size_of::() + std::mem::size_of::() +); + +/// `size_t (*BMK_benchFn_t)(const void*, size_t, void*, size_t, void*)` +pub type BMK_benchFn_t = Option< + unsafe extern "C" fn( + src: *const c_void, + srcSize: usize, + dst: *mut c_void, + dstCapacity: usize, + customPayload: *mut c_void, + ) -> usize, +>; +/// `size_t (*BMK_initFn_t)(void*)` +pub type BMK_initFn_t = Option usize>; +/// `unsigned (*BMK_errorFn_t)(size_t)` +pub type BMK_errorFn_t = Option c_uint>; + +/// Parameters of `BMK_benchFunction`, passed by value (`BMK_benchParams_t`). +#[repr(C)] +#[derive(Clone, Copy)] +pub struct BMK_benchParams_t { + pub benchFn: BMK_benchFn_t, + pub benchPayload: *mut c_void, + pub initFn: BMK_initFn_t, + pub initPayload: *mut c_void, + pub errorFn: BMK_errorFn_t, + pub blockCount: usize, + pub srcBuffers: *const *const c_void, + pub srcSizes: *const usize, + pub dstBuffers: *const *mut c_void, + pub dstCapacities: *const usize, + pub blockResults: *mut usize, +} + +/// Aborts, like benchfn.c's `CONTROL`, when an accessor is used on the wrong +/// outcome variant. +fn control(condition: bool) { + if !condition { + std::process::abort(); + } +} + +fn error_outcome(errorResult: usize) -> BMK_runOutcome_t { + BMK_runOutcome_t { + internal_never_ever_use_directly: BMK_runTime_t { + nanoSecPerRun: 0.0, + sumOfReturn: 0, + }, + error_result_never_ever_use_directly: errorResult, + error_tag_never_ever_use_directly: 1, + } +} + +fn valid_outcome(runTime: BMK_runTime_t) -> BMK_runOutcome_t { + BMK_runOutcome_t { + internal_never_ever_use_directly: runTime, + error_result_never_ever_use_directly: 0, + error_tag_never_ever_use_directly: 0, + } +} + +/// Tells if the outcome carries a valid measurement. +#[no_mangle] +pub extern "C" fn BMK_isSuccessful_runOutcome(outcome: BMK_runOutcome_t) -> c_int { + c_int::from(outcome.error_tag_never_ever_use_directly == 0) +} + +/// Extracts the measurement; aborts if the outcome is an error, so validity +/// must be checked first with `BMK_isSuccessful_runOutcome`. +#[no_mangle] +pub extern "C" fn BMK_extract_runTime(outcome: BMK_runOutcome_t) -> BMK_runTime_t { + control(outcome.error_tag_never_ever_use_directly == 0); + outcome.internal_never_ever_use_directly +} + +/// Extracts the faulty `benchFn` return value; aborts if the outcome is +/// valid, so failure must be checked first. +#[no_mangle] +pub extern "C" fn BMK_extract_errorResult(outcome: BMK_runOutcome_t) -> usize { + control(outcome.error_tag_never_ever_use_directly != 0); + outcome.error_result_never_ever_use_directly +} + +/// Runs `initFn` once, then `benchFn` `nbLoops` times over every block, and +/// reports the mean time per loop. On the first loop, per-block results are +/// stored into `blockResults` (when provided) and checked with `errorFn` +/// (when provided); the first failing block aborts the measurement and +/// produces an error outcome carrying the faulty return value. +#[no_mangle] +pub unsafe extern "C" fn BMK_benchFunction( + p: BMK_benchParams_t, + mut nbLoops: c_uint, +) -> BMK_runOutcome_t { + // Minimum nbLoops is 1. + nbLoops += c_uint::from(nbLoops == 0); + + // Warm up and erase the result buffers. + for blockNb in 0..p.blockCount { + unsafe { + let dst = *p.dstBuffers.add(blockNb); + ptr::write_bytes(dst.cast::(), 0xE5, *p.dstCapacities.add(blockNb)); + } + } + + let benchFn = p.benchFn.expect("benchFn is mandatory"); + let mut dstSize = 0usize; + let clockStart = UTIL_getTime(); + if let Some(initFn) = p.initFn { + unsafe { initFn(p.initPayload) }; + } + for loopNb in 0..nbLoops { + for blockNb in 0..p.blockCount { + let res = unsafe { + benchFn( + *p.srcBuffers.add(blockNb), + *p.srcSizes.add(blockNb), + *p.dstBuffers.add(blockNb), + *p.dstCapacities.add(blockNb), + p.benchPayload, + ) + }; + if loopNb == 0 { + if !p.blockResults.is_null() { + unsafe { *p.blockResults.add(blockNb) = res }; + } + if let Some(errorFn) = p.errorFn { + if unsafe { errorFn(res) } != 0 { + return error_outcome(res); + } + } + dstSize = dstSize.wrapping_add(res); + } + } + } + + let totalTime = UTIL_clockSpanNano(clockStart); + valid_outcome(BMK_runTime_t { + nanoSecPerRun: totalTime as f64 / f64::from(nbLoops), + sumOfReturn: dstSize, + }) +} + +/// Benchmark session state (`struct BMK_timedFnState_s`), opaque to C. +#[repr(C)] +pub struct BMK_timedFnState_t { + timeSpent_ns: PTime, + timeBudget_ns: PTime, + runBudget_ns: PTime, + fastestRun: BMK_runTime_t, + nbLoops: c_uint, + coolTime: UTIL_time_t, +} + +/// `BMK_TIMEDFNSTATE_SIZE` in benchfn.h: capacity of the caller-provided +/// `BMK_timedFnState_shell`, which the state must always fit. +const BMK_TIMEDFNSTATE_SIZE: usize = 64; +const _: () = assert!(std::mem::size_of::() <= BMK_TIMEDFNSTATE_SIZE); +// The shell aligns via a `long long` member; the state must not need more. +const _: () = assert!(std::mem::align_of::() <= std::mem::align_of::()); + +/// Allocates and initializes a benchmark session lasting a minimum of +/// `total_ms`, paced at intervals of approximately `run_ms`. Uses `malloc` +/// so ownership stays interchangeable with the original C implementation. +#[no_mangle] +pub extern "C" fn BMK_createTimedFnState( + total_ms: c_uint, + run_ms: c_uint, +) -> *mut BMK_timedFnState_t { + let state = unsafe { libc::malloc(std::mem::size_of::()) } + .cast::(); + if state.is_null() { + return ptr::null_mut(); + } + unsafe { BMK_resetTimedFnState(state, total_ms, run_ms) }; + state +} + +/// Releases a state obtained from `BMK_createTimedFnState`. +#[no_mangle] +pub unsafe extern "C" fn BMK_freeTimedFnState(state: *mut BMK_timedFnState_t) { + unsafe { libc::free(state.cast()) }; +} + +/// Places the session state into a caller-provided buffer, typically a +/// `BMK_timedFnState_shell`. Returns NULL when the buffer is missing, too +/// small, or misaligned. +#[no_mangle] +pub unsafe extern "C" fn BMK_initStatic_timedFnState( + buffer: *mut c_void, + size: usize, + total_ms: c_uint, + run_ms: c_uint, +) -> *mut BMK_timedFnState_t { + if buffer.is_null() { + return ptr::null_mut(); + } + if size < std::mem::size_of::() { + return ptr::null_mut(); + } + if !(buffer as usize).is_multiple_of(std::mem::align_of::()) { + return ptr::null_mut(); + } + let state = buffer.cast::(); + unsafe { BMK_resetTimedFnState(state, total_ms, run_ms) }; + state +} + +/// Re-arms a session for a new benchmark of `total_ms`, paced at `run_ms`. +#[no_mangle] +pub unsafe extern "C" fn BMK_resetTimedFnState( + timedFnState: *mut BMK_timedFnState_t, + total_ms: c_uint, + run_ms: c_uint, +) { + let total_ms = if total_ms == 0 { 1 } else { total_ms }; + let mut run_ms = if run_ms == 0 { 1 } else { run_ms }; + if run_ms > total_ms { + run_ms = total_ms; + } + let state = BMK_timedFnState_t { + timeSpent_ns: 0, + timeBudget_ns: PTime::from(total_ms) * TIMELOOP_NANOSEC / 1000, + runBudget_ns: PTime::from(run_ms) * TIMELOOP_NANOSEC / 1000, + fastestRun: BMK_runTime_t { + // Must be larger than any potential measurement. + nanoSecPerRun: TIMELOOP_NANOSEC as f64 * 2_000_000_000.0, + sumOfReturn: usize::MAX, + }, + nbLoops: 1, + coolTime: UTIL_getTime(), + }; + unsafe { timedFnState.write(state) }; +} + +/// Tells if the total time budget of the session is spent. Also reports 1 +/// after `BMK_benchTimedFn` returned an error. +#[no_mangle] +pub unsafe extern "C" fn BMK_isCompleted_TimedFn(timedFnState: *const BMK_timedFnState_t) -> c_int { + let state = unsafe { &*timedFnState }; + c_int::from(state.timeSpent_ns >= state.timeBudget_ns) +} + +/// Runs one measurement supposed to last about `run_ms`, automatically +/// scaling `nbLoops`. Runs shorter than half the run budget are re-tried +/// with a larger workload instead of being reported, limiting rounding-error +/// risks; the best (fastest) qualifying run is returned. +#[no_mangle] +pub unsafe extern "C" fn BMK_benchTimedFn( + cont: *mut BMK_timedFnState_t, + p: BMK_benchParams_t, +) -> BMK_runOutcome_t { + let cont = unsafe { &mut *cont }; + let runBudget_ns = cont.runBudget_ns; + let runTimeMin_ns = runBudget_ns / 2; + let mut bestRunTime = cont.fastestRun; + + loop { + let runResult = unsafe { BMK_benchFunction(p, cont.nbLoops) }; + + if BMK_isSuccessful_runOutcome(runResult) == 0 { + // Error: move out. + return runResult; + } + + let newRunTime = BMK_extract_runTime(runResult); + let loopDuration_ns = newRunTime.nanoSecPerRun * f64::from(cont.nbLoops); + + cont.timeSpent_ns = cont.timeSpent_ns.wrapping_add(loopDuration_ns as PTime); + + // Estimate nbLoops for the next run to last approximately run_ms. + if loopDuration_ns > runBudget_ns as f64 / 50.0 { + let fastestRun_ns = bestRunTime.nanoSecPerRun.min(newRunTime.nanoSecPerRun); + cont.nbLoops = ((runBudget_ns as f64 / fastestRun_ns) as c_uint).wrapping_add(1); + } else { + // Previous run was too short: blindly increase workload by a + // x10 multiplier. + const MULTIPLIER: c_uint = 10; + debug_assert!(cont.nbLoops < c_uint::MAX / MULTIPLIER); // avoid overflow + cont.nbLoops = cont.nbLoops.wrapping_mul(MULTIPLIER); + } + + if loopDuration_ns < runTimeMin_ns as f64 { + // Don't report results when the run time was too small, which + // increases the risk of rounding errors. + continue; + } + if newRunTime.nanoSecPerRun < bestRunTime.nanoSecPerRun { + bestRunTime = newRunTime; + } + return valid_outcome(bestRunTime); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Test payload observed through `benchPayload`/`initPayload` pointers. + #[derive(Default)] + struct CallLog { + bench_calls: usize, + init_calls: usize, + } + + /// Counts invocations and reports `srcSize`, like a size-preserving codec. + unsafe extern "C" fn counting_bench_fn( + _src: *const c_void, + srcSize: usize, + _dst: *mut c_void, + _dstCapacity: usize, + payload: *mut c_void, + ) -> usize { + let log = unsafe { &mut *payload.cast::() }; + log.bench_calls += 1; + srcSize + } + + unsafe extern "C" fn counting_init_fn(payload: *mut c_void) -> usize { + let log = unsafe { &mut *payload.cast::() }; + log.init_calls += 1; + 0 + } + + /// Flags results of 5 bytes and above as errors. + unsafe extern "C" fn error_on_5(result: usize) -> c_uint { + c_uint::from(result >= 5) + } + + struct Fixture { + srcs: Vec>, + dsts: Vec>, + src_ptrs: Vec<*const c_void>, + src_sizes: Vec, + dst_ptrs: Vec<*mut c_void>, + dst_capacities: Vec, + block_results: Vec, + log: CallLog, + } + + impl Fixture { + fn new(block_sizes: &[usize]) -> Box { + let srcs: Vec> = block_sizes.iter().map(|size| vec![0u8; *size]).collect(); + let mut dsts: Vec> = block_sizes.iter().map(|size| vec![0u8; *size]).collect(); + let src_ptrs = srcs.iter().map(|src| src.as_ptr().cast()).collect(); + let src_sizes = srcs.iter().map(Vec::len).collect(); + let dst_ptrs = dsts.iter_mut().map(|dst| dst.as_mut_ptr().cast()).collect(); + let dst_capacities = dsts.iter().map(Vec::len).collect(); + let block_results = vec![0usize; block_sizes.len()]; + Box::new(Self { + srcs, + dsts, + src_ptrs, + src_sizes, + dst_ptrs, + dst_capacities, + block_results, + log: CallLog::default(), + }) + } + + fn params(&mut self, errorFn: BMK_errorFn_t) -> BMK_benchParams_t { + let payload: *mut CallLog = &mut self.log; + BMK_benchParams_t { + benchFn: Some(counting_bench_fn), + benchPayload: payload.cast(), + initFn: Some(counting_init_fn), + initPayload: payload.cast(), + errorFn, + blockCount: self.srcs.len(), + srcBuffers: self.src_ptrs.as_ptr(), + srcSizes: self.src_sizes.as_ptr(), + dstBuffers: self.dst_ptrs.as_ptr(), + dstCapacities: self.dst_capacities.as_ptr(), + blockResults: self.block_results.as_mut_ptr(), + } + } + } + + #[test] + fn bench_function_accounts_loops_blocks_and_first_loop_results() { + let mut fixture = Fixture::new(&[3, 8]); + let params = fixture.params(None); + + let outcome = unsafe { BMK_benchFunction(params, 4) }; + + assert_eq!(BMK_isSuccessful_runOutcome(outcome), 1); + let run_time = BMK_extract_runTime(outcome); + // benchFn ran nbLoops times over each block; initFn ran once. + assert_eq!(fixture.log.bench_calls, 4 * 2); + assert_eq!(fixture.log.init_calls, 1); + // sumOfReturn and blockResults reflect the first loop only. + assert_eq!(run_time.sumOfReturn, 3 + 8); + assert_eq!(fixture.block_results, vec![3, 8]); + assert!(run_time.nanoSecPerRun >= 0.0); + } + + #[test] + fn bench_function_treats_zero_loops_as_one_and_warms_up_buffers() { + let mut fixture = Fixture::new(&[4]); + let params = fixture.params(None); + + let outcome = unsafe { BMK_benchFunction(params, 0) }; + + assert_eq!(BMK_isSuccessful_runOutcome(outcome), 1); + assert_eq!(fixture.log.bench_calls, 1); + // The result buffer was erased with the 0xE5 warm-up pattern. + assert_eq!(fixture.dsts[0], vec![0xE5; 4]); + } + + #[test] + fn bench_function_reports_the_first_failing_block() { + let mut fixture = Fixture::new(&[3, 5, 7]); + let params = fixture.params(Some(error_on_5)); + + let outcome = unsafe { BMK_benchFunction(params, 10) }; + + assert_eq!(BMK_isSuccessful_runOutcome(outcome), 0); + assert_eq!(BMK_extract_errorResult(outcome), 5); + // Execution stopped at the failing block, before the third one. + assert_eq!(fixture.log.bench_calls, 2); + // blockResults were recorded up to and including the failure. + assert_eq!(fixture.block_results[..2], [3, 5]); + } + + #[test] + fn reset_clamps_budgets_and_rearms_the_loop_counter() { + let state = BMK_createTimedFnState(0, 7); + assert!(!state.is_null()); + + { + let state = unsafe { &*state }; + // total_ms 0 becomes 1ms, and run_ms is clamped to total_ms. + assert_eq!(state.timeBudget_ns, 1_000_000); + assert_eq!(state.runBudget_ns, 1_000_000); + assert_eq!(state.nbLoops, 1); + assert_eq!(state.timeSpent_ns, 0); + assert_eq!(state.fastestRun.sumOfReturn, usize::MAX); + } + assert_eq!(unsafe { BMK_isCompleted_TimedFn(state) }, 0); + + unsafe { BMK_resetTimedFnState(state, 2_000, 500) }; + { + let state = unsafe { &*state }; + assert_eq!(state.timeBudget_ns, 2_000_000_000); + assert_eq!(state.runBudget_ns, 500_000_000); + } + + unsafe { BMK_freeTimedFnState(state) }; + } + + #[test] + fn static_state_initialization_validates_its_buffer() { + let mut shell = [0u64; BMK_TIMEDFNSTATE_SIZE / 8]; + let buffer: *mut c_void = shell.as_mut_ptr().cast(); + + // A properly sized and aligned buffer is accepted. + let state = unsafe { BMK_initStatic_timedFnState(buffer, 64, 1_000, 100) }; + assert!(!state.is_null()); + assert_eq!(unsafe { BMK_isCompleted_TimedFn(state) }, 0); + + // NULL, undersized, and misaligned buffers are rejected. + let too_small = std::mem::size_of::() - 1; + unsafe { + assert!(BMK_initStatic_timedFnState(ptr::null_mut(), 64, 1, 1).is_null()); + assert!(BMK_initStatic_timedFnState(buffer, too_small, 1, 1).is_null()); + assert!( + BMK_initStatic_timedFnState(buffer.cast::().add(1).cast(), 63, 1, 1).is_null() + ); + } + } + + #[test] + fn timed_runs_grow_the_workload_and_spend_the_budget() { + let mut fixture = Fixture::new(&[16]); + let params = fixture.params(None); + let state = BMK_createTimedFnState(4, 2); + assert!(!state.is_null()); + + let mut rounds = 0usize; + while unsafe { BMK_isCompleted_TimedFn(state) } == 0 { + let outcome = unsafe { BMK_benchTimedFn(state, params) }; + assert_eq!(BMK_isSuccessful_runOutcome(outcome), 1); + let run_time = BMK_extract_runTime(outcome); + assert_eq!(run_time.sumOfReturn, 16); + rounds += 1; + assert!(rounds < 1_000, "the time budget must eventually be spent"); + } + + { + let state = unsafe { &*state }; + // A reported run had to last at least runBudget/2, which is only + // reachable for this trivial function with a grown loop counter. + assert!(state.nbLoops > 1); + assert!(state.timeSpent_ns >= state.timeBudget_ns); + } + // Every reported outcome came from a run of >= runBudget/2, and the + // budget accounting matches BMK_isCompleted_TimedFn. + assert!(rounds >= 1); + assert!(fixture.log.bench_calls >= rounds); + + unsafe { BMK_freeTimedFnState(state) }; + } + + #[test] + fn timed_runs_propagate_errors_without_aborting() { + let mut fixture = Fixture::new(&[9]); + let params = fixture.params(Some(error_on_5)); + let state = BMK_createTimedFnState(1_000, 100); + + let outcome = unsafe { BMK_benchTimedFn(state, params) }; + + assert_eq!(BMK_isSuccessful_runOutcome(outcome), 0); + assert_eq!(BMK_extract_errorResult(outcome), 9); + + unsafe { BMK_freeTimedFnState(state) }; + } +} diff --git a/rust/src/zstd_cli.rs b/rust/src/zstd_cli.rs index 3dbe1e34b..8504051ac 100644 --- a/rust/src/zstd_cli.rs +++ b/rust/src/zstd_cli.rs @@ -10,9 +10,15 @@ //! writes, dictionary loading, streaming, and metadata preservation remain in //! `programs/fileio.c` for this first migration step. //! +//! Benchmark mode (`-b`) parses here and dispatches through the +//! `ZSTD_rust_cli_bench` bridge in `programs/zstdcli.c`: the run/timing loop +//! (benchfn, timefn) is Rust, while orchestration and result formatting +//! (`benchzstd.c`) remain C behind the preprocessor-gated bridge, so builds +//! with `ZSTD_NOBENCH` never reference benchmark symbols. +//! //! Remaining C-only CLI boundaries are called out in `unsupported()` below: -//! benchmark execution, dictionary training, recursive/file-list expansion, -//! tracing, alternate-format selection, and the advanced directory modes. +//! dictionary training, recursive/file-list expansion, tracing, +//! alternate-format selection, and the advanced directory modes. use std::env; use std::ffi::{CStr, CString, OsStr, OsString}; @@ -30,6 +36,7 @@ use std::os::unix::fs::FileTypeExt; const DEFAULT_CLEVEL: i32 = 3; #[cfg(feature = "compression")] const DEFAULT_MAX_CLEVEL: i32 = 19; +const DEFAULT_BENCH_NB_SECONDS: u32 = 3; const DEFAULT_MEM_LIMIT: u32 = 1 << 27; const DEFAULT_LONG_WINDOW_LOG: u32 = 27; const MAX_FAST_ACCELERATION: i32 = 128 << 10; @@ -173,6 +180,22 @@ unsafe extern "C" { output: *const c_char, dict: *const c_char, ) -> c_int; + + /// Benchmark bridge implemented by the `programs/zstdcli.c` shim, which + /// owns the `ZSTD_NOBENCH` preprocessor decision. Returns the benchmark + /// result (>= 0), or -1 when benchmarking is compiled out. + fn ZSTD_rust_cli_bench( + file_names: *const *const c_char, + nb_files: c_uint, + dict_file_name: *const c_char, + start_level: c_int, + end_level: c_int, + compression_params: *const ZSTD_compressionParameters, + display_level: c_int, + nb_seconds: c_uint, + block_size: usize, + nb_workers: c_int, + ) -> c_int; } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -180,6 +203,7 @@ enum Operation { Compress, Decompress, Test, + Bench, } #[derive(Debug)] @@ -230,6 +254,8 @@ struct Cli { row_match_finder: i32, exclude_compressed: bool, compression_params: ZSTD_compressionParameters, + bench_end_level: Option, + bench_nb_seconds: Option, unsupported_program: Option, } @@ -275,6 +301,8 @@ impl Cli { row_match_finder: ZSTD_PS_AUTO, exclude_compressed: false, compression_params: ZSTD_compressionParameters::default(), + bench_end_level: None, + bench_nb_seconds: None, unsupported_program: None, }; @@ -404,9 +432,22 @@ fn usage(advanced: bool) { out, " --adapt[=min=#,max=#], --rsyncable, --[no-]row-match-finder" ); + let _ = writeln!(out, "\nBenchmark options:"); let _ = writeln!( out, - "\nNot yet migrated: benchmark, dictionary training, recursive/file-list expansion," + " -b# Benchmark file(s) at compression level #" + ); + let _ = writeln!( + out, + " -e# Test all levels from -b# up to # included" + ); + let _ = writeln!( + out, + " -i# Set the minimum evaluation time to # seconds" + ); + let _ = writeln!( + out, + "\nNot yet migrated: dictionary training, recursive/file-list expansion," ); let _ = writeln!(out, "trace, alternate formats, and output-directory modes."); } @@ -905,6 +946,33 @@ fn parse_short_options( 'd' => cli.operation = Operation::Decompress, 'z' => cli.operation = Operation::Compress, 't' => cli.operation = Operation::Test, + 'b' => cli.operation = Operation::Bench, + 'e' | 'i' => { + // Benchmark range end (-e#) and duration (-i#): like the C + // parser, digits attach directly and default to 0. + let mut digits_end = offset + 1; + while digits_end < bytes.len() && bytes[digits_end].is_ascii_digit() { + digits_end += 1; + } + let digits = &value[offset + 1..digits_end]; + if option == 'e' { + cli.bench_end_level = Some(if digits.is_empty() { + 0 + } else { + parse_i32(digits, "benchmark end level")? + }); + } else { + cli.bench_nb_seconds = Some(if digits.is_empty() { + 0 + } else { + digits + .parse::() + .map_err(|_| format!("invalid benchmark duration: {digits:?}"))? + }); + } + offset = digits_end; + continue; + } 'c' => { cli.output = Some(cstring(STDOUT_MARK)?); cli.force_stdout = true; @@ -940,7 +1008,7 @@ fn parse_short_options( } break; } - 'b' | 'e' | 'i' | 'l' | 'p' | 'P' | 'r' | 's' | 'S' => { + 'l' | 'p' | 'P' | 'r' | 's' | 'S' => { unsupported(&format!("-{option}"))?; } _ => return Err(format!("unknown option -{option}")), @@ -1231,16 +1299,52 @@ unsafe fn run_decompress( dictionary, ) }, - Operation::Compress => unreachable!("compression is dispatched separately"), + Operation::Compress | Operation::Bench => { + unreachable!("compression and benchmark are dispatched separately") + } } } +/// Runs benchmark mode through the C bridge. Level clamping against +/// `ZSTD_maxCLevel()` happens on the C side, where the symbol is always +/// available when benchmarking is compiled in. No input file means a +/// synthetic-sample benchmark, matching the C CLI. +fn run_bench(cli: &Cli) -> Result { + let inputs: Vec<*const c_char> = cli.inputs.iter().map(|value| value.as_ptr()).collect(); + let dictionary = cli + .dictionary + .as_ref() + .map_or(ptr::null(), |value| value.as_ptr()); + let result = unsafe { + ZSTD_rust_cli_bench( + inputs.as_ptr(), + inputs.len() as c_uint, + dictionary, + cli.level, + cli.bench_end_level.unwrap_or(cli.level), + &cli.compression_params, + cli.display_level, + cli.bench_nb_seconds.unwrap_or(DEFAULT_BENCH_NB_SECONDS), + cli.block_size.unwrap_or(0), + // The C CLI benchmarks single-threaded unless -T was given. + cli.workers.unwrap_or(1), + ) + }; + if result < 0 { + return Err("benchmark mode is not available in this build".to_owned()); + } + Ok(result) +} + fn run_cli(mut cli: Cli) -> Result { if let Some(program_name) = &cli.unsupported_program { return Err(format!( "{program_name} compatibility mode is not yet implemented by the Rust CLI frontend" )); } + if cli.operation == Operation::Bench { + return run_bench(&cli); + } let explicit_input_count = cli.inputs.len(); filter_symlink_inputs(&mut cli); if explicit_input_count > 0 && cli.inputs.is_empty() { @@ -1347,6 +1451,7 @@ fn run_cli(mut cli: Cli) -> Result { #[cfg(not(feature = "decompression"))] unreachable!("unsupported decompression was rejected above") } + Operation::Bench => unreachable!("benchmark mode was dispatched earlier"), } }; @@ -1581,4 +1686,41 @@ mod tests { assert!(error.contains("not yet implemented")); } + + #[test] + fn bench_mode_parses_level_duration_and_defaults() { + let cli = parse(&["zstd", "-b1", "-i0", "input"]); + + assert_eq!(cli.operation, Operation::Bench); + assert_eq!(cli.level, 1); + assert_eq!(cli.bench_nb_seconds, Some(0)); + assert_eq!(cli.bench_end_level, None); + assert_eq!( + cli.inputs + .iter() + .map(|input| input.as_bytes()) + .collect::>(), + vec![&b"input"[..]] + ); + } + + #[test] + fn bench_range_aggregates_within_a_single_argument() { + let cli = parse(&["zstd", "-b5e6i2", "input"]); + + assert_eq!(cli.operation, Operation::Bench); + assert_eq!(cli.level, 5); + assert_eq!(cli.bench_end_level, Some(6)); + assert_eq!(cli.bench_nb_seconds, Some(2)); + } + + #[test] + fn bench_duration_without_digits_defaults_to_zero() { + let cli = parse(&["zstd", "-b", "-e", "-i"]); + + assert_eq!(cli.operation, Operation::Bench); + assert_eq!(cli.level, DEFAULT_CLEVEL); + assert_eq!(cli.bench_end_level, Some(0)); + assert_eq!(cli.bench_nb_seconds, Some(0)); + } } diff --git a/tests/Makefile b/tests/Makefile index 15c1f1cdf..f709b0417 100644 --- a/tests/Makefile +++ b/tests/Makefile @@ -101,7 +101,7 @@ RUST_CLI_DIR := $(RUST_DIR)/cli RUST_CLI_MANIFEST := $(RUST_CLI_DIR)/Cargo.toml RUST_CLI_HELPER_SOURCES := $(RUST_CLI_MANIFEST) $(RUST_CLI_DIR)/Cargo.lock \ $(RUST_CLI_DIR)/src/lib.rs \ - $(RUST_DIR)/src/timefn.rs + $(RUST_DIR)/src/timefn.rs $(RUST_DIR)/src/benchfn.rs RUST_CLI_HELPERS_TARGET_DIR := $(RUST_DIR)/target/cli-helpers RUST_CLI_HELPERS_STATICLIB := $(RUST_CLI_HELPERS_TARGET_DIR)/release/libzstd_cli_rs.a RUST_CLI_HELPERS_STATICLIB_32 := $(RUST_CLI_HELPERS_TARGET_DIR)/$(RUST_TARGET_32)/release/libzstd_cli_rs.a