Compare commits

...

1 Commits

Author SHA1 Message Date
Vito
ae9cb05521 benchmarks 2026-10-01 12:33:37 +02:00
22 changed files with 4177 additions and 0 deletions

View File

@@ -23,6 +23,8 @@ ignoreRegExpList:
- /\b(XRPL|BEAST)_[A-Z_0-9]+_H+/g # include guards
- /::[a-z:_]+/g # things from other namespaces
- /\blib[a-z]+/g # libraries
- /\b(mpd|MPD)_[A-Za-z_0-9]+/g # libmpdec (mpdecimal) API
- /\bbid(64|128)_[a-z_0-9]+/g # Intel decimal library API
- /\b[0-9]{4}-[0-9]{2}-[0-9]{2}[,:][A-Za-zÀ-ÖØ-öø-ÿ.\s]+/g # copyright dates
- /\b[0-9]{4}[,:]?\s*[A-Za-zÀ-ÖØ-öø-ÿ.\s]+/g # copyright years
- /\[[A-Za-z0-9-]+\]\(https:\/\/github.com\/[A-Za-z0-9-]+\)/g # Github usernames
@@ -44,6 +46,7 @@ suggestWords:
- synch->sync
words:
- abempty
- allcr
- AMMID
- AMMMPT
- AMMMPToken
@@ -184,6 +187,7 @@ words:
- Merkle
- misprediction
- missingok
- mpdecimal
- mptbalance
- MPTDEX
- mptflags

View File

@@ -31,6 +31,12 @@ if(tests)
endif()
option(benchmark "Build benchmarks" ON)
# Opt-in: the Number divergence harness (xrpl.bench.number_diff) and the
# mpdecimal benchmark subjects. Requires the `bench_mpdecimal` Conan option.
option(bench_mpdecimal "Build benchmarks that need mpdecimal" OFF)
# Opt-in: Intel Decimal Floating-Point Math Library benchmark subjects. The
# library is downloaded and built at configure/build time (benchmarks only).
option(bench_intel_dfp "Build benchmarks that need Intel's decimal library" OFF)
# When OFF, the crates directory is not added to the build at all: no Rust
# toolchain is required, no cxxbridge bindings are generated, and the C++ tests

View File

@@ -14,6 +14,7 @@
"openssl/3.6.3#f806de8933e3bf6f01016c6a888cee2e%1783945160.863288",
"nudb/2.0.9#11149c73f8f2baff9a0198fe25971fc7%1782392402.297166",
"mpt-crypto/1.0.2#b313cef0c1a493eb970ad185b2e9bab7%1784285108.866483",
"mpdecimal/4.0.0#2fe1a7688d6c8b4bf209ea23e8b0c6ba%1736849696.519",
"lz4/1.10.0#982d9b673900f665a1da109e09c17cab%1782392402.164188",
"libiconv/1.17#9923bc6dc6f106646d6967e0039a5ada%1782392792.775744",
"libbacktrace/cci.20210118#a7691bfccd8caaf66309df196790a5a1%1782392402.420732",

View File

@@ -25,12 +25,15 @@ rm -f conan.lock
conan lock create . \
--options '&:jemalloc=True' \
--options '&:rocksdb=True' \
--options '&:bench_mpdecimal=True' \
--profile:all=conan/lockfile/linux.profile
conan lock create . \
--options '&:jemalloc=True' \
--options '&:rocksdb=True' \
--options '&:bench_mpdecimal=True' \
--profile:all=conan/lockfile/macos.profile
conan lock create . \
--options '&:jemalloc=True' \
--options '&:rocksdb=True' \
--options '&:bench_mpdecimal=True' \
--profile:all=conan/lockfile/windows.profile

View File

@@ -16,6 +16,7 @@ class Xrpl(ConanFile):
options = {
"assertions": [True, False],
"benchmark": [True, False],
"bench_mpdecimal": [True, False],
"coverage": [True, False],
"fPIC": [True, False],
"jemalloc": [True, False],
@@ -51,6 +52,7 @@ class Xrpl(ConanFile):
default_options = {
"assertions": False,
"benchmark": True,
"bench_mpdecimal": False,
"coverage": False,
"fPIC": True,
"jemalloc": False,
@@ -66,6 +68,7 @@ class Xrpl(ConanFile):
"boost/*:without_coroutine2": False,
"date/*:header_only": True,
"ed25519/*:shared": False,
"mpdecimal/*:cxx": False,
"grpc/*:shared": False,
"grpc/*:secure": True,
"grpc/*:codegen": True,
@@ -137,6 +140,8 @@ class Xrpl(ConanFile):
def requirements(self):
if self.options.benchmark:
self.requires("benchmark/1.9.5")
if self.options.bench_mpdecimal:
self.requires("mpdecimal/4.0.0")
self.requires("boost/1.91.0", force=True, transitive_headers=True)
self.requires("date/3.0.4", transitive_headers=True)
if self.options.jemalloc:
@@ -172,6 +177,7 @@ class Xrpl(ConanFile):
tc = CMakeToolchain(self)
tc.variables["tests"] = self.options.tests
tc.variables["benchmark"] = self.options.benchmark
tc.variables["bench_mpdecimal"] = self.options.bench_mpdecimal
tc.variables["assert"] = self.options.assertions
tc.variables["coverage"] = self.options.coverage
tc.variables["jemalloc"] = self.options.jemalloc

View File

@@ -0,0 +1,409 @@
# Number vs. decimal floating-point libraries: benchmark plan
Status: complete (phases 1-5). Goal: before extending `xrpl::Number` to a 128-bit
mantissa, (1) measure how `Number` performs against established base-10
floating-point libraries, and (2) quantify where `Number` diverges from
IEEE 754-2008/2019 decimal arithmetic.
Target precisions for a future 128-bit `Number`: **both 34 digits** (matches
IEEE decimal128) **and 38 digits** (largest `10^k - 1` that fits in a
`uint128`, the direct analogue of today's 19-digit-in-`uint64` design).
## Summary
Measured on one x86-64 machine with GCC 15.2; see "Findings so far" for
method, caveats, and every number.
**Correctness.** Today's `Number` (`Large330`) is correctly rounded for
`+ - * /` and conversion to int64 in all four rounding modes: 9.2 million of
9.2 million samples, including the int64-cap cases older scales got wrong.
Its `root2` (up to ~5 ULP) and `power` (up to ~2800 ULP for n = 360) are not
correctly rounded. No IEEE library tested has an accurate integer `power`
either, and loan payments lose only ~4 digits to it at every precision.
Boost.Decimal 1.91 has several rounding defects (see "Validating the
oracle"); Intel's library had none in `+ - * /`, `sqrt`, or int64
conversion.
**Divergence from IEEE 754.** `Number` is always normalized (no cohorts),
has no infinities, NaNs, signed zero, subnormals, or status flags, throws on
overflow and division by zero, flushes underflow to zero, has a wider
exponent range, and only ~18.96 digits of uniform precision because the
external mantissa is capped at int64. See "How Number diverges from IEEE 754
decimal".
**Speed.** Today's `Number` is 6-25x slower than 64-bit IEEE types at `+`
and `*`, and slower than Intel's 34-digit BID128 at everything except
comparison: 87 vs 12 ns for `+`, 204 vs 38 ns for `*`, parity for `/`. Most
of the cost is normalization one digit at a time (52% of instructions in
`Guard::doDropDigit<unsigned __int128>`) plus a heap-allocated error string
on every `+` and `*`, not the 64-bit width.
![Cost of one operation](images/number/number-core-ops.svg)
**Precision pays off in ledger formulas.** Each formula loses about the same
number of digits at any precision, so a 34-digit type keeps ~15 more correct
digits than today's `Number`, and 38 digits ~4 more again. The AMM swap-in is
the clearest case: 6.5 correct digits in the worst case today, against 22
(34 digits) or 26 (38 digits). 16-digit types are ruled out: they get 60-65%
of vault share conversions wrong.
![AMM swap-in: accuracy and cost](images/number/number-amm-swap-in.svg)
![360-payment amortization: accuracy and cost](images/number/number-loan-amortize360.svg)
**Implications for a 128-bit `Number`.**
- Speed is not a reason to stay at 64 bits. Intel BID128 runs every ledger
formula 2-3x faster than today's `Number` while keeping ~15 more digits.
Removing several digits per normalization step (one divide by 10^k, with
the remainder feeding the guard) and passing error locations as
`char const*` instead of `std::string` are likely the largest gains, at
any width.
- 34 vs 38 digits: mpdecimal runs both at the same speed, so the choice is
~4 extra digits against matching IEEE decimal128 exactly (and being able
to use Intel's library as a drop-in reference for testing).
- `power` needs a better algorithm (wider intermediates or a correctly
rounded method) regardless of width, if Lending needs it accurate to the
last digit.
- A prototype can be measured with no changes here: add
`include/xrpl/basics/Number128.h` (see "Measuring a Number128 prototype").
## Library shortlist
| Library | Types | Role | Import |
| -------------------------------------------------------- | ---------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
| Boost.Decimal (Boost >= 1.91) | `decimal64_t`, `decimal128_t`, `decimal_fast64_t`, `decimal_fast128_t` | Primary competitor; header-only C++ with operator overloads. The `fast` types are always normalized (no cohorts), the same design as `Number`. | Already a dependency (`boost/1.91.0`). |
| Intel Decimal Floating-Point Math Library (libbid 2.0u4) | BID64, BID128 | IEEE reference and expected performance ceiling; backs GCC's `_Decimal*` on x86. Per-call rounding mode and status flags. | Fetched from netlib and built at build time (`-Dbench_intel_dfp=ON`); a Conan recipe if it is adopted beyond benchmarking. |
| mpdecimal (libmpdec 4.0) | arbitrary precision (configured to 19 / 34 / 38 digits) | Correctly-rounded oracle for the divergence harness; previews 34- and 38-digit results. | ConanCenter `mpdecimal/4.0.0`. |
| IBM decNumber (optional) | `decDouble`, `decQuad` | Second independent IEEE implementation, only if Boost and Intel disagree. | Vendored. |
Excluded: GCC `_Decimal*` built-ins (Intel's library underneath, not portable
to Clang), Rust `rust_decimal` (96-bit, non-IEEE, FFI distorts timings),
fixed-point libraries.
Baselines: `double`; `Number` under each `MantissaScale` (`Small`,
`LargeLegacy`, `Large330`).
## How Number diverges from IEEE 754 decimal
Items marked _(to verify)_ are hypotheses from reading the code; the
divergence harness below will confirm or refute them.
| Aspect | IEEE 754 decimal | `Number` |
| ----------------------------- | --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| Precision | 16 digits (decimal64), 34 (decimal128) | Internal mantissa `[10^18, 10^19-1]`, but the external `mantissa()` is capped at `int64` (`kMaxRep`); mantissas above it are stored with a forced trailing zero, so precision is non-uniform (~18.96 digits). |
| Cohorts / quantum | Preserved (`1.0` and `1.00` are distinct representations) | Always normalized; one representation per value; `==` compares fields. |
| Exponent range | ±384 (decimal64), ±6144 (decimal128) | ±32768 (`int` exponent). |
| Underflow | Gradual (subnormals) | Flush to zero (`doNormalize`). |
| Overflow | ±∞ plus a status flag | Throws `std::overflow_error`. |
| Division by zero | ±∞ / NaN | Throws. |
| NaN / ∞ | Yes | None; comparisons are a plain total order. |
| Signed zero | Yes | `-0` collapses to `0`. |
| Rounding modes | 5 (incl. ties-away) | 4; thread-local global mode, similar to `fenv`. (Boost.Decimal's mode is a plain process-wide global, not thread-local.) |
| Status flags | Inexact, underflow, overflow, … | None. |
| Correct rounding of + − × ÷ √ | Required | `×` uses a 128-bit product with guard digits (likely correct). `÷` uses a fixed 10^17 scale plus a staged correction; the legacy `Small` path may not be correctly rounded _(to verify)_. `root`/`root2` (Newton iteration) and `power` (repeated squaring) are not guaranteed correctly rounded. |
| Amendment-gated variants | n/a | `Small`, `LargeLegacy` (keeps a known cusp-rounding error), `Large320`, `Large330`; all must stay bit-identical. |
| Representable range | Symmetric | −2^63 is not exactly representable. |
| Encoding | 8 / 16 bytes (BID or DPD) | 24 bytes in memory (`bool`, `uint64`, `int`, with padding); 12 bytes serialized as `STNumber` (`int64` + `int32`). |
| Other operations | fma, quantize, sameQuantum, … | None. |
## Integration
Nothing in `libxrpl` changes; all code lives under
`src/benchmarks/libxrpl/number/` and builds only when `benchmark=ON`.
- `xrpl_add_benchmark(number)` in `src/benchmarks/libxrpl/CMakeLists.txt`.
- `number/NumberLike.h`: a concept capturing the subset of `Number`'s
interface used by ledger code: construction from `(int64 mantissa, int
exponent)` and from `int64`; `+ - * /`; `< ==`; explicit conversion to
`int64`; `to_string`; `abs`; `power(x, n)`; `root2`; rounding-mode control
mapped onto `Number::RoundingMode`; and `decompose()`, which returns
`(sign, coefficient, exponent)` for digit-exact comparison.
- `number/adapters/*.h`: one thin wrapper per library, holding a single
backend value, all functions inline. `Number` satisfies the concept
directly.
- Optional backends are gated behind options that are off by default, so the
default benchmark build gains no dependencies. mpdecimal: Conan option
`bench_mpdecimal` (which also sets the CMake option of the same name).
Intel: CMake option `bench_intel_dfp`; the tarball is downloaded (SHA-256
pinned) and built with its own makefile via `ExternalProject`, with
`CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0` and `-O3`.
## Benchmarks (each templated on the NumberLike type)
1. Single operations: add (equal exponents; exponents differing by 1, 5, and
18), subtract with heavy cancellation, multiply, divide, compare,
construct from `int64` / `(mantissa, exponent)`, convert to `int64`,
`to_string`, `power`, `root2`.
2. Operand sets, generated up front into arrays (never compile-time
constants; Boost.Decimal is `constexpr`, so constants would be folded
away): random full-width coefficients; IOU-like values (16 digits); XRP
drop amounts; MPT integers up to 2^63−1; wide exponent spread.
3. Throughput (independent operations) and latency (each result feeds the
next) variants.
4. Realistic formula kernels, as templated copies: AMM swap in/out
(`AMMHelpers`), Vault share/asset conversion, Lending interest and
amortization (`power`, `root`). Report speed and the size of
end-result differences.
5. Measurement discipline: Release, `-O3`, pinned core, fixed CPU frequency,
`--benchmark_repetitions=10`, report medians and variation, compare runs
with Google Benchmark's `compare.py`. Report the cost of `Number`'s
thread-local loads (`mode`, `kRange`) separately.
## Divergence harness
A separate executable, `xrpl.bench.number_diff`, runs millions of random and
edge-case operations against mpdecimal configured to match `Number`'s
precision and exponent range. It tallies results that are exact, off by
1 ULP, or worse, per operation, mantissa scale, and rounding mode. It also
checks Boost.Decimal and Intel against mpdecimal to validate the oracle, and
becomes the regression oracle for a 128-bit `Number` at 34 and 38 digits.
## Findings so far
Divergence results are from `xrpl.bench.number_diff --samples 100000` (all
four rounding modes, every mantissa scale). "Correctly rounded" means identical to the exact result
rounded onto the type's own grid of representable values.
### Validating the oracle
The harness was checked against IEEE implementations, and every disagreement
was cross-checked against both mpdecimal at the native precision and Python's
`decimal` module. The oracle agreed with both in every case; the
disagreements are Boost.Decimal 1.91 defects:
- `llrint` returns wrong values for inputs that are already integers at full
precision (e.g. `1.23e17` gives `12345678901235`), and divides by zero
(SIGFPE) when the significand exponent is exactly 0. Fixed upstream; the
adapter works around it.
- Directed rounding is ignored when an add/sub operand is far below the
other's last digit: `6111703207134976 - 5.857e-10` toward zero returns
`...976` instead of `...975` (`decimal64_t` and `decimal_fast64_t`, about
10% of IOU-dataset subtractions in `TowardsZero`).
- `decimal128_t` division misrounds near ties (1 in 2000 full-width
quotients).
- `sqrt` is not correctly rounded (errors up to ~4.5 ULP), although IEEE
requires it.
- The `fast` types are not correctly rounded for `*` and `/` (up to ~1 ULP),
as documented.
Intel's library is correctly rounded for `+ - * /`, `sqrt`, and conversion
to int64 in all four modes (9.2 million of 9.2 million samples for both
BID64 and BID128), including the cases Boost gets wrong. Its `pow` is not
correctly rounded (allowed by IEEE): up to ~1500 ULP for x^360 with x in
[1, 1.01). Example confirmed with Python's `decimal`: 1.002740590413819539^360
is `2.678516411804073475375264650425851`, Intel BID128 returns
`...650426258`. So no IEEE library here provides an accurate integer power;
Number's `power` errors are of the same order as Intel's.
### Number
- **`Large330` (current):** `+ - * /` and conversion to int64 are correctly
rounded in all four modes on every dataset, including the exponent-gap,
cancellation, tie, and int64-cap ("cusp") cases: 9.2 million of 9.2
million samples.
- **Older scales:** the harness reproduces what each amendment fixed.
`LargeLegacy` misrounds at the int64 cap in `Upward` (the documented
`fixCleanup3_2_0` defect); `LargeLegacy` and `Large320` misround at the cap
in the other modes and misround subtraction in `TowardsZero` (fixed in
`Large330`); `Small` misrounds 3-5% of divisions.
- **`root2`** is not correctly rounded in any scale: errors up to ~2.7 ULP
to nearest and ~5 ULP in directed modes.
- **`power`** accumulates error through repeated squaring without extra
precision: ~5 ULP for n = 12 and ~800-2800 ULP for n = 360. Relevant to
Lending amortization, and a design input for a 128-bit Number: wider
intermediates would remove most of it.
### Performance (phase 1)
Median ns per operation, throughput (independent operations), from
`xrpl.bench.number --benchmark_repetitions=3`. GCC 15.2, Release,
single pinned core of a 16-core x86-64 machine. CPU frequency scaling was on
(`powersave`), so treat differences under ~10% as noise; the median
coefficient of variation across benchmarks was 1.1%.
N = Number, B = `BoostDecimal`, BF = `BoostDecimalFast`, Mpd = mpdecimal at
that precision.
| op | data | N.Small | N.LargeLegacy | N.Large330 | B64 | BF64 | B128 | BF128 | Mpd19 | Mpd34 | Mpd38 | double |
| --------- | ---- | ------: | ------------: | ---------: | ---: | ---: | ---: | ----: | ----: | ----: | ----: | -----: |
| add | full | 105 | 105 | 109 | 17.6 | 21.9 | 118 | 118 | 42.1 | 26.1 | 25.7 | 0.6 |
| sub | full | 128 | 125 | 110 | 17.7 | 21.3 | 113 | 130 | 40.1 | 26.4 | 26.3 | 0.6 |
| mul | full | 192 | 210 | 248 | 9.2 | 18.0 | 77.4 | 139 | 30.3 | 31.9 | 15.7 | 0.6 |
| div | full | 40.8 | 66.2 | 70.2 | 38.6 | 44.3 | 336 | 136 | 55.4 | 54.3 | 57.0 | 0.9 |
| lt | full | 1.2 | 1.4 | 1.4 | 3.8 | 2.1 | 8.6 | 2.1 | 4.1 | 4.3 | 4.4 | 0.6 |
| construct | full | 24.3 | 19.1 | 18.4 | 4.7 | 4.8 | 1.0 | 6.5 | 12.7 | 12.7 | 12.7 | 11.0 |
| to_int64 | mpt | 26.4 | 30.1 | 34.8 | 11.7 | 4.4 | 86.4 | 56.7 | 9.4 | 9.5 | 9.5 | 1.5 |
| root2 | full | 1575 | 1734 | 2077 | 120 | 159 | 856 | 906 | 929 | 1212 | 1492 | 1.3 |
| add_gap0 | full | 48.5 | 52.1 | 34.7 | 14.4 | 20.4 | 63.7 | 56.0 | 24.3 | 18.3 | 18.3 | 0.6 |
| add_gap18 | full | 207 | 204 | 204 | 18.6 | 21.8 | 188 | 162 | 43.6 | 44.8 | 29.1 | 0.6 |
| power360 | rate | 1973 | 2503 | 2229 | 221 | 278 | 1421 | 1567 | 788 | 1259 | 1246 | 2.9 |
| mul_chain | rate | 180 | 204 | 284 | 13.1 | 25.0 | 135 | 127 | 32.7 | 36.0 | 36.7 | 0.8 |
| div_chain | rate | 47.2 | 69.9 | 103 | 42.3 | 38.3 | 350 | 130 | 54.4 | 58.0 | 60.7 | 2.7 |
Observations:
- Number's `+` and `*` are 6-25x slower than Boost `decimal64_t`, and 4-15x
slower than mpdecimal at 38 digits, even though mpdecimal is arbitrary
precision and heap-capable. Comparison is Number's only clear win.
- Under callgrind, 52% of the instructions in Number's multiply-and-add loop
are in `Guard::doDropDigit<unsigned __int128>`, which removes surplus digits
one at a time with a 128-bit divide each. `add_gap0` vs `add_gap18` (35 vs
204 ns) shows the same per-digit cost in addition. About another 10% is
`malloc`/`free`: `doRoundUp`/`doRound` take the error-location message as a
`std::string` by value, so every `*` builds one with `std::to_string`, and
every `+` builds `"Number::addition overflow"`, which is too long for the
small-string optimization.
- So Number's cost today is digit-at-a-time normalization, not mantissa
width. A 128-bit Number does not need to be slower than today's if it
removes several digits per step (divide by 10^k with a remainder for the
guard). mpdecimal at 38 digits (add 26 ns, mul 16 ns, div 57 ns) is a
realistic target.
- Boost `decimal128_t` is slower than mpdecimal at 34 digits for `+` and `/`
(`/` 336 vs 54 ns), so Boost.Decimal is not a performance reference at 128
bits. Intel BID128 is (next section).
- `mpdecimal` at 34 and 38 digits costs about the same; 38 is often faster
because 19-digit x 19-digit products fit without rounding. Precision alone
does not separate the 34- and 38-digit options on speed.
### Performance with Intel (phase 3)
A separate run (same build, same conditions) of the Intel types with three
anchor subjects repeated from the table above. The anchors moved by less than
5%, except Number's `add`, which was ~15% faster in this run; read the two
tables as consistent to within about that margin.
| op | data | N.Large330 | B64 | Mpd38 | Intel64 | Intel128 |
| --------- | ---- | ---------: | ---: | ----: | ------: | -------: |
| add | full | 89.8 | 16.3 | 25.5 | 12.3 | 11.8 |
| sub | full | 94.2 | 16.8 | 25.9 | 12.3 | 13.0 |
| mul | full | 206 | 9.1 | 15.7 | 18.3 | 37.6 |
| div | full | 65.5 | 36.0 | 57.7 | 19.7 | 64.1 |
| lt | full | 1.4 | 3.6 | 4.4 | 4.2 | 6.6 |
| construct | full | 16.6 | 4.2 | 12.6 | 11.1 | 4.0 |
| to_int64 | mpt | 27.8 | 9.4 | 9.6 | 4.1 | 4.8 |
| root2 | full | 1819 | 114 | 1491 | 14.8 | 108 |
| add_gap18 | full | 193 | 17.9 | 29.0 | 8.3 | 24.2 |
| power360 | rate | 2196 | 193 | 1275 | 259 | 1149 |
| mul_chain | rate | 197 | 13.1 | 38.9 | 19.2 | 49.7 |
| div_chain | rate | 70.0 | 41.8 | 60.8 | 28.2 | 69.2 |
Observations:
- Intel BID128 (34 digits, correctly rounded) adds in ~12 ns and multiplies
in 25-38 ns: 7-8x faster than today's 64-bit Number for `+` and 5-8x for
`*`, with division at parity. It is the realistic speed target for a
128-bit Number.
- Intel's `sqrt` is two orders of magnitude faster than Number's `root2`
(15 vs ~1800 ns at 16 digits; 108 ns at 34 digits) and correctly rounded.
### Ledger formulas (phase 4)
`number/Kernels.h` holds templated copies of production formulas, run
identically on every type, including the rounding-mode switches:
`swapAssetIn`/`swapAssetOut` (fixAMMv1_1 branch), `assetsToSharesDeposit`
and `sharesToAssetsDeposit`, and `loanPeriodicRate` plus
`loanPeriodicPayment` (fixCleanup3_2_0 branch). `loan_amortize` runs a full
monthly schedule (12 or 360 payments) and measures the leftover balance,
which is exactly 0 in exact arithmetic, relative to the principal. Inputs:
AMM pools of 10^6 to 10^12 with trades of 10^-6 to 10^-2 of the pool and
fees up to 1%; vaults of 10^9 to 10^17 integer assets with share supply up
to 10^18; loans of 10^3 to 10^9 at 0.1% to 30% a year.
ns per call
| kernel | N.Small | N.Large330 | B64 | Intel64 | double | B128 | Intel128 | Mpd34 | Mpd38 |
| ---------------------- | ------: | ---------: | ----: | ------: | -----: | ----: | -------: | ----: | ----: |
| amm_swap_in | 685 | 770 | 148 | 128 | 23 | 646 | 266 | 390 | 404 |
| amm_swap_out | 508 | 598 | 174 | 127 | 27 | 590 | 289 | 385 | 415 |
| vault_assets_to_shares | 237 | 291 | 66 | 45 | 10 | 432 | 95 | 91 | 90 |
| vault_shares_to_assets | 219 | 266 | 48 | 39 | 0.8 | 333 | 79 | 64 | 63 |
| loan_payment12 | 1870 | 2115 | 293 | 351 | 4.3 | 1679 | 1041 | 1158 | 1296 |
| loan_payment360 | 3599 | 3555 | 399 | 487 | 6.7 | 2616 | 1867 | 2454 | 2200 |
| loan_amortize12 | 5108 | 5803 | 1024 | 1055 | 16 | 5385 | 3308 | 3167 | 3091 |
| loan_amortize360 | 98902 | 114119 | 19552 | 18359 | 888 | 96989 | 58486 | 55230 | 45190 |
Correct significant digits against a 96-digit evaluation of the same
formula from the same inputs, worst case (median) over 256 cases; for
integer results, cases that differ from the exact truncated result:
| kernel | N.Small | N.Large330 | B64 | Intel64 | double | B128 | Intel128 | Mpd34 | Mpd38 |
| ---------------------- | ------------: | ----------: | ------------: | ------------: | ------------: | ----------: | ----------: | ----------: | ----------: |
| amm_swap_in | 3.3 (10.9) | 6.5 (13.9) | 3.3 (10.8) | 3.3 (10.8) | 4.0 (11.8) | 21.9 (29.7) | 21.9 (29.8) | 21.9 (29.8) | 26.3 (33.7) |
| amm_swap_out | 8.5 (11.3) | 11.4 (14.1) | 8.5 (11.2) | 8.5 (11.2) | 8.9 (11.4) | 26.9 (29.9) | 26.8 (30.0) | 26.8 (30.0) | 30.5 (33.9) |
| vault_assets_to_shares | 165/256 wrong | 0/256 wrong | 165/256 wrong | 165/256 wrong | 153/256 wrong | 0/256 wrong | 0/256 wrong | 0/256 wrong | 0/256 wrong |
| vault_shares_to_assets | 15.2 (16.0) | 18.1 (19.0) | 15.2 (16.0) | 15.2 (16.0) | 15.8 (16.3) | 33.4 (34.3) | 33.4 (34.3) | 33.4 (34.3) | 37.4 (38.2) |
| loan_payment12 | 11.9 (13.7) | 14.8 (16.7) | 11.9 (13.7) | 11.9 (13.7) | 12.4 (14.4) | 29.4 (31.7) | 29.4 (31.7) | 29.7 (31.8) | 33.7 (35.8) |
| loan_payment360 | 11.9 (15.1) | 15.0 (18.1) | 11.8 (15.2) | 11.8 (15.2) | 12.6 (15.7) | 30.0 (33.1) | 30.0 (33.1) | 30.1 (33.1) | 34.1 (37.1) |
| loan_amortize12 | 11.9 (13.7) | 14.8 (16.7) | 11.9 (13.7) | 11.9 (13.7) | 12.4 (14.3) | 29.4 (31.7) | 29.4 (31.7) | 29.7 (31.7) | 33.7 (35.7) |
| loan_amortize360 | 11.2 (12.8) | 14.1 (15.8) | 11.2 (12.8) | 11.2 (12.8) | 11.5 (13.3) | 29.1 (30.8) | 29.1 (30.8) | 29.1 (30.8) | 33.0 (34.8) |
Observations:
- Wider precision pays off directly in the results. For every formula the
34-digit types keep about 15 more correct digits than today's Number, and
38 digits adds about 4 more. The relative loss is roughly constant: each
formula loses the same number of digits at any precision, so the extra
digits carry through to the result.
- `amm_swap_in` is where it matters most: `poolOut - ratio` cancels for
small trades, leaving 6.5 correct digits in the worst case with today's
Number and about 3 with 16-digit types, against 22 (34 digits) and 26
(38 digits).
- 16-digit types get 60-65% of vault share conversions wrong, because share
counts above 10^16 do not fit. Today's Number and all wider types get every
case right. Any 64-bit IEEE type is ruled out for vaults.
- Intel BID128 runs every formula 2-3x faster than today's Number while
keeping ~15 more digits. A 128-bit Number does not have to trade speed for
precision.
- Loans lose about 4 digits at every precision, even though `power` is off
by up to thousands of ULP: the payment formula is not very sensitive to
that error.
- `Number.Small` and `Number.Large330` cost about the same.
## Phases
1. `Number` and Boost.Decimal adapters, plus the single-operation benchmarks
(no new dependencies). (Done.)
2. The divergence harness and the mpdecimal dependency, at 19, 34, and 38
digits. (Done.)
3. The Intel BID adapter, fetched and built at build time. (Done.)
4. The realistic formula kernels. (Done.)
5. Write-up with figures, and a slot for a `Number128` prototype. (Done.)
## Measuring a Number128 prototype
Add `include/xrpl/basics/Number128.h` defining `xrpl::Number128` with
Number's interface (the `DecomposableNumber` concept in
`src/benchmarks/libxrpl/number/NumberLike.h`) and
`static constexpr unsigned kDigits`. On the next build CMake notices the
header (a `CONFIGURE_DEPENDS` glob) and defines `XRPL_BENCH_NUMBER128`, and
the type appears in every single-operation and formula benchmark and in
`number_diff`, which checks it against a `kDigits`-digit grid. Nothing in
the benchmark sources needs to change. See
`src/benchmarks/libxrpl/number/Number128.h`.
## Running
```bash
# From the build directory. bench_mpdecimal is only needed for number_diff
# and the MpDecimal benchmark subjects; bench_intel_dfp downloads and builds
# Intel's library (needs network access and `make`).
conan install .. --output-folder . --build missing -s build_type=Release \
-o '&:benchmark=True' -o '&:bench_mpdecimal=True'
cmake -G Ninja -DCMAKE_TOOLCHAIN_FILE=build/generators/conan_toolchain.cmake \
-DCMAKE_BUILD_TYPE=Release -Dbench_intel_dfp=ON ..
cmake --build . --target xrpl.bench.number xrpl.bench.number_diff
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_repetitions=10 \
--benchmark_report_aggregates_only=true
./src/benchmarks/libxrpl/xrpl.bench.number_diff --samples 100000 >divergence.md
# Regenerate the figures (standard-library Python only).
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_filter='^(add|mul|div)/full/' \
--benchmark_repetitions=5 --benchmark_report_aggregates_only=true \
--benchmark_out=core.json --benchmark_out_format=json
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_filter='^kernel/' \
--benchmark_repetitions=3 --benchmark_report_aggregates_only=true \
--benchmark_out=kernels.json --benchmark_out_format=json
../src/benchmarks/libxrpl/number/plot_results.py core.json kernels.json \
../docs/images/number
```

View File

@@ -0,0 +1,105 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 276" width="834" height="276" role="img" aria-label="AMM swap-in: correct digits (higher is better) and cost (lower is faster)">
<style>
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
</style>
<rect class="bg" width="834" height="276"/>
<text class="t1" x="0" y="18" font-size="15" font-weight="600">AMM swap-in: correct digits (higher is better) and cost (lower is faster)</text>
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="232.8" y="42">16-digit types</text>
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
<text class="t1" x="0" y="94.0">Number (Large330)</text>
<text class="t1" x="0" y="118.0">Boost decimal64</text>
<text class="t1" x="0" y="142.0">Intel BID64</text>
<text class="t1" x="0" y="166.0">Boost decimal128</text>
<text class="t1" x="0" y="190.0">Intel BID128</text>
<text class="t1" x="0" y="214.0">mpdecimal 34</text>
<text class="t1" x="0" y="238.0">mpdecimal 38</text>
<text class="t1" x="150" y="64" font-weight="600">Correct digits, worst case</text>
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="246"/>
<text class="t2" x="150.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="246"/>
<text class="t2" x="190.0" y="262" text-anchor="middle" font-size="11">10</text>
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="246"/>
<text class="t2" x="230.0" y="262" text-anchor="middle" font-size="11">20</text>
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="246"/>
<text class="t2" x="270.0" y="262" text-anchor="middle" font-size="11">30</text>
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="246"/>
<text class="t2" x="310.0" y="262" text-anchor="middle" font-size="11">40</text>
<text class="t2" x="310.0" y="274" text-anchor="end" font-size="11">significant digits</text>
<path class="f-number" d="M150.0,84.0 H172.0 Q176.0,84.0 176.0,88.0 V92.0 Q176.0,96.0 172.0,96.0 H150.0 Z"><title>Number (Large330): 6.5 significant digits</title></path>
<text class="t2" x="181.0" y="94.0" font-size="11">6.5</text>
<path class="f-ieee64" d="M150.0,108.0 H159.2 Q163.2,108.0 163.2,112.0 V116.0 Q163.2,120.0 159.2,120.0 H150.0 Z"><title>Boost decimal64: 3.3 significant digits</title></path>
<text class="t2" x="168.2" y="118.0" font-size="11">3.3</text>
<path class="f-ieee64" d="M150.0,132.0 H159.2 Q163.2,132.0 163.2,136.0 V140.0 Q163.2,144.0 159.2,144.0 H150.0 Z"><title>Intel BID64: 3.3 significant digits</title></path>
<text class="t2" x="168.2" y="142.0" font-size="11">3.3</text>
<path class="f-wide" d="M150.0,156.0 H233.7 Q237.7,156.0 237.7,160.0 V164.0 Q237.7,168.0 233.7,168.0 H150.0 Z"><title>Boost decimal128: 21.9 significant digits</title></path>
<text class="t2" x="242.7" y="166.0" font-size="11">21.9</text>
<path class="f-wide" d="M150.0,180.0 H233.7 Q237.7,180.0 237.7,184.0 V188.0 Q237.7,192.0 233.7,192.0 H150.0 Z"><title>Intel BID128: 21.9 significant digits</title></path>
<text class="t2" x="242.7" y="190.0" font-size="11">21.9</text>
<path class="f-wide" d="M150.0,204.0 H233.7 Q237.7,204.0 237.7,208.0 V212.0 Q237.7,216.0 233.7,216.0 H150.0 Z"><title>mpdecimal 34: 21.9 significant digits</title></path>
<text class="t2" x="242.7" y="214.0" font-size="11">21.9</text>
<path class="f-wide" d="M150.0,228.0 H251.2 Q255.2,228.0 255.2,232.0 V236.0 Q255.2,240.0 251.2,240.0 H150.0 Z"><title>mpdecimal 38: 26.3 significant digits</title></path>
<text class="t2" x="260.2" y="238.0" font-size="11">26.3</text>
<line class="base" x1="150" y1="78" x2="150" y2="246"/>
<text class="t1" x="378" y="64" font-weight="600">Correct digits, median</text>
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="246"/>
<text class="t2" x="378.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="246"/>
<text class="t2" x="418.0" y="262" text-anchor="middle" font-size="11">10</text>
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="246"/>
<text class="t2" x="458.0" y="262" text-anchor="middle" font-size="11">20</text>
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="246"/>
<text class="t2" x="498.0" y="262" text-anchor="middle" font-size="11">30</text>
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="246"/>
<text class="t2" x="538.0" y="262" text-anchor="middle" font-size="11">40</text>
<text class="t2" x="538.0" y="274" text-anchor="end" font-size="11">significant digits</text>
<path class="f-number" d="M378.0,84.0 H429.6 Q433.6,84.0 433.6,88.0 V92.0 Q433.6,96.0 429.6,96.0 H378.0 Z"><title>Number (Large330): 13.9 significant digits</title></path>
<text class="t2" x="438.6" y="94.0" font-size="11">13.9</text>
<path class="f-ieee64" d="M378.0,108.0 H417.3 Q421.3,108.0 421.3,112.0 V116.0 Q421.3,120.0 417.3,120.0 H378.0 Z"><title>Boost decimal64: 10.8 significant digits</title></path>
<text class="t2" x="426.3" y="118.0" font-size="11">10.8</text>
<path class="f-ieee64" d="M378.0,132.0 H417.3 Q421.3,132.0 421.3,136.0 V140.0 Q421.3,144.0 417.3,144.0 H378.0 Z"><title>Intel BID64: 10.8 significant digits</title></path>
<text class="t2" x="426.3" y="142.0" font-size="11">10.8</text>
<path class="f-wide" d="M378.0,156.0 H492.6 Q496.6,156.0 496.6,160.0 V164.0 Q496.6,168.0 492.6,168.0 H378.0 Z"><title>Boost decimal128: 29.7 significant digits</title></path>
<text class="t2" x="501.6" y="166.0" font-size="11">29.7</text>
<path class="f-wide" d="M378.0,180.0 H493.3 Q497.3,180.0 497.3,184.0 V188.0 Q497.3,192.0 493.3,192.0 H378.0 Z"><title>Intel BID128: 29.8 significant digits</title></path>
<text class="t2" x="502.3" y="190.0" font-size="11">29.8</text>
<path class="f-wide" d="M378.0,204.0 H493.3 Q497.3,204.0 497.3,208.0 V212.0 Q497.3,216.0 493.3,216.0 H378.0 Z"><title>mpdecimal 34: 29.8 significant digits</title></path>
<text class="t2" x="502.3" y="214.0" font-size="11">29.8</text>
<path class="f-wide" d="M378.0,228.0 H508.7 Q512.7,228.0 512.7,232.0 V236.0 Q512.7,240.0 508.7,240.0 H378.0 Z"><title>mpdecimal 38: 33.7 significant digits</title></path>
<text class="t2" x="517.7" y="238.0" font-size="11">33.7</text>
<line class="base" x1="378" y1="78" x2="378" y2="246"/>
<text class="t1" x="606" y="64" font-weight="600">Cost</text>
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="246"/>
<text class="t2" x="606.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="638.0" y1="78" x2="638.0" y2="246"/>
<text class="t2" x="638.0" y="262" text-anchor="middle" font-size="11">200</text>
<line class="grid" x1="670.0" y1="78" x2="670.0" y2="246"/>
<text class="t2" x="670.0" y="262" text-anchor="middle" font-size="11">400</text>
<line class="grid" x1="702.0" y1="78" x2="702.0" y2="246"/>
<text class="t2" x="702.0" y="262" text-anchor="middle" font-size="11">600</text>
<line class="grid" x1="734.0" y1="78" x2="734.0" y2="246"/>
<text class="t2" x="734.0" y="262" text-anchor="middle" font-size="11">800</text>
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="246"/>
<text class="t2" x="766.0" y="262" text-anchor="middle" font-size="11">1000</text>
<text class="t2" x="766.0" y="274" text-anchor="end" font-size="11">ns per call</text>
<path class="f-number" d="M606.0,84.0 H725.2 Q729.2,84.0 729.2,88.0 V92.0 Q729.2,96.0 725.2,96.0 H606.0 Z"><title>Number (Large330): 770 ns per call</title></path>
<text class="t2" x="734.2" y="94.0" font-size="11">770</text>
<path class="f-ieee64" d="M606.0,108.0 H625.7 Q629.7,108.0 629.7,112.0 V116.0 Q629.7,120.0 625.7,120.0 H606.0 Z"><title>Boost decimal64: 148 ns per call</title></path>
<text class="t2" x="634.7" y="118.0" font-size="11">148</text>
<path class="f-ieee64" d="M606.0,132.0 H622.4 Q626.4,132.0 626.4,136.0 V140.0 Q626.4,144.0 622.4,144.0 H606.0 Z"><title>Intel BID64: 128 ns per call</title></path>
<text class="t2" x="631.4" y="142.0" font-size="11">128</text>
<path class="f-wide" d="M606.0,156.0 H705.3 Q709.3,156.0 709.3,160.0 V164.0 Q709.3,168.0 705.3,168.0 H606.0 Z"><title>Boost decimal128: 646 ns per call</title></path>
<text class="t2" x="714.3" y="166.0" font-size="11">646</text>
<path class="f-wide" d="M606.0,180.0 H644.6 Q648.6,180.0 648.6,184.0 V188.0 Q648.6,192.0 644.6,192.0 H606.0 Z"><title>Intel BID128: 266 ns per call</title></path>
<text class="t2" x="653.6" y="190.0" font-size="11">266</text>
<path class="f-wide" d="M606.0,204.0 H664.4 Q668.4,204.0 668.4,208.0 V212.0 Q668.4,216.0 664.4,216.0 H606.0 Z"><title>mpdecimal 34: 390 ns per call</title></path>
<text class="t2" x="673.4" y="214.0" font-size="11">390</text>
<path class="f-wide" d="M606.0,228.0 H666.7 Q670.7,228.0 670.7,232.0 V236.0 Q670.7,240.0 666.7,240.0 H606.0 Z"><title>mpdecimal 38: 404 ns per call</title></path>
<text class="t2" x="675.7" y="238.0" font-size="11">404</text>
<line class="base" x1="606" y1="78" x2="606" y2="246"/>
</svg>

After

Width:  |  Height:  |  Size: 9.2 KiB

View File

@@ -0,0 +1,117 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 324" width="834" height="324" role="img" aria-label="Cost of one operation (full-width operands, lower is faster)">
<style>
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
</style>
<rect class="bg" width="834" height="324"/>
<text class="t1" x="0" y="18" font-size="15" font-weight="600">Cost of one operation (full-width operands, lower is faster)</text>
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="232.8" y="42">16-digit types</text>
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
<text class="t1" x="0" y="94.0">Number (Large330)</text>
<text class="t1" x="0" y="118.0">Boost decimal64</text>
<text class="t1" x="0" y="142.0">Boost decimal_fast64</text>
<text class="t1" x="0" y="166.0">Intel BID64</text>
<text class="t1" x="0" y="190.0">Boost decimal128</text>
<text class="t1" x="0" y="214.0">Boost decimal_fast128</text>
<text class="t1" x="0" y="238.0">Intel BID128</text>
<text class="t1" x="0" y="262.0">mpdecimal 34</text>
<text class="t1" x="0" y="286.0">mpdecimal 38</text>
<text class="t1" x="150" y="64" font-weight="600">Add</text>
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="294"/>
<text class="t2" x="150.0" y="310" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="294"/>
<text class="t2" x="190.0" y="310" text-anchor="middle" font-size="11">100</text>
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="294"/>
<text class="t2" x="230.0" y="310" text-anchor="middle" font-size="11">200</text>
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="294"/>
<text class="t2" x="270.0" y="310" text-anchor="middle" font-size="11">300</text>
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="294"/>
<text class="t2" x="310.0" y="310" text-anchor="middle" font-size="11">400</text>
<text class="t2" x="310.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
<path class="f-number" d="M150.0,84.0 H180.9 Q184.9,84.0 184.9,88.0 V92.0 Q184.9,96.0 180.9,96.0 H150.0 Z"><title>Number (Large330): 87 ns per operation</title></path>
<text class="t2" x="189.9" y="94.0" font-size="11">87</text>
<path class="f-ieee64" d="M150.0,108.0 H152.3 Q156.3,108.0 156.3,112.0 V116.0 Q156.3,120.0 152.3,120.0 H150.0 Z"><title>Boost decimal64: 16 ns per operation</title></path>
<text class="t2" x="161.3" y="118.0" font-size="11">16</text>
<path class="f-ieee64" d="M150.0,132.0 H154.2 Q158.2,132.0 158.2,136.0 V140.0 Q158.2,144.0 154.2,144.0 H150.0 Z"><title>Boost decimal_fast64: 21 ns per operation</title></path>
<text class="t2" x="163.2" y="142.0" font-size="11">21</text>
<path class="f-ieee64" d="M150.0,156.0 H150.6 Q154.6,156.0 154.6,160.0 V164.0 Q154.6,168.0 150.6,168.0 H150.0 Z"><title>Intel BID64: 11 ns per operation</title></path>
<text class="t2" x="159.6" y="166.0" font-size="11">11</text>
<path class="f-wide" d="M150.0,180.0 H190.4 Q194.4,180.0 194.4,184.0 V188.0 Q194.4,192.0 190.4,192.0 H150.0 Z"><title>Boost decimal128: 111 ns per operation</title></path>
<text class="t2" x="199.4" y="190.0" font-size="11">111</text>
<path class="f-wide" d="M150.0,204.0 H187.2 Q191.2,204.0 191.2,208.0 V212.0 Q191.2,216.0 187.2,216.0 H150.0 Z"><title>Boost decimal_fast128: 103 ns per operation</title></path>
<text class="t2" x="196.2" y="214.0" font-size="11">103</text>
<path class="f-wide" d="M150.0,228.0 H150.7 Q154.7,228.0 154.7,232.0 V236.0 Q154.7,240.0 150.7,240.0 H150.0 Z"><title>Intel BID128: 12 ns per operation</title></path>
<text class="t2" x="159.7" y="238.0" font-size="11">12</text>
<path class="f-wide" d="M150.0,252.0 H156.2 Q160.2,252.0 160.2,256.0 V260.0 Q160.2,264.0 156.2,264.0 H150.0 Z"><title>mpdecimal 34: 26 ns per operation</title></path>
<text class="t2" x="165.2" y="262.0" font-size="11">26</text>
<path class="f-wide" d="M150.0,276.0 H156.0 Q160.0,276.0 160.0,280.0 V284.0 Q160.0,288.0 156.0,288.0 H150.0 Z"><title>mpdecimal 38: 25 ns per operation</title></path>
<text class="t2" x="165.0" y="286.0" font-size="11">25</text>
<line class="base" x1="150" y1="78" x2="150" y2="294"/>
<text class="t1" x="378" y="64" font-weight="600">Multiply</text>
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="294"/>
<text class="t2" x="378.0" y="310" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="294"/>
<text class="t2" x="418.0" y="310" text-anchor="middle" font-size="11">100</text>
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="294"/>
<text class="t2" x="458.0" y="310" text-anchor="middle" font-size="11">200</text>
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="294"/>
<text class="t2" x="498.0" y="310" text-anchor="middle" font-size="11">300</text>
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="294"/>
<text class="t2" x="538.0" y="310" text-anchor="middle" font-size="11">400</text>
<text class="t2" x="538.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
<path class="f-number" d="M378.0,84.0 H455.6 Q459.6,84.0 459.6,88.0 V92.0 Q459.6,96.0 455.6,96.0 H378.0 Z"><title>Number (Large330): 204 ns per operation</title></path>
<text class="t2" x="464.6" y="94.0" font-size="11">204</text>
<path class="f-ieee64" d="M378.0,108.0 H378.0 Q381.4,108.0 381.4,111.4 V116.6 Q381.4,120.0 378.0,120.0 H378.0 Z"><title>Boost decimal64: 8.5 ns per operation</title></path>
<text class="t2" x="386.4" y="118.0" font-size="11">8.5</text>
<path class="f-ieee64" d="M378.0,132.0 H380.6 Q384.6,132.0 384.6,136.0 V140.0 Q384.6,144.0 380.6,144.0 H378.0 Z"><title>Boost decimal_fast64: 17 ns per operation</title></path>
<text class="t2" x="389.6" y="142.0" font-size="11">17</text>
<path class="f-ieee64" d="M378.0,156.0 H381.3 Q385.3,156.0 385.3,160.0 V164.0 Q385.3,168.0 381.3,168.0 H378.0 Z"><title>Intel BID64: 18 ns per operation</title></path>
<text class="t2" x="390.3" y="166.0" font-size="11">18</text>
<path class="f-wide" d="M378.0,180.0 H403.1 Q407.1,180.0 407.1,184.0 V188.0 Q407.1,192.0 403.1,192.0 H378.0 Z"><title>Boost decimal128: 73 ns per operation</title></path>
<text class="t2" x="412.1" y="190.0" font-size="11">73</text>
<path class="f-wide" d="M378.0,204.0 H421.7 Q425.7,204.0 425.7,208.0 V212.0 Q425.7,216.0 421.7,216.0 H378.0 Z"><title>Boost decimal_fast128: 119 ns per operation</title></path>
<text class="t2" x="430.7" y="214.0" font-size="11">119</text>
<path class="f-wide" d="M378.0,228.0 H389.3 Q393.3,228.0 393.3,232.0 V236.0 Q393.3,240.0 389.3,240.0 H378.0 Z"><title>Intel BID128: 38 ns per operation</title></path>
<text class="t2" x="398.3" y="238.0" font-size="11">38</text>
<path class="f-wide" d="M378.0,252.0 H386.4 Q390.4,252.0 390.4,256.0 V260.0 Q390.4,264.0 386.4,264.0 H378.0 Z"><title>mpdecimal 34: 31 ns per operation</title></path>
<text class="t2" x="395.4" y="262.0" font-size="11">31</text>
<path class="f-wide" d="M378.0,276.0 H380.1 Q384.1,276.0 384.1,280.0 V284.0 Q384.1,288.0 380.1,288.0 H378.0 Z"><title>mpdecimal 38: 15 ns per operation</title></path>
<text class="t2" x="389.1" y="286.0" font-size="11">15</text>
<line class="base" x1="378" y1="78" x2="378" y2="294"/>
<text class="t1" x="606" y="64" font-weight="600">Divide</text>
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="294"/>
<text class="t2" x="606.0" y="310" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="646.0" y1="78" x2="646.0" y2="294"/>
<text class="t2" x="646.0" y="310" text-anchor="middle" font-size="11">100</text>
<line class="grid" x1="686.0" y1="78" x2="686.0" y2="294"/>
<text class="t2" x="686.0" y="310" text-anchor="middle" font-size="11">200</text>
<line class="grid" x1="726.0" y1="78" x2="726.0" y2="294"/>
<text class="t2" x="726.0" y="310" text-anchor="middle" font-size="11">300</text>
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="294"/>
<text class="t2" x="766.0" y="310" text-anchor="middle" font-size="11">400</text>
<text class="t2" x="766.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
<path class="f-number" d="M606.0,84.0 H627.7 Q631.7,84.0 631.7,88.0 V92.0 Q631.7,96.0 627.7,96.0 H606.0 Z"><title>Number (Large330): 64 ns per operation</title></path>
<text class="t2" x="636.7" y="94.0" font-size="11">64</text>
<path class="f-ieee64" d="M606.0,108.0 H615.7 Q619.7,108.0 619.7,112.0 V116.0 Q619.7,120.0 615.7,120.0 H606.0 Z"><title>Boost decimal64: 34 ns per operation</title></path>
<text class="t2" x="624.7" y="118.0" font-size="11">34</text>
<path class="f-ieee64" d="M606.0,132.0 H616.8 Q620.8,132.0 620.8,136.0 V140.0 Q620.8,144.0 616.8,144.0 H606.0 Z"><title>Boost decimal_fast64: 37 ns per operation</title></path>
<text class="t2" x="625.8" y="142.0" font-size="11">37</text>
<path class="f-ieee64" d="M606.0,156.0 H610.0 Q614.0,156.0 614.0,160.0 V164.0 Q614.0,168.0 610.0,168.0 H606.0 Z"><title>Intel BID64: 20 ns per operation</title></path>
<text class="t2" x="619.0" y="166.0" font-size="11">20</text>
<path class="f-wide" d="M606.0,180.0 H726.2 Q730.2,180.0 730.2,184.0 V188.0 Q730.2,192.0 726.2,192.0 H606.0 Z"><title>Boost decimal128: 310 ns per operation</title></path>
<text class="t2" x="735.2" y="190.0" font-size="11">310</text>
<path class="f-wide" d="M606.0,204.0 H648.7 Q652.7,204.0 652.7,208.0 V212.0 Q652.7,216.0 648.7,216.0 H606.0 Z"><title>Boost decimal_fast128: 117 ns per operation</title></path>
<text class="t2" x="657.7" y="214.0" font-size="11">117</text>
<path class="f-wide" d="M606.0,228.0 H627.4 Q631.4,228.0 631.4,232.0 V236.0 Q631.4,240.0 627.4,240.0 H606.0 Z"><title>Intel BID128: 63 ns per operation</title></path>
<text class="t2" x="636.4" y="238.0" font-size="11">63</text>
<path class="f-wide" d="M606.0,252.0 H623.2 Q627.2,252.0 627.2,256.0 V260.0 Q627.2,264.0 623.2,264.0 H606.0 Z"><title>mpdecimal 34: 53 ns per operation</title></path>
<text class="t2" x="632.2" y="262.0" font-size="11">53</text>
<path class="f-wide" d="M606.0,276.0 H624.5 Q628.5,276.0 628.5,280.0 V284.0 Q628.5,288.0 624.5,288.0 H606.0 Z"><title>mpdecimal 38: 56 ns per operation</title></path>
<text class="t2" x="633.5" y="286.0" font-size="11">56</text>
<line class="base" x1="606" y1="78" x2="606" y2="294"/>
</svg>

After

Width:  |  Height:  |  Size: 10 KiB

View File

@@ -0,0 +1,105 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 276" width="834" height="276" role="img" aria-label="360-payment amortization: correct digits (higher is better) and cost (lower is faster)">
<style>
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
</style>
<rect class="bg" width="834" height="276"/>
<text class="t1" x="0" y="18" font-size="15" font-weight="600">360-payment amortization: correct digits (higher is better) and cost (lower is faster)</text>
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="232.8" y="42">16-digit types</text>
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
<text class="t1" x="0" y="94.0">Number (Large330)</text>
<text class="t1" x="0" y="118.0">Boost decimal64</text>
<text class="t1" x="0" y="142.0">Intel BID64</text>
<text class="t1" x="0" y="166.0">Boost decimal128</text>
<text class="t1" x="0" y="190.0">Intel BID128</text>
<text class="t1" x="0" y="214.0">mpdecimal 34</text>
<text class="t1" x="0" y="238.0">mpdecimal 38</text>
<text class="t1" x="150" y="64" font-weight="600">Correct digits, worst case</text>
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="246"/>
<text class="t2" x="150.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="246"/>
<text class="t2" x="190.0" y="262" text-anchor="middle" font-size="11">10</text>
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="246"/>
<text class="t2" x="230.0" y="262" text-anchor="middle" font-size="11">20</text>
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="246"/>
<text class="t2" x="270.0" y="262" text-anchor="middle" font-size="11">30</text>
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="246"/>
<text class="t2" x="310.0" y="262" text-anchor="middle" font-size="11">40</text>
<text class="t2" x="310.0" y="274" text-anchor="end" font-size="11">significant digits</text>
<path class="f-number" d="M150.0,84.0 H202.3 Q206.3,84.0 206.3,88.0 V92.0 Q206.3,96.0 202.3,96.0 H150.0 Z"><title>Number (Large330): 14.1 significant digits</title></path>
<text class="t2" x="211.3" y="94.0" font-size="11">14.1</text>
<path class="f-ieee64" d="M150.0,108.0 H190.8 Q194.8,108.0 194.8,112.0 V116.0 Q194.8,120.0 190.8,120.0 H150.0 Z"><title>Boost decimal64: 11.2 significant digits</title></path>
<text class="t2" x="199.8" y="118.0" font-size="11">11.2</text>
<path class="f-ieee64" d="M150.0,132.0 H190.8 Q194.8,132.0 194.8,136.0 V140.0 Q194.8,144.0 190.8,144.0 H150.0 Z"><title>Intel BID64: 11.2 significant digits</title></path>
<text class="t2" x="199.8" y="142.0" font-size="11">11.2</text>
<path class="f-wide" d="M150.0,156.0 H262.5 Q266.5,156.0 266.5,160.0 V164.0 Q266.5,168.0 262.5,168.0 H150.0 Z"><title>Boost decimal128: 29.1 significant digits</title></path>
<text class="t2" x="271.5" y="166.0" font-size="11">29.1</text>
<path class="f-wide" d="M150.0,180.0 H262.5 Q266.5,180.0 266.5,184.0 V188.0 Q266.5,192.0 262.5,192.0 H150.0 Z"><title>Intel BID128: 29.1 significant digits</title></path>
<text class="t2" x="271.5" y="190.0" font-size="11">29.1</text>
<path class="f-wide" d="M150.0,204.0 H262.6 Q266.6,204.0 266.6,208.0 V212.0 Q266.6,216.0 262.6,216.0 H150.0 Z"><title>mpdecimal 34: 29.1 significant digits</title></path>
<text class="t2" x="271.6" y="214.0" font-size="11">29.1</text>
<path class="f-wide" d="M150.0,228.0 H277.9 Q281.9,228.0 281.9,232.0 V236.0 Q281.9,240.0 277.9,240.0 H150.0 Z"><title>mpdecimal 38: 33.0 significant digits</title></path>
<text class="t2" x="286.9" y="238.0" font-size="11">33.0</text>
<line class="base" x1="150" y1="78" x2="150" y2="246"/>
<text class="t1" x="378" y="64" font-weight="600">Correct digits, median</text>
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="246"/>
<text class="t2" x="378.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="246"/>
<text class="t2" x="418.0" y="262" text-anchor="middle" font-size="11">10</text>
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="246"/>
<text class="t2" x="458.0" y="262" text-anchor="middle" font-size="11">20</text>
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="246"/>
<text class="t2" x="498.0" y="262" text-anchor="middle" font-size="11">30</text>
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="246"/>
<text class="t2" x="538.0" y="262" text-anchor="middle" font-size="11">40</text>
<text class="t2" x="538.0" y="274" text-anchor="end" font-size="11">significant digits</text>
<path class="f-number" d="M378.0,84.0 H437.1 Q441.1,84.0 441.1,88.0 V92.0 Q441.1,96.0 437.1,96.0 H378.0 Z"><title>Number (Large330): 15.8 significant digits</title></path>
<text class="t2" x="446.1" y="94.0" font-size="11">15.8</text>
<path class="f-ieee64" d="M378.0,108.0 H425.4 Q429.4,108.0 429.4,112.0 V116.0 Q429.4,120.0 425.4,120.0 H378.0 Z"><title>Boost decimal64: 12.8 significant digits</title></path>
<text class="t2" x="434.4" y="118.0" font-size="11">12.8</text>
<path class="f-ieee64" d="M378.0,132.0 H425.4 Q429.4,132.0 429.4,136.0 V140.0 Q429.4,144.0 425.4,144.0 H378.0 Z"><title>Intel BID64: 12.8 significant digits</title></path>
<text class="t2" x="434.4" y="142.0" font-size="11">12.8</text>
<path class="f-wide" d="M378.0,156.0 H497.2 Q501.2,156.0 501.2,160.0 V164.0 Q501.2,168.0 497.2,168.0 H378.0 Z"><title>Boost decimal128: 30.8 significant digits</title></path>
<text class="t2" x="506.2" y="166.0" font-size="11">30.8</text>
<path class="f-wide" d="M378.0,180.0 H497.2 Q501.2,180.0 501.2,184.0 V188.0 Q501.2,192.0 497.2,192.0 H378.0 Z"><title>Intel BID128: 30.8 significant digits</title></path>
<text class="t2" x="506.2" y="190.0" font-size="11">30.8</text>
<path class="f-wide" d="M378.0,204.0 H497.2 Q501.2,204.0 501.2,208.0 V212.0 Q501.2,216.0 497.2,216.0 H378.0 Z"><title>mpdecimal 34: 30.8 significant digits</title></path>
<text class="t2" x="506.2" y="214.0" font-size="11">30.8</text>
<path class="f-wide" d="M378.0,228.0 H513.3 Q517.3,228.0 517.3,232.0 V236.0 Q517.3,240.0 513.3,240.0 H378.0 Z"><title>mpdecimal 38: 34.8 significant digits</title></path>
<text class="t2" x="522.3" y="238.0" font-size="11">34.8</text>
<line class="base" x1="378" y1="78" x2="378" y2="246"/>
<text class="t1" x="606" y="64" font-weight="600">Cost</text>
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="246"/>
<text class="t2" x="606.0" y="262" text-anchor="middle" font-size="11">0</text>
<line class="grid" x1="638.0" y1="78" x2="638.0" y2="246"/>
<text class="t2" x="638.0" y="262" text-anchor="middle" font-size="11">25</text>
<line class="grid" x1="670.0" y1="78" x2="670.0" y2="246"/>
<text class="t2" x="670.0" y="262" text-anchor="middle" font-size="11">50</text>
<line class="grid" x1="702.0" y1="78" x2="702.0" y2="246"/>
<text class="t2" x="702.0" y="262" text-anchor="middle" font-size="11">75</text>
<line class="grid" x1="734.0" y1="78" x2="734.0" y2="246"/>
<text class="t2" x="734.0" y="262" text-anchor="middle" font-size="11">100</text>
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="246"/>
<text class="t2" x="766.0" y="262" text-anchor="middle" font-size="11">125</text>
<text class="t2" x="766.0" y="274" text-anchor="end" font-size="11">µs per schedule</text>
<path class="f-number" d="M606.0,84.0 H748.1 Q752.1,84.0 752.1,88.0 V92.0 Q752.1,96.0 748.1,96.0 H606.0 Z"><title>Number (Large330): 114 µs per schedule</title></path>
<text class="t2" x="757.1" y="94.0" font-size="11">114</text>
<path class="f-ieee64" d="M606.0,108.0 H627.0 Q631.0,108.0 631.0,112.0 V116.0 Q631.0,120.0 627.0,120.0 H606.0 Z"><title>Boost decimal64: 20 µs per schedule</title></path>
<text class="t2" x="636.0" y="118.0" font-size="11">20</text>
<path class="f-ieee64" d="M606.0,132.0 H625.5 Q629.5,132.0 629.5,136.0 V140.0 Q629.5,144.0 625.5,144.0 H606.0 Z"><title>Intel BID64: 18 µs per schedule</title></path>
<text class="t2" x="634.5" y="142.0" font-size="11">18</text>
<path class="f-wide" d="M606.0,156.0 H726.1 Q730.1,156.0 730.1,160.0 V164.0 Q730.1,168.0 726.1,168.0 H606.0 Z"><title>Boost decimal128: 97 µs per schedule</title></path>
<text class="t2" x="735.1" y="166.0" font-size="11">97</text>
<path class="f-wide" d="M606.0,180.0 H676.9 Q680.9,180.0 680.9,184.0 V188.0 Q680.9,192.0 676.9,192.0 H606.0 Z"><title>Intel BID128: 58 µs per schedule</title></path>
<text class="t2" x="685.9" y="190.0" font-size="11">58</text>
<path class="f-wide" d="M606.0,204.0 H672.7 Q676.7,204.0 676.7,208.0 V212.0 Q676.7,216.0 672.7,216.0 H606.0 Z"><title>mpdecimal 34: 55 µs per schedule</title></path>
<text class="t2" x="681.7" y="214.0" font-size="11">55</text>
<path class="f-wide" d="M606.0,228.0 H659.8 Q663.8,228.0 663.8,232.0 V236.0 Q663.8,240.0 659.8,240.0 H606.0 Z"><title>mpdecimal 38: 45 µs per schedule</title></path>
<text class="t2" x="668.8" y="238.0" font-size="11">45</text>
<line class="base" x1="606" y1="78" x2="606" y2="246"/>
</svg>

After

Width:  |  Height:  |  Size: 9.3 KiB

View File

@@ -20,3 +20,103 @@ target_link_libraries(
xrpl_add_benchmark(nodestore)
target_link_libraries(xrpl.bench.nodestore PRIVATE xrpl.imports.bench)
add_dependencies(xrpl.benchmarks xrpl.bench.nodestore)
# Number vs. decimal floating-point libraries. See docs/NumberDecimalBenchmark.md.
# Boost.Decimal is header-only and ships with Boost >= 1.91.
xrpl_add_benchmark(number)
target_link_libraries(xrpl.bench.number PRIVATE xrpl.imports.bench)
add_dependencies(xrpl.benchmarks xrpl.bench.number)
# Benchmark a 128-bit Number prototype if one exists (see number/Number128.h).
# CONFIGURE_DEPENDS re-checks the glob on every build, so adding or removing
# the header is enough to reconfigure and rebuild.
file(
GLOB number128_header
CONFIGURE_DEPENDS
"${CMAKE_SOURCE_DIR}/include/xrpl/basics/Number128.h"
)
if(number128_header)
set(XRPL_BENCH_NUMBER128 ON)
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_NUMBER128)
endif()
if(bench_intel_dfp)
# Intel Decimal Floating-Point Math Library, fetched and built with its own
# makefile for local benchmarking only. If it is ever used beyond that, it
# should become a Conan recipe instead.
#
# CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0: arguments by value, rounding
# mode and status flags passed to every call. number/IntelBid.h defines the
# matching DECIMAL_* macros. The upstream makefile adds no optimization
# flags of its own, so pass -O3 to match the Release build.
include(ExternalProject)
find_program(XRPL_MAKE_EXECUTABLE NAMES gmake make REQUIRED)
set(intel_dfp_dir "${CMAKE_CURRENT_BINARY_DIR}/intel_dfp")
ExternalProject_Add(
intel_dfp
URL https://netlib.org/misc/intel/IntelRDFPMathLib20U4.tar.gz
URL_HASH
SHA256=1df86132e7a31fd74d784fee1c679b21a088f73a8ec979cfaf784c200392e125
DOWNLOAD_EXTRACT_TIMESTAMP TRUE
SOURCE_DIR "${intel_dfp_dir}"
BUILD_IN_SOURCE TRUE
CONFIGURE_COMMAND ""
BUILD_COMMAND
${XRPL_MAKE_EXECUTABLE} -C LIBRARY CC=${CMAKE_C_COMPILER}
CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0 UNCHANGED_BINARY_FLAGS=0
"CFLAGS_OPT=-O3 -fPIC"
INSTALL_COMMAND ""
BUILD_BYPRODUCTS "${intel_dfp_dir}/LIBRARY/libbid.a"
)
# Imported targets require their include directory to exist at generate time.
file(MAKE_DIRECTORY "${intel_dfp_dir}/LIBRARY/src")
add_library(xrpl.imports.intel_dfp STATIC IMPORTED)
set_target_properties(
xrpl.imports.intel_dfp
PROPERTIES
IMPORTED_LOCATION "${intel_dfp_dir}/LIBRARY/libbid.a"
INTERFACE_INCLUDE_DIRECTORIES "${intel_dfp_dir}/LIBRARY/src"
)
add_dependencies(xrpl.imports.intel_dfp intel_dfp)
target_link_libraries(xrpl.bench.number PRIVATE xrpl.imports.intel_dfp)
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_INTEL_DFP)
endif()
if(bench_mpdecimal)
find_package(mpdecimal REQUIRED)
target_link_libraries(xrpl.bench.number PRIVATE mpdecimal::libmpdecimal)
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_MPDECIMAL)
# The divergence harness is a plain executable, not a Google Benchmark, so
# it is defined here rather than with xrpl_add_benchmark. It shares the
# adapters in number/.
add_executable(xrpl.bench.number_diff number_diff/main.cpp)
isolate_headers(
xrpl.bench.number_diff
"${CMAKE_SOURCE_DIR}/src"
"${CMAKE_CURRENT_SOURCE_DIR}/number"
PRIVATE
)
target_link_libraries(
xrpl.bench.number_diff
PRIVATE xrpl.libxrpl mpdecimal::libmpdecimal
)
if(bench_intel_dfp)
target_link_libraries(
xrpl.bench.number_diff
PRIVATE xrpl.imports.intel_dfp
)
target_compile_definitions(
xrpl.bench.number_diff
PRIVATE XRPL_BENCH_INTEL_DFP
)
endif()
if(XRPL_BENCH_NUMBER128)
target_compile_definitions(
xrpl.bench.number_diff
PRIVATE XRPL_BENCH_NUMBER128
)
endif()
add_dependencies(xrpl.benchmarks xrpl.bench.number_diff)
endif()

View File

@@ -0,0 +1,263 @@
// Single-operation benchmarks for xrpl::Number and the decimal floating-point
// types it is being compared against. See docs/NumberDecimalBenchmark.md.
//
// Benchmark names are "<operation>/<dataset>/<type>", e.g.
// "mul/full/Number.Large330", so they can be filtered with
// --benchmark_filter='^mul/.*/(Number|BoostDecimalFast128)'.
#include <xrpl/basics/Number.h>
#include <benchmark/benchmark.h>
#include <benchmarks/libxrpl/number/BoostDecimal.h>
#include <benchmarks/libxrpl/number/Double.h>
#include <benchmarks/libxrpl/number/Number128.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <benchmarks/libxrpl/number/Operands.h>
#ifdef XRPL_BENCH_MPDECIMAL
#include <benchmarks/libxrpl/number/MpDecimal.h>
#endif
#ifdef XRPL_BENCH_INTEL_DFP
#include <benchmarks/libxrpl/number/IntelBid.h>
#endif
#include <array>
#include <cstddef>
#include <cstdint>
#include <format>
#include <functional>
#include <string>
#include <string_view>
#include <utility>
#include <vector>
namespace xrpl::number_bench {
namespace {
constexpr std::uint64_t kSeedLhs = 0x5eed'0001;
constexpr std::uint64_t kSeedRhs = 0x5eed'0002;
/**
* Environment for types that need no per-benchmark setup.
*/
struct NoEnv
{
};
/**
* Runs Number under a specific mantissa scale. Operands are materialized after
* the guard is in place, so they are normalized for that scale.
*/
template <MantissaRange::MantissaScale Scale>
struct NumberScaleEnv
{
NumberMantissaScaleGuard guard{Scale};
};
/**
* Independent operations over a batch: measures throughput.
*/
template <NumberLike T, class Env, class Op>
auto
binaryThroughput(std::vector<RawOperand> lhs, std::vector<RawOperand> rhs, Op op)
{
return [lhs = std::move(lhs), rhs = std::move(rhs), op](benchmark::State& state) {
[[maybe_unused]] Env const env{};
auto const a = materialize<T>(lhs);
auto const b = materialize<T>(rhs);
for (auto _ : state)
{
for (std::size_t i = 0; i < a.size(); ++i)
{
auto r = op(a[i], b[i]);
benchmark::DoNotOptimize(r);
}
}
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
};
}
template <NumberLike T, class Env, class Op>
auto
unaryThroughput(std::vector<RawOperand> operands, Op op)
{
return [operands = std::move(operands), op](benchmark::State& state) {
[[maybe_unused]] Env const env{};
auto const a = materialize<T>(operands);
for (auto _ : state)
{
for (auto const& x : a)
{
auto r = op(x);
benchmark::DoNotOptimize(r);
}
}
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
};
}
/**
* Construction from (mantissa, exponent), the path every STNumber/STAmount read takes.
*/
template <NumberLike T, class Env>
auto
constructThroughput(std::vector<RawOperand> operands)
{
return [operands = std::move(operands)](benchmark::State& state) {
[[maybe_unused]] Env const env{};
for (auto _ : state)
{
for (auto const& raw : operands)
{
T x{raw.mantissa, raw.exponent};
benchmark::DoNotOptimize(x);
}
}
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * operands.size()));
};
}
/**
* Each result feeds the next operation: measures latency.
*/
template <NumberLike T, class Env, class Op>
auto
chainLatency(std::vector<RawOperand> operands, Op op)
{
return [operands = std::move(operands), op](benchmark::State& state) {
[[maybe_unused]] Env const env{};
auto const a = materialize<T>(operands);
for (auto _ : state)
{
T acc{std::int64_t{1}};
for (auto const& x : a)
acc = op(acc, x);
benchmark::DoNotOptimize(acc);
}
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
};
}
template <class T, class Fn>
void
add(std::string_view op, std::string_view data, std::string_view subject, Fn fn)
{
benchmark::RegisterBenchmark(
std::format("{}/{}/{}", op, data, subject), [fn = std::move(fn)](benchmark::State& state) {
fn(state);
state.counters["sizeof"] = sizeof(T);
});
}
template <NumberLike T, class Env>
void
registerSubject(std::string_view subject)
{
constexpr auto kGeneral =
std::to_array({Dataset::Full, Dataset::Iou, Dataset::Drops, Dataset::Mpt});
for (auto const d : kGeneral)
{
auto const name = toString(d);
auto const lhs = makeRaw(d, kSeedLhs);
auto const rhs = makeRaw(d, kSeedRhs);
add<T>("add", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::plus<>{}));
add<T>("sub", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::minus<>{}));
add<T>("mul", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::multiplies<>{}));
add<T>("div", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::divides<>{}));
add<T>("lt", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::less<>{}));
add<T>("construct", name, subject, constructThroughput<T, Env>(lhs));
add<T>("to_string", name, subject, unaryThroughput<T, Env>(lhs, [](T const& x) {
return to_string(x);
}));
add<T>("root2", name, subject, unaryThroughput<T, Env>(lhs, [](T const& x) {
return root2(x);
}));
}
// Only integer-valued datasets fit in int64 without overflow.
for (auto const d : {Dataset::Drops, Dataset::Mpt})
{
add<T>(
"to_int64",
toString(d),
subject,
unaryThroughput<T, Env>(
makeRaw(d, kSeedLhs), [](T const& x) { return static_cast<std::int64_t>(x); }));
}
for (int const gap : {0, 1, 5, 18})
{
auto [lhs, rhs] = makeExponentGapPairs(gap, kSeedLhs);
add<T>(
std::format("add_gap{}", gap),
toString(Dataset::Full),
subject,
binaryThroughput<T, Env>(std::move(lhs), std::move(rhs), std::plus<>{}));
}
{
auto [lhs, rhs] = makeCancellationPairs(kSeedLhs);
add<T>(
"sub_cancel",
toString(Dataset::Full),
subject,
binaryThroughput<T, Env>(std::move(lhs), std::move(rhs), std::minus<>{}));
}
// 12 and 360: monthly payments for one year and for thirty years.
for (unsigned const n : {12u, 360u})
{
add<T>(
std::format("power{}", n),
toString(Dataset::Rate),
subject,
unaryThroughput<T, Env>(
makeRaw(Dataset::Rate, kSeedLhs), [n](T const& x) { return power(x, n); }));
}
add<T>(
"add_chain",
toString(Dataset::Full),
subject,
chainLatency<T, Env>(makeRaw(Dataset::Full, kSeedLhs), std::plus<>{}));
add<T>(
"mul_chain",
toString(Dataset::Rate),
subject,
chainLatency<T, Env>(makeRaw(Dataset::Rate, kSeedLhs), std::multiplies<>{}));
add<T>(
"div_chain",
toString(Dataset::Rate),
subject,
chainLatency<T, Env>(makeRaw(Dataset::Rate, kSeedLhs), std::divides<>{}));
}
[[maybe_unused]] bool const kRegistered = [] {
using enum MantissaRange::MantissaScale;
registerSubject<Number, NumberScaleEnv<Small>>("Number.Small");
registerSubject<Number, NumberScaleEnv<LargeLegacy>>("Number.LargeLegacy");
registerSubject<Number, NumberScaleEnv<Large330>>("Number.Large330");
registerSubject<BoostDecimal64, NoEnv>("BoostDecimal64");
registerSubject<BoostDecimalFast64, NoEnv>("BoostDecimalFast64");
registerSubject<BoostDecimal128, NoEnv>("BoostDecimal128");
registerSubject<BoostDecimalFast128, NoEnv>("BoostDecimalFast128");
registerSubject<Double, NoEnv>("Double");
#ifdef XRPL_BENCH_MPDECIMAL
registerSubject<MpDecimal19, NoEnv>("MpDecimal19");
registerSubject<MpDecimal34, NoEnv>("MpDecimal34");
registerSubject<MpDecimal38, NoEnv>("MpDecimal38");
#endif
#ifdef XRPL_BENCH_INTEL_DFP
registerSubject<IntelBid64, NoEnv>("IntelBid64");
registerSubject<IntelBid128, NoEnv>("IntelBid128");
#endif
#ifdef XRPL_BENCH_HAS_NUMBER128
registerSubject<Number128, NoEnv>("Number128");
#endif
return true;
}();
} // namespace
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,269 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <boost/decimal.hpp>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <array>
#include <cstdint>
#include <string>
namespace xrpl::number_bench {
namespace detail {
constexpr boost::decimal::rounding_mode
toBoost(Number::RoundingMode mode)
{
using enum Number::RoundingMode;
using boost::decimal::rounding_mode;
switch (mode)
{
case ToNearest:
return rounding_mode::fe_dec_to_nearest;
case TowardsZero:
return rounding_mode::fe_dec_toward_zero;
case Downward:
return rounding_mode::fe_dec_downward;
case Upward:
return rounding_mode::fe_dec_upward;
}
return rounding_mode::fe_dec_to_nearest;
}
constexpr Number::RoundingMode
fromBoost(boost::decimal::rounding_mode mode)
{
using enum Number::RoundingMode;
using boost::decimal::rounding_mode;
switch (mode)
{
case rounding_mode::fe_dec_toward_zero:
return TowardsZero;
case rounding_mode::fe_dec_downward:
return Downward;
case rounding_mode::fe_dec_upward:
return Upward;
default:
return ToNearest;
}
}
} // namespace detail
/**
* Wraps a Boost.Decimal type behind Number's interface.
*
* Every member is a one-line inline forward, so at -O2 and above the wrapper
* adds no cost over using the Boost type directly.
*
* Semantic differences from Number that the wrapper deliberately does NOT
* hide (they are part of what we are measuring):
* - Overflow produces infinity and division by zero produces inf/NaN
* instead of throwing.
* - The rounding mode is a process-wide global in Boost.Decimal, not
* thread-local like Number's.
* - decimal64_t / decimal_fast64_t carry 16 digits, so constructing from a
* 19-digit int64 mantissa rounds.
*/
template <class D>
class BoostDecimal final
{
D value_{};
explicit constexpr BoostDecimal(D value) noexcept : value_{value}
{
}
public:
using rep = std::int64_t;
using value_type = D;
constexpr BoostDecimal() = default;
// Implicit, like Number(rep).
constexpr BoostDecimal(rep mantissa) noexcept : value_{mantissa, 0} // NOLINT
{
}
explicit constexpr BoostDecimal(rep mantissa, int exponent) noexcept
: value_{mantissa, exponent}
{
}
[[nodiscard]] constexpr D
value() const noexcept
{
return value_;
}
constexpr BoostDecimal
operator-() const noexcept
{
return BoostDecimal{-value_};
}
constexpr BoostDecimal&
operator+=(BoostDecimal const& x) noexcept
{
value_ += x.value_;
return *this;
}
constexpr BoostDecimal&
operator-=(BoostDecimal const& x) noexcept
{
value_ -= x.value_;
return *this;
}
constexpr BoostDecimal&
operator*=(BoostDecimal const& x) noexcept
{
value_ *= x.value_;
return *this;
}
constexpr BoostDecimal&
operator/=(BoostDecimal const& x) noexcept
{
value_ /= x.value_;
return *this;
}
friend constexpr BoostDecimal
operator+(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return BoostDecimal{x.value_ + y.value_};
}
friend constexpr BoostDecimal
operator-(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return BoostDecimal{x.value_ - y.value_};
}
friend constexpr BoostDecimal
operator*(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return BoostDecimal{x.value_ * y.value_};
}
friend constexpr BoostDecimal
operator/(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return BoostDecimal{x.value_ / y.value_};
}
friend constexpr bool
operator==(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return x.value_ == y.value_;
}
friend constexpr bool
operator<(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return x.value_ < y.value_;
}
friend constexpr bool
operator>(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return y < x;
}
friend constexpr bool
operator<=(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return !(y < x);
}
friend constexpr bool
operator>=(BoostDecimal const& x, BoostDecimal const& y) noexcept
{
return !(x < y);
}
// Round to an integer using the current rounding mode, like Number.
explicit
operator rep() const noexcept
{
// Boost 1.91's llrint mishandles values that are already integers at full
// precision (significand exponent >= 0): it returns a wrong result, or
// divides by zero when the exponent is exactly 0. Fixed upstream (the
// same early return as here); remove this once Boost includes the fix.
int exponent = 0;
boost::decimal::frexp10(value_, &exponent);
if (exponent >= 0)
return static_cast<rep>(value_);
return static_cast<rep>(boost::decimal::llrint(value_));
}
static Number::RoundingMode
getround() noexcept
{
return detail::fromBoost(boost::decimal::fegetround());
}
// Returns the previous mode, like Number::setround.
static Number::RoundingMode
setround(Number::RoundingMode mode) noexcept
{
auto const old = getround();
boost::decimal::fesetround(detail::toBoost(mode));
return old;
}
friend std::string
to_string(BoostDecimal const& x)
{
std::array<char, 64> buffer{};
auto const result =
boost::decimal::to_chars(buffer.data(), buffer.data() + buffer.size(), x.value_);
return {buffer.data(), result.ptr};
}
friend constexpr BoostDecimal
abs(BoostDecimal const& x) noexcept
{
return BoostDecimal{boost::decimal::abs(x.value_)};
}
friend constexpr BoostDecimal
power(BoostDecimal const& x, unsigned n) noexcept
{
return BoostDecimal{boost::decimal::pow(x.value_, n)};
}
friend constexpr BoostDecimal
root2(BoostDecimal const& x) noexcept
{
return BoostDecimal{boost::decimal::sqrt(x.value_)};
}
friend constexpr Decomposed
decompose(BoostDecimal const& x) noexcept
{
int exponent = 0;
auto const coefficient = boost::decimal::frexp10(x.value_, &exponent);
return {
.negative = boost::decimal::signbit(x.value_),
.coefficient = static_cast<unsigned __int128>(coefficient),
.exponent = exponent};
}
};
using BoostDecimal64 = BoostDecimal<boost::decimal::decimal64_t>;
using BoostDecimal128 = BoostDecimal<boost::decimal::decimal128_t>;
using BoostDecimalFast64 = BoostDecimal<boost::decimal::decimal_fast64_t>;
using BoostDecimalFast128 = BoostDecimal<boost::decimal::decimal_fast128_t>;
static_assert(DecomposableNumber<BoostDecimal64>);
static_assert(DecomposableNumber<BoostDecimal128>);
static_assert(DecomposableNumber<BoostDecimalFast64>);
static_assert(DecomposableNumber<BoostDecimalFast128>);
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,200 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <cfenv>
#include <cmath>
#include <cstdint>
#include <sstream>
#include <string>
namespace xrpl::number_bench {
/**
* Binary `double` behind Number's interface: a hardware speed-of-light
* baseline, not a candidate (base 2, 53-bit significand).
*
* It is not DecomposableNumber because a binary value has no exact short
* decimal coefficient.
*/
class Double final
{
double value_ = 0.0;
explicit constexpr Double(double value) noexcept : value_{value}
{
}
public:
using rep = std::int64_t;
constexpr Double() = default;
// Implicit, like Number(rep).
Double(rep mantissa) noexcept : value_{static_cast<double>(mantissa)} // NOLINT
{
}
explicit Double(rep mantissa, int exponent) noexcept
: value_{static_cast<double>(mantissa) * std::pow(10.0, exponent)}
{
}
constexpr Double
operator-() const noexcept
{
return Double{-value_};
}
constexpr Double&
operator+=(Double const& x) noexcept
{
value_ += x.value_;
return *this;
}
constexpr Double&
operator-=(Double const& x) noexcept
{
value_ -= x.value_;
return *this;
}
constexpr Double&
operator*=(Double const& x) noexcept
{
value_ *= x.value_;
return *this;
}
constexpr Double&
operator/=(Double const& x) noexcept
{
value_ /= x.value_;
return *this;
}
friend constexpr Double
operator+(Double const& x, Double const& y) noexcept
{
return Double{x.value_ + y.value_};
}
friend constexpr Double
operator-(Double const& x, Double const& y) noexcept
{
return Double{x.value_ - y.value_};
}
friend constexpr Double
operator*(Double const& x, Double const& y) noexcept
{
return Double{x.value_ * y.value_};
}
friend constexpr Double
operator/(Double const& x, Double const& y) noexcept
{
return Double{x.value_ / y.value_};
}
friend constexpr bool
operator==(Double const& x, Double const& y) noexcept
{
return x.value_ == y.value_;
}
friend constexpr auto
operator<=>(Double const& x, Double const& y) noexcept
{
return x.value_ <=> y.value_;
}
explicit
operator rep() const noexcept
{
return static_cast<rep>(std::llrint(value_));
}
static Number::RoundingMode
getround() noexcept
{
using enum Number::RoundingMode;
switch (std::fegetround())
{
case FE_TOWARDZERO:
return TowardsZero;
case FE_DOWNWARD:
return Downward;
case FE_UPWARD:
return Upward;
default:
return ToNearest;
}
}
static Number::RoundingMode
setround(Number::RoundingMode mode) noexcept
{
using enum Number::RoundingMode;
auto const old = getround();
switch (mode)
{
case ToNearest:
std::fesetround(FE_TONEAREST);
break;
case TowardsZero:
std::fesetround(FE_TOWARDZERO);
break;
case Downward:
std::fesetround(FE_DOWNWARD);
break;
case Upward:
std::fesetround(FE_UPWARD);
break;
}
return old;
}
friend std::string
to_string(Double const& x)
{
std::ostringstream os;
os.precision(17);
os << x.value_;
return os.str();
}
friend Double
abs(Double const& x) noexcept
{
return Double{std::fabs(x.value_)};
}
// Repeated squaring, the same algorithm as xrpl::power(Number, unsigned).
friend constexpr Double
power(Double const& x, unsigned n) noexcept
{
double result = 1.0;
double base = x.value_;
for (; n != 0; n >>= 1)
{
if ((n & 1u) != 0)
result *= base;
base *= base;
}
return Double{result};
}
friend Double
root2(Double const& x) noexcept
{
return Double{std::sqrt(x.value_)};
}
};
static_assert(NumberLike<Double>);
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,377 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <array>
#include <cstdint>
#include <string>
#include <utility>
// The library must be built with the same three settings (see the
// bench_intel_dfp block in src/benchmarks/libxrpl/CMakeLists.txt): arguments
// by value, rounding mode passed to each call, status flags passed to each
// call. That is the configuration closest to Number, whose rounding mode is a
// per-thread setting rather than process-global state.
#define DECIMAL_CALL_BY_REFERENCE 0 // NOLINT(cppcoreguidelines-macro-usage)
#define DECIMAL_GLOBAL_ROUNDING 0 // NOLINT(cppcoreguidelines-macro-usage)
#define DECIMAL_GLOBAL_EXCEPTION_FLAGS 0 // NOLINT(cppcoreguidelines-macro-usage)
#include <bid_conf.h>
#include <bid_functions.h>
namespace xrpl::number_bench {
namespace detail {
/**
* One function table per interchange width, so IntelBid is written once.
*/
struct Bid64Ops
{
using Value = BID_UINT64;
static constexpr unsigned kDigits = 16;
static constexpr int kBias = 398;
// clang-format off
static Value add(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_add(x, y, r, f); }
static Value sub(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_sub(x, y, r, f); }
static Value mul(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_mul(x, y, r, f); }
static Value div(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_div(x, y, r, f); }
static Value sqrt(Value x, _IDEC_round r, _IDEC_flags* f) { return bid64_sqrt(x, r, f); }
static Value pow(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_pow(x, y, r, f); }
static Value fromInt64(std::int64_t x, _IDEC_round r, _IDEC_flags* f) { return bid64_from_int64(x, r, f); }
static Value scalbn(Value x, int n, _IDEC_round r, _IDEC_flags* f) { return bid64_scalbn(x, n, r, f); }
static Value negate(Value x) { return bid64_negate(x); }
static Value abs(Value x) { return bid64_abs(x); }
static bool less(Value x, Value y, _IDEC_flags* f) { return bid64_quiet_less(x, y, f) != 0; }
static bool equal(Value x, Value y, _IDEC_flags* f) { return bid64_quiet_equal(x, y, f) != 0; }
static std::int64_t toInt64Nearest(Value x, _IDEC_flags* f) { return bid64_to_int64_rnint(x, f); }
static std::int64_t toInt64TowardZero(Value x, _IDEC_flags* f) { return bid64_to_int64_int(x, f); }
static std::int64_t toInt64Floor(Value x, _IDEC_flags* f) { return bid64_to_int64_floor(x, f); }
static std::int64_t toInt64Ceil(Value x, _IDEC_flags* f) { return bid64_to_int64_ceil(x, f); }
static void toString(char* s, Value x, _IDEC_flags* f) { bid64_to_string(s, x, f); }
// clang-format on
/**
* Decodes the BID64 layout. Precondition: finite and canonical.
*/
static Decomposed
decompose(Value x) noexcept
{
bool const negative = (x >> 63) != 0;
bool const largeForm = ((x >> 61) & 3) == 3;
auto const exponent = static_cast<int>(largeForm ? (x >> 51) & 0x3ff : (x >> 53) & 0x3ff);
auto const coefficient = largeForm
? (x & 0x0007'ffff'ffff'ffffULL) | 0x0020'0000'0000'0000ULL
: x & 0x001f'ffff'ffff'ffffULL;
return {.negative = negative, .coefficient = coefficient, .exponent = exponent - kBias};
}
};
struct Bid128Ops
{
using Value = BID_UINT128;
static constexpr unsigned kDigits = 34;
static constexpr int kBias = 6176;
// clang-format off
static Value add(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_add(x, y, r, f); }
static Value sub(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_sub(x, y, r, f); }
static Value mul(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_mul(x, y, r, f); }
static Value div(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_div(x, y, r, f); }
static Value sqrt(Value x, _IDEC_round r, _IDEC_flags* f) { return bid128_sqrt(x, r, f); }
static Value pow(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_pow(x, y, r, f); }
static Value fromInt64(std::int64_t x, _IDEC_round, _IDEC_flags*) { return bid128_from_int64(x); }
static Value scalbn(Value x, int n, _IDEC_round r, _IDEC_flags* f) { return bid128_scalbn(x, n, r, f); }
static Value negate(Value x) { return bid128_negate(x); }
static Value abs(Value x) { return bid128_abs(x); }
static bool less(Value x, Value y, _IDEC_flags* f) { return bid128_quiet_less(x, y, f) != 0; }
static bool equal(Value x, Value y, _IDEC_flags* f) { return bid128_quiet_equal(x, y, f) != 0; }
static std::int64_t toInt64Nearest(Value x, _IDEC_flags* f) { return bid128_to_int64_rnint(x, f); }
static std::int64_t toInt64TowardZero(Value x, _IDEC_flags* f) { return bid128_to_int64_int(x, f); }
static std::int64_t toInt64Floor(Value x, _IDEC_flags* f) { return bid128_to_int64_floor(x, f); }
static std::int64_t toInt64Ceil(Value x, _IDEC_flags* f) { return bid128_to_int64_ceil(x, f); }
static void toString(char* s, Value x, _IDEC_flags* f) { bid128_to_string(s, x, f); }
// clang-format on
/**
* Decodes the BID128 layout. Precondition: finite and canonical.
*/
static Decomposed
decompose(Value x) noexcept
{
auto const hi = x.w[BID_HIGH_128W];
auto const lo = x.w[BID_LOW_128W];
bool const negative = (hi >> 63) != 0;
auto const exponent = static_cast<int>((hi >> 49) & 0x3fff);
auto const coefficient =
(static_cast<unsigned __int128>(hi & 0x0001'ffff'ffff'ffffULL) << 64) | lo;
return {.negative = negative, .coefficient = coefficient, .exponent = exponent - kBias};
}
};
} // namespace detail
/**
* Intel's Decimal Floating-Point Math Library (BID encoding) behind Number's
* interface.
*
* The rounding mode is thread-local, like Number's, and passed to each call.
* Status flags from every call accumulate in a thread-local; read and clear
* them with `takeStatus()`.
*
* Like the other IEEE types, overflow produces infinity and division by zero
* produces inf/NaN instead of throwing; decimal64 rounds 19-digit int64
* mantissas to 16 digits on construction.
*/
template <class Ops>
class IntelBid final
{
using Value = typename Ops::Value;
Value value_{};
struct State
{
_IDEC_round round = BID_ROUNDING_TO_NEAREST;
_IDEC_flags flags = 0;
};
static State&
state() noexcept
{
static thread_local State s;
return s;
}
static IntelBid
wrap(Value v) noexcept
{
IntelBid result;
result.value_ = v;
return result;
}
public:
using rep = std::int64_t;
IntelBid() : value_{Ops::fromInt64(0, BID_ROUNDING_TO_NEAREST, &state().flags)}
{
}
// Implicit, like Number(rep).
IntelBid(rep mantissa) : IntelBid(mantissa, 0) // NOLINT
{
}
explicit IntelBid(rep mantissa, int exponent)
{
auto& s = state();
value_ = Ops::fromInt64(mantissa, s.round, &s.flags);
if (exponent != 0)
value_ = Ops::scalbn(value_, exponent, s.round, &s.flags);
}
IntelBid
operator-() const noexcept
{
return wrap(Ops::negate(value_));
}
friend IntelBid
operator+(IntelBid const& x, IntelBid const& y) noexcept
{
auto& s = state();
return wrap(Ops::add(x.value_, y.value_, s.round, &s.flags));
}
friend IntelBid
operator-(IntelBid const& x, IntelBid const& y) noexcept
{
auto& s = state();
return wrap(Ops::sub(x.value_, y.value_, s.round, &s.flags));
}
friend IntelBid
operator*(IntelBid const& x, IntelBid const& y) noexcept
{
auto& s = state();
return wrap(Ops::mul(x.value_, y.value_, s.round, &s.flags));
}
friend IntelBid
operator/(IntelBid const& x, IntelBid const& y) noexcept
{
auto& s = state();
return wrap(Ops::div(x.value_, y.value_, s.round, &s.flags));
}
IntelBid&
operator+=(IntelBid const& x) noexcept
{
return *this = *this + x;
}
IntelBid&
operator-=(IntelBid const& x) noexcept
{
return *this = *this - x;
}
IntelBid&
operator*=(IntelBid const& x) noexcept
{
return *this = *this * x;
}
IntelBid&
operator/=(IntelBid const& x) noexcept
{
return *this = *this / x;
}
friend bool
operator==(IntelBid const& x, IntelBid const& y) noexcept
{
return Ops::equal(x.value_, y.value_, &state().flags);
}
friend bool
operator<(IntelBid const& x, IntelBid const& y) noexcept
{
return Ops::less(x.value_, y.value_, &state().flags);
}
friend bool
operator>(IntelBid const& x, IntelBid const& y) noexcept
{
return y < x;
}
friend bool
operator<=(IntelBid const& x, IntelBid const& y) noexcept
{
return !(y < x);
}
friend bool
operator>=(IntelBid const& x, IntelBid const& y) noexcept
{
return !(x < y);
}
// Round to an integer using the current rounding mode, like Number.
explicit
operator rep() const noexcept
{
auto& s = state();
switch (s.round)
{
case BID_ROUNDING_TO_ZERO:
return Ops::toInt64TowardZero(value_, &s.flags);
case BID_ROUNDING_DOWN:
return Ops::toInt64Floor(value_, &s.flags);
case BID_ROUNDING_UP:
return Ops::toInt64Ceil(value_, &s.flags);
default:
return Ops::toInt64Nearest(value_, &s.flags);
}
}
static Number::RoundingMode
getround() noexcept
{
using enum Number::RoundingMode;
switch (state().round)
{
case BID_ROUNDING_TO_ZERO:
return TowardsZero;
case BID_ROUNDING_DOWN:
return Downward;
case BID_ROUNDING_UP:
return Upward;
default:
return ToNearest;
}
}
static Number::RoundingMode
setround(Number::RoundingMode mode) noexcept
{
using enum Number::RoundingMode;
auto const old = getround();
auto& round = state().round;
switch (mode)
{
case ToNearest:
round = BID_ROUNDING_TO_NEAREST;
break;
case TowardsZero:
round = BID_ROUNDING_TO_ZERO;
break;
case Downward:
round = BID_ROUNDING_DOWN;
break;
case Upward:
round = BID_ROUNDING_UP;
break;
}
return old;
}
/**
* Returns the status flags raised since the last call, and clears them.
*/
static _IDEC_flags
takeStatus() noexcept
{
return std::exchange(state().flags, 0);
}
friend std::string
to_string(IntelBid const& x)
{
std::array<char, 64> buffer{};
Ops::toString(buffer.data(), x.value_, &state().flags);
return buffer.data();
}
friend IntelBid
abs(IntelBid const& x) noexcept
{
return wrap(Ops::abs(x.value_));
}
friend IntelBid
power(IntelBid const& x, unsigned n)
{
auto& s = state();
IntelBid const exponent{static_cast<rep>(n)};
return wrap(Ops::pow(x.value_, exponent.value_, s.round, &s.flags));
}
friend IntelBid
root2(IntelBid const& x) noexcept
{
auto& s = state();
return wrap(Ops::sqrt(x.value_, s.round, &s.flags));
}
/**
* Precondition: x is finite.
*/
friend Decomposed
decompose(IntelBid const& x) noexcept
{
return Ops::decompose(x.value_);
}
};
using IntelBid64 = IntelBid<detail::Bid64Ops>;
using IntelBid128 = IntelBid<detail::Bid128Ops>;
static_assert(DecomposableNumber<IntelBid64>);
static_assert(DecomposableNumber<IntelBid128>);
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,355 @@
// Ledger-formula benchmarks: AMM swaps, vault share conversion, and loan
// amortization (see Kernels.h), run identically on every NumberLike type.
// See docs/NumberDecimalBenchmark.md.
//
// Benchmark names are "kernel/<kernel>/<type>". Each reports throughput and,
// when built with mpdecimal (bench_mpdecimal), accuracy counters comparing the
// type's result with the same formula evaluated at 96 digits from the type's
// own inputs, so construction rounding is excluded:
//
// digits_min / digits_median correct significant digits, -log10(relative
// error), worst case and median over the
// cases; 40 means exact. For loan_amortize the
// error is the final balance relative to the
// principal.
// mismatches for integer results: cases that differ from
// the exact truncated result.
#include <benchmarks/libxrpl/number/Kernels.h>
#include <xrpl/basics/Number.h>
#include <benchmark/benchmark.h>
#include <benchmarks/libxrpl/number/BoostDecimal.h>
#include <benchmarks/libxrpl/number/Double.h>
#include <benchmarks/libxrpl/number/Number128.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <benchmarks/libxrpl/number/Operands.h>
#ifdef XRPL_BENCH_MPDECIMAL
#include <benchmarks/libxrpl/number/MpDecimal.h>
#endif
#ifdef XRPL_BENCH_INTEL_DFP
#include <benchmarks/libxrpl/number/IntelBid.h>
#endif
#include <algorithm>
#include <array>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <format>
#include <random>
#include <string>
#include <string_view>
#include <type_traits>
#include <utility>
#include <vector>
namespace xrpl::number_bench {
namespace {
constexpr std::size_t kCases = 256;
constexpr std::uint64_t kSeed = 0x6b65'726e;
/**
* Inputs for one kernel evaluation: up to three decimal values and two
* integer parameters (fee, rate, interval, payment count).
*/
struct KernelCase
{
std::array<RawOperand, 3> values;
std::array<std::uint32_t, 3> integers;
};
template <class T>
using Values = std::array<T, 3>;
// ---- Case generation ----
/**
* A positive value with a 16-digit mantissa and magnitude 10^lo to 10^hi.
*/
RawOperand
amount(std::mt19937_64& rng, int lo, int hi)
{
constexpr auto kE15 = static_cast<std::int64_t>(kPowerOfTen[15]);
auto const mantissa = std::uniform_int_distribution<std::int64_t>{kE15, (kE15 * 10) - 1}(rng);
return {mantissa, std::uniform_int_distribution<int>{lo, hi}(rng)-15};
}
/**
* `base` scaled by a factor 10^lo to 10^hi, with a fresh mantissa.
*/
RawOperand
fractionOf(std::mt19937_64& rng, RawOperand base, int lo, int hi)
{
auto result = amount(rng, 0, 0);
result.exponent = base.exponent + std::uniform_int_distribution<int>{lo, hi}(rng);
return result;
}
/**
* Pools of 10^6 to 10^12 per side, trades of 10^-6 to 10^-1 of the pool,
* fees of 0 to 1%.
*/
std::vector<KernelCase>
ammCases()
{
std::mt19937_64 rng{kSeed};
std::vector<KernelCase> cases;
for (std::size_t i = 0; i < kCases; ++i)
{
auto const poolIn = amount(rng, 6, 12);
auto const poolOut = amount(rng, 6, 12);
// Trades are a fraction of the out pool for swap-out to stay solvent.
auto const trade = fractionOf(rng, poolOut, -6, -2);
auto const fee = std::uniform_int_distribution<std::uint32_t>{0, 1000}(rng);
cases.push_back({{poolIn, poolOut, trade}, {fee, 0, 0}});
}
return cases;
}
/**
* Vaults holding 10^9 to 10^17 drops-like integer assets, share supply 10^6
* times larger (vault scale 6) within a factor of 2, but at most 10^18 since
* MPT supply is capped at 2^63-1; deposits of 10^-6 to 10^-1 of the assets.
*/
std::vector<KernelCase>
vaultCases()
{
std::mt19937_64 rng{kSeed + 1};
std::vector<KernelCase> cases;
for (std::size_t i = 0; i < kCases; ++i)
{
auto const assetDigits = std::uniform_int_distribution<int>{10, 17}(rng);
auto const assetTotal = std::uniform_int_distribution<std::int64_t>{
static_cast<std::int64_t>(kPowerOfTen[assetDigits - 1]),
static_cast<std::int64_t>(kPowerOfTen[assetDigits]) - 1}(rng);
auto const ratio = std::uniform_real_distribution<double>{0.5, 2.0}(rng);
auto const shareMantissa =
static_cast<std::int64_t>(static_cast<double>(assetTotal) * ratio) | 1;
auto const shareExponent = std::min(6, 18 - assetDigits);
auto const deposit = std::uniform_int_distribution<std::int64_t>{
std::max<std::int64_t>(1, assetTotal / 1'000'000), assetTotal / 10}(rng);
cases.push_back(
{{RawOperand{shareMantissa, shareExponent},
RawOperand{assetTotal, 0},
RawOperand{deposit, 0}},
{}});
}
return cases;
}
/**
* Principal of 10^3 to 10^9, annual rates of 0.1% to 30% (in 1/10 bips),
* monthly payments.
*/
std::vector<KernelCase>
loanCases(std::uint32_t payments)
{
constexpr std::uint32_t kMonth = 30 * 24 * 60 * 60;
std::mt19937_64 rng{kSeed + 2 + payments};
std::vector<KernelCase> cases;
for (std::size_t i = 0; i < kCases; ++i)
{
auto const principal = amount(rng, 3, 9);
auto const rate = std::uniform_int_distribution<std::uint32_t>{100, 30'000}(rng);
cases.push_back(
{{principal, RawOperand{0, 0}, RawOperand{0, 0}}, {rate, kMonth, payments}});
}
return cases;
}
// ---- Kernels, adapted to the common case shape ----
auto const kAmmSwapIn = [](auto const& v, auto const& i) {
return ammSwapIn(v[0], v[1], v[2], static_cast<std::uint16_t>(i[0]));
};
auto const kAmmSwapOut = [](auto const& v, auto const& i) {
return ammSwapOut(v[0], v[1], v[2], static_cast<std::uint16_t>(i[0]));
};
auto const kVaultAssetsToShares = [](auto const& v, auto const&) {
return vaultAssetsToShares(v[0], v[1], v[2]);
};
auto const kVaultSharesToAssets = [](auto const& v, auto const&) {
// Redeem the same magnitude of shares as the deposit's asset amount.
return vaultSharesToAssets(v[1], v[0], v[2]);
};
auto const kLoanPayment = [](auto const& v, auto const& i) {
using T = std::decay_t<decltype(v[0])>;
return loanPeriodicPayment(v[0], loanPeriodicRate<T>(i[0], i[1]), i[2]);
};
auto const kLoanAmortize = [](auto const& v, auto const& i) {
using T = std::decay_t<decltype(v[0])>;
return loanAmortize(v[0], loanPeriodicRate<T>(i[0], i[1]), i[2]);
};
// ---- Accuracy against a 96-digit evaluation ----
#ifdef XRPL_BENCH_MPDECIMAL
using Exact = MpDecimal<96>;
template <NumberLike T>
Exact
toExact(T const& x)
{
if constexpr (DecomposableNumber<T>)
{
return Exact::fromDecomposed(decompose(x));
}
else
{
Exact result;
mpd_context_t ctx;
mpd_maxcontext(&ctx);
std::uint32_t status = 0;
mpd_qset_string(result.get(), to_string(x).c_str(), &ctx, &status);
return result;
}
}
double
correctDigits(Exact const& error, Exact const& scale)
{
constexpr double kExact = 40;
if (mpd_iszero(error.get()) != 0)
return kExact;
auto const relative = std::stod(to_string(abs(error) / abs(scale)));
return std::min(kExact, -std::log10(relative));
}
/**
* Sets digits_min/digits_median (or mismatches) on `state`.
*/
template <NumberLike T, class Fn>
void
measureAccuracy(
benchmark::State& state,
std::vector<Values<T>> const& inputs,
std::vector<KernelCase> const& cases,
Fn fn,
bool relativeToFirstInput)
{
std::vector<double> digits;
std::size_t mismatches = 0;
for (std::size_t c = 0; c < inputs.size(); ++c)
{
Values<Exact> const exactInputs{
toExact(inputs[c][0]), toExact(inputs[c][1]), toExact(inputs[c][2])};
auto const result = fn(inputs[c], cases[c].integers);
auto const exact = fn(exactInputs, cases[c].integers);
if constexpr (std::is_integral_v<std::decay_t<decltype(result)>>)
{
mismatches += result == exact ? 0 : 1;
}
else
{
auto const& scale = relativeToFirstInput ? exactInputs[0] : exact;
digits.push_back(correctDigits(toExact(result) - exact, scale));
}
}
if (digits.empty())
{
state.counters["mismatches"] = static_cast<double>(mismatches);
return;
}
std::ranges::sort(digits);
state.counters["digits_min"] = digits.front();
state.counters["digits_median"] = digits[digits.size() / 2];
}
#endif
// ---- Benchmarks ----
struct NoEnv
{
};
template <MantissaRange::MantissaScale Scale>
struct NumberScaleEnv
{
NumberMantissaScaleGuard guard{Scale};
};
template <NumberLike T, class Env, class Fn>
void
registerKernel(
std::string_view kernel,
std::string_view subject,
std::vector<KernelCase> const& cases,
Fn fn,
bool relativeToFirstInput = false)
{
benchmark::RegisterBenchmark(
std::format("kernel/{}/{}", kernel, subject),
[cases, fn, relativeToFirstInput](benchmark::State& state) {
[[maybe_unused]] Env const env{};
std::vector<Values<T>> inputs;
inputs.reserve(cases.size());
for (auto const& c : cases)
{
inputs.push_back(
{T{c.values[0].mantissa, c.values[0].exponent},
T{c.values[1].mantissa, c.values[1].exponent},
T{c.values[2].mantissa, c.values[2].exponent}});
}
for (auto _ : state)
{
for (std::size_t c = 0; c < inputs.size(); ++c)
{
auto r = fn(inputs[c], cases[c].integers);
benchmark::DoNotOptimize(r);
}
}
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * inputs.size()));
#ifdef XRPL_BENCH_MPDECIMAL
measureAccuracy<T>(state, inputs, cases, fn, relativeToFirstInput);
#else
(void)relativeToFirstInput;
#endif
});
}
template <NumberLike T, class Env>
void
registerSubject(std::string_view subject)
{
static auto const kAmm = ammCases();
static auto const kVault = vaultCases();
static auto const kLoan12 = loanCases(12);
static auto const kLoan360 = loanCases(360);
registerKernel<T, Env>("amm_swap_in", subject, kAmm, kAmmSwapIn);
registerKernel<T, Env>("amm_swap_out", subject, kAmm, kAmmSwapOut);
registerKernel<T, Env>("vault_assets_to_shares", subject, kVault, kVaultAssetsToShares);
registerKernel<T, Env>("vault_shares_to_assets", subject, kVault, kVaultSharesToAssets);
registerKernel<T, Env>("loan_payment12", subject, kLoan12, kLoanPayment);
registerKernel<T, Env>("loan_payment360", subject, kLoan360, kLoanPayment);
registerKernel<T, Env>("loan_amortize12", subject, kLoan12, kLoanAmortize, true);
registerKernel<T, Env>("loan_amortize360", subject, kLoan360, kLoanAmortize, true);
}
[[maybe_unused]] bool const kRegistered = [] {
using enum MantissaRange::MantissaScale;
registerSubject<Number, NumberScaleEnv<Small>>("Number.Small");
registerSubject<Number, NumberScaleEnv<Large330>>("Number.Large330");
registerSubject<BoostDecimal64, NoEnv>("BoostDecimal64");
registerSubject<BoostDecimal128, NoEnv>("BoostDecimal128");
registerSubject<Double, NoEnv>("Double");
#ifdef XRPL_BENCH_MPDECIMAL
registerSubject<MpDecimal34, NoEnv>("MpDecimal34");
registerSubject<MpDecimal38, NoEnv>("MpDecimal38");
#endif
#ifdef XRPL_BENCH_INTEL_DFP
registerSubject<IntelBid64, NoEnv>("IntelBid64");
registerSubject<IntelBid128, NoEnv>("IntelBid128");
#endif
#ifdef XRPL_BENCH_HAS_NUMBER128
registerSubject<Number128, NoEnv>("Number128");
#endif
return true;
}();
} // namespace
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,220 @@
#pragma once
// Ledger formulas, written once against NumberLike so every library runs the
// identical computation. Each kernel mirrors a production function; keep them
// in sync if those change:
//
// ammSwapIn / ammSwapOut swapAssetIn / swapAssetOut, fixAMMv1_1 branch
// (include/xrpl/ledger/helpers/AMMHelpers.h)
// vaultAssetsToShares assetsToSharesDeposit
// vaultSharesToAssets sharesToAssetsDeposit
// (src/libxrpl/ledger/helpers/VaultHelpers.cpp)
// loanPeriodicRate loanPeriodicRate
// loanPowerMinusOne computePowerMinusOneHybrid
// loanPeriodicPayment loanPeriodicPayment, fixCleanup3_2_0 branch
// (src/libxrpl/ledger/helpers/LendingHelpers.cpp)
//
// loanAmortize is not a production function: it runs a full payment schedule
// (interest = balance * rate; balance -= payment - interest) so error
// accumulates the way it does over a loan's life. The exact remaining balance
// after the last payment is 0.
//
// The kernels operate on bare values, not STAmount or ledger entries, and
// omit the final rounding to the asset's scale, so they measure arithmetic
// alone.
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <cstdint>
namespace xrpl::number_bench {
namespace detail {
template <NumberLike T>
T
integer(std::int64_t value)
{
return T{value};
}
/**
* Sets a rounding mode for its lifetime, then restores the previous one.
*/
template <NumberLike T>
class RoundGuard
{
Number::RoundingMode saved_;
public:
explicit RoundGuard(Number::RoundingMode mode) : saved_{T::setround(mode)}
{
}
~RoundGuard()
{
T::setround(saved_);
}
RoundGuard(RoundGuard const&) = delete;
RoundGuard&
operator=(RoundGuard const&) = delete;
};
} // namespace detail
/**
* Amount of the out asset received for swapping `assetIn` into the pool.
* Rounding at each step favors the AMM.
*/
template <NumberLike T>
T
ammSwapIn(T const& poolIn, T const& poolOut, T const& assetIn, std::uint16_t tfee)
{
using enum Number::RoundingMode;
auto const saved = T::getround();
T::setround(Upward);
auto const numerator = poolIn * poolOut;
auto const fee = detail::integer<T>(tfee) / detail::integer<T>(100'000);
T::setround(Downward);
auto const denom = poolIn + (assetIn * (detail::integer<T>(1) - fee));
T::setround(Upward);
auto const ratio = numerator / denom;
T::setround(Downward);
auto const swapOut = poolOut - ratio;
T::setround(saved);
return swapOut;
}
/**
* Amount of the in asset required to take `assetOut` out of the pool.
* Rounding at each step favors the AMM.
*/
template <NumberLike T>
T
ammSwapOut(T const& poolIn, T const& poolOut, T const& assetOut, std::uint16_t tfee)
{
using enum Number::RoundingMode;
auto const saved = T::getround();
T::setround(Upward);
auto const numerator = poolIn * poolOut;
T::setround(Downward);
auto const denom = poolOut - assetOut;
T::setround(Upward);
auto const ratio = numerator / denom;
auto const numerator2 = ratio - poolIn;
auto const fee = detail::integer<T>(tfee) / detail::integer<T>(100'000);
T::setround(Downward);
auto const feeMult = detail::integer<T>(1) - fee;
T::setround(Upward);
auto const swapIn = numerator2 / feeMult;
T::setround(saved);
return swapIn;
}
/**
* Shares minted for depositing `assets` into a non-empty vault, truncated to
* an integer (vault shares are MPTs).
*/
template <NumberLike T>
std::int64_t
vaultAssetsToShares(T const& shareTotal, T const& assetTotal, T const& assets)
{
auto const shares = (shareTotal * assets) / assetTotal;
detail::RoundGuard<T> const guard{Number::RoundingMode::TowardsZero};
return static_cast<std::int64_t>(shares);
}
/**
* Assets corresponding to `shares` of a non-empty vault.
*/
template <NumberLike T>
T
vaultSharesToAssets(T const& assetTotal, T const& shareTotal, T const& shares)
{
return (assetTotal * shares) / shareTotal;
}
constexpr std::uint32_t kSecondsInYear = 365 * 24 * 60 * 60;
/**
* Interest rate per payment interval, from an annual rate in 1/10 bips.
*/
template <NumberLike T>
T
loanPeriodicRate(std::uint32_t interestRateTenthBips, std::uint32_t paymentInterval)
{
auto const interval = detail::integer<T>(paymentInterval);
return interval * detail::integer<T>(interestRateTenthBips) / detail::integer<T>(100'000) /
detail::integer<T>(kSecondsInYear);
}
/**
* (1 + r)^n - 1: closed form, or a binomial series when r * n is tiny and the
* closed form would cancel catastrophically.
*/
template <NumberLike T>
T
loanPowerMinusOne(T const& rate, std::uint32_t payments)
{
auto const n = detail::integer<T>(payments);
if (n * rate >= T{1, -9})
return power(detail::integer<T>(1) + rate, payments) - detail::integer<T>(1);
T term = n * rate;
T sum = term;
for (std::uint32_t k = 1; k < payments; ++k)
{
term = term * rate * detail::integer<T>(payments - k) / detail::integer<T>(k + 1);
T const next = sum + term;
if (next == sum)
break;
sum = next;
}
return sum;
}
/**
* Level payment that amortizes `principal` over `payments` periods.
*/
template <NumberLike T>
T
loanPeriodicPayment(T const& principal, T const& rate, std::uint32_t payments)
{
auto const raisedMinusOne = loanPowerMinusOne(rate, payments);
auto const raised = detail::integer<T>(1) + raisedMinusOne;
return principal * ((rate * raised) / raisedMinusOne);
}
/**
* Remaining balance after making every scheduled payment. Exactly 0 in exact
* arithmetic; anything else is accumulated rounding error.
*/
template <NumberLike T>
T
loanAmortize(T const& principal, T const& rate, std::uint32_t payments)
{
auto const payment = loanPeriodicPayment(principal, rate, payments);
T balance = principal;
for (std::uint32_t i = 0; i < payments; ++i)
{
auto const interest = balance * rate;
balance -= payment - interest;
}
return balance;
}
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,395 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <mpdecimal.h>
#include <cstdint>
#include <memory>
#include <string>
#include <utility>
namespace xrpl::number_bench {
/**
* mpdecimal (libmpdec) at a fixed precision, behind Number's interface.
*
* This is the correctly-rounded reference for the divergence harness: with
* `allcr` set, every operation, including `power`, is correctly rounded in
* the active rounding mode.
*
* `Digits` is the coefficient precision: 19 to mirror today's Number, 34 for
* IEEE decimal128, 38 for a full-uint128 Number. The exponent range follows
* Number's convention: value = coefficient * 10^e with e in
* [Number::kMinExponent, Number::kMaxExponent] for a `Digits`-digit
* coefficient. mpdecimal uses adjusted exponents (the exponent of the leading
* digit), hence the `Digits - 1` offset.
*
* Coefficients live in inline storage sized for `Digits`, so copying never
* allocates. Library temporaries may still allocate; that is part of
* mpdecimal's cost. Precisions above 38 digits exist for the divergence
* harness, which needs a near-exact result to round from; `decompose` is only
* available up to 38 digits.
*
* Unlike Number, results that overflow become infinity and underflow becomes
* subnormal or zero. Each operation ORs its mpdecimal status into a
* thread-local accumulator; read and clear it with `takeStatus()`.
*/
template <unsigned Digits>
class MpDecimal final
{
static_assert(Digits >= 16 && Digits <= 114);
// Enough 19-digit limbs for a Digits-digit coefficient, and never fewer
// than MPD_MINALLOC_MIN.
static constexpr mpd_ssize_t kLimbs = (Digits / MPD_RDIGITS) + 2;
mpd_uint_t data_[kLimbs]{};
// Zero: one limb holding 0.
mpd_t dec_{MPD_STATIC | MPD_STATIC_DATA, 0, 1, 1, kLimbs, data_};
struct Context
{
mpd_context_t ctx{};
uint32_t status = 0;
Context()
{
mpd_defaultcontext(&ctx);
ctx.prec = Digits;
ctx.emax = Number::kMaxExponent + static_cast<int>(Digits) - 1;
ctx.emin = Number::kMinExponent + static_cast<int>(Digits) - 1;
ctx.round = MPD_ROUND_HALF_EVEN;
ctx.traps = 0;
ctx.clamp = 0;
ctx.allcr = 1;
}
};
static Context&
context() noexcept
{
static thread_local Context c;
return c;
}
template <class F>
static MpDecimal
compute(F f)
{
MpDecimal result;
auto& c = context();
f(&result.dec_, &c.ctx, &c.status);
return result;
}
public:
using rep = std::int64_t;
static constexpr unsigned kDigits = Digits;
MpDecimal() noexcept = default;
~MpDecimal()
{
// Frees the coefficient only if an operation outgrew inline storage.
mpd_del(&dec_);
}
MpDecimal(MpDecimal const& other) noexcept
{
mpd_qcopy(&dec_, &other.dec_, &context().status);
}
MpDecimal&
operator=(MpDecimal const& other) noexcept
{
if (this != &other)
mpd_qcopy(&dec_, &other.dec_, &context().status);
return *this;
}
// Implicit, like Number(rep).
MpDecimal(rep mantissa) : MpDecimal(mantissa, 0) // NOLINT
{
}
explicit MpDecimal(rep mantissa, int exponent)
{
auto& c = context();
mpd_qset_i64_exact(&dec_, mantissa, &c.status);
dec_.exp += exponent;
mpd_qfinalize(&dec_, &c.ctx, &c.status);
}
MpDecimal
operator-() const
{
return compute(
[this](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qminus(r, &dec_, ctx, s); });
}
friend MpDecimal
operator+(MpDecimal const& x, MpDecimal const& y)
{
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qadd(r, &x.dec_, &y.dec_, ctx, s);
});
}
friend MpDecimal
operator-(MpDecimal const& x, MpDecimal const& y)
{
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qsub(r, &x.dec_, &y.dec_, ctx, s);
});
}
friend MpDecimal
operator*(MpDecimal const& x, MpDecimal const& y)
{
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qmul(r, &x.dec_, &y.dec_, ctx, s);
});
}
friend MpDecimal
operator/(MpDecimal const& x, MpDecimal const& y)
{
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qdiv(r, &x.dec_, &y.dec_, ctx, s);
});
}
MpDecimal&
operator+=(MpDecimal const& x)
{
return *this = *this + x;
}
MpDecimal&
operator-=(MpDecimal const& x)
{
return *this = *this - x;
}
MpDecimal&
operator*=(MpDecimal const& x)
{
return *this = *this * x;
}
MpDecimal&
operator/=(MpDecimal const& x)
{
return *this = *this / x;
}
friend bool
operator==(MpDecimal const& x, MpDecimal const& y) noexcept
{
return mpd_qcmp(&x.dec_, &y.dec_, &context().status) == 0;
}
friend bool
operator<(MpDecimal const& x, MpDecimal const& y) noexcept
{
return mpd_qcmp(&x.dec_, &y.dec_, &context().status) == -1;
}
friend bool
operator>(MpDecimal const& x, MpDecimal const& y) noexcept
{
return y < x;
}
friend bool
operator<=(MpDecimal const& x, MpDecimal const& y) noexcept
{
return x < y || x == y;
}
friend bool
operator>=(MpDecimal const& x, MpDecimal const& y) noexcept
{
return y <= x;
}
// Round to an integer using the current rounding mode, like Number.
explicit
operator rep() const
{
auto const rounded = compute([this](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qround_to_int(r, &dec_, ctx, s);
});
return mpd_qget_i64(&rounded.dec_, &context().status);
}
static Number::RoundingMode
getround() noexcept
{
using enum Number::RoundingMode;
switch (context().ctx.round)
{
case MPD_ROUND_DOWN:
return TowardsZero;
case MPD_ROUND_FLOOR:
return Downward;
case MPD_ROUND_CEILING:
return Upward;
default:
return ToNearest;
}
}
static Number::RoundingMode
setround(Number::RoundingMode mode) noexcept
{
using enum Number::RoundingMode;
auto const old = getround();
auto& round = context().ctx.round;
switch (mode)
{
case ToNearest:
round = MPD_ROUND_HALF_EVEN;
break;
case TowardsZero:
round = MPD_ROUND_DOWN;
break;
case Downward:
round = MPD_ROUND_FLOOR;
break;
case Upward:
round = MPD_ROUND_CEILING;
break;
}
return old;
}
/**
* Returns the mpdecimal status flags raised since the last call, and clears them.
*/
static uint32_t
takeStatus() noexcept
{
return std::exchange(context().status, 0);
}
friend std::string
to_string(MpDecimal const& x)
{
char* raw = nullptr;
auto const size = mpd_to_sci_size(&raw, &x.dec_, 0);
std::unique_ptr<char, void (*)(void*)> const owned{raw, mpd_free};
return size < 0 ? std::string{} : std::string{raw, static_cast<std::size_t>(size)};
}
friend MpDecimal
abs(MpDecimal const& x)
{
return compute(
[&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qabs(r, &x.dec_, ctx, s); });
}
friend MpDecimal
power(MpDecimal const& x, unsigned n)
{
MpDecimal const exponent{static_cast<rep>(n)};
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
mpd_qpow(r, &x.dec_, &exponent.dec_, ctx, s);
});
}
friend MpDecimal
root2(MpDecimal const& x)
{
return compute(
[&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qsqrt(r, &x.dec_, ctx, s); });
}
/**
* The underlying libmpdec value, for the divergence harness.
*/
[[nodiscard]] mpd_t const*
get() const noexcept
{
return &dec_;
}
[[nodiscard]] mpd_t*
get() noexcept
{
return &dec_;
}
/**
* Rounds `x` to this type's precision in the current rounding mode.
*/
template <unsigned OtherDigits>
static MpDecimal
roundFrom(MpDecimal<OtherDigits> const& x)
{
MpDecimal result;
auto& c = context();
mpd_qcopy(&result.dec_, x.get(), &c.status);
mpd_qfinalize(&result.dec_, &c.ctx, &c.status);
return result;
}
/**
* Exact conversion; rounds only if `d` has more than Digits digits.
*/
static MpDecimal
fromDecomposed(Decomposed const& d)
{
auto& c = context();
mpd_context_t exact;
mpd_maxcontext(&exact);
exact.traps = 0;
// coefficient = hi * 2^64 + lo, assembled at unlimited precision.
MpDecimal result;
MpDecimal lo;
MpDecimal shift;
mpd_qset_u64_exact(
&result.dec_, static_cast<std::uint64_t>(d.coefficient >> 64), &c.status);
mpd_qset_u64_exact(&lo.dec_, static_cast<std::uint64_t>(d.coefficient), &c.status);
mpd_qset_u64_exact(&shift.dec_, std::uint64_t{1} << 32, &c.status);
mpd_qmul(&shift.dec_, &shift.dec_, &shift.dec_, &exact, &c.status);
mpd_qmul(&result.dec_, &result.dec_, &shift.dec_, &exact, &c.status);
mpd_qadd(&result.dec_, &result.dec_, &lo.dec_, &exact, &c.status);
if (d.negative)
mpd_set_negative(&result.dec_);
result.dec_.exp += d.exponent;
mpd_qfinalize(&result.dec_, &c.ctx, &c.status);
return result;
}
/**
* Precondition: x is finite (check takeStatus() for overflow/invalid first).
*/
friend Decomposed
decompose(MpDecimal const& x) noexcept
requires(Digits <= 38)
{
unsigned __int128 coefficient = 0;
for (auto i = x.dec_.len; i-- > 0;)
coefficient = (coefficient * MPD_RADIX) + x.dec_.data[i];
return {
.negative = (x.dec_.flags & MPD_NEG) != 0,
.coefficient = coefficient,
.exponent = static_cast<int>(x.dec_.exp)};
}
};
using MpDecimal19 = MpDecimal<19>;
using MpDecimal34 = MpDecimal<34>;
using MpDecimal38 = MpDecimal<38>;
static_assert(DecomposableNumber<MpDecimal19>);
static_assert(DecomposableNumber<MpDecimal34>);
static_assert(DecomposableNumber<MpDecimal38>);
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,37 @@
#pragma once
// Slot for a 128-bit Number prototype.
//
// When <xrpl/basics/Number128.h> exists, CMake defines XRPL_BENCH_NUMBER128
// (see src/benchmarks/libxrpl/CMakeLists.txt) and the type is benchmarked and
// checked by every tool here with no further changes: xrpl.bench.number (single
// operations and ledger formulas) and xrpl.bench.number_diff (rounding
// correctness in all four modes). The header must provide `xrpl::Number128`
// satisfying DecomposableNumber (see NumberLike.h), plus:
//
// static constexpr unsigned kDigits; // mantissa precision, e.g. 34 or 38
//
// number_diff rounds exact results onto a plain kDigits-digit grid. Up to 38
// digits that is exact: 10^38 - 1 is below the signed 128-bit maximum, so
// there is no capped region like today's Number has above 2^63 - 1. If the
// prototype's grid differs, extend `Grid` in number_diff/main.cpp.
#ifdef XRPL_BENCH_NUMBER128
#include <xrpl/basics/Number128.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
// NOLINTNEXTLINE(cppcoreguidelines-macro-usage)
#define XRPL_BENCH_HAS_NUMBER128 1
namespace xrpl::number_bench {
static_assert(
DecomposableNumber<Number128>,
"Number128 must provide Number's interface; see NumberLike.h");
static_assert(Number128::kDigits <= 38, "Decomposed holds at most 38 digits");
} // namespace xrpl::number_bench
#endif

View File

@@ -0,0 +1,91 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <concepts>
#include <cstdint>
#include <string>
#include <string_view>
namespace xrpl::number_bench {
/**
* Sign, coefficient, and exponent of a finite decimal value:
* value = (negative ? -1 : 1) * coefficient * 10^exponent.
*
* The coefficient is 128 bits wide so the same struct can describe 34- and
* 38-digit types. Different libraries normalize to different precisions, so
* compare decompositions with `canonical()`, which strips trailing zeros.
*/
struct Decomposed
{
bool negative = false;
unsigned __int128 coefficient = 0;
int exponent = 0;
[[nodiscard]] constexpr Decomposed
canonical() const noexcept
{
if (coefficient == 0)
return {};
Decomposed result = *this;
while (result.coefficient % 10 == 0)
{
result.coefficient /= 10;
++result.exponent;
}
return result;
}
friend constexpr bool
operator==(Decomposed const&, Decomposed const&) = default;
};
[[nodiscard]] inline Decomposed
decompose(Number const& x) noexcept
{
auto const m = x.mantissa();
// Negating INT64_MIN is UB, but Number never produces it (see Number.h).
auto const magnitude = static_cast<std::uint64_t>(m < 0 ? -m : m);
return {.negative = m < 0, .coefficient = magnitude, .exponent = x.exponent()};
}
/**
* The subset of Number's interface that ledger code relies on. Benchmarks and
* the divergence harness are written against this concept, so any type
* satisfying it, including a future 128-bit Number, can be dropped in.
*/
template <class T>
concept NumberLike = std::copyable<T> && std::totally_ordered<T> &&
requires(T a, T b, std::int64_t mantissa, int exponent, unsigned n, Number::RoundingMode mode) {
T{mantissa};
T{mantissa, exponent};
{ a + b } -> std::same_as<T>;
{ a - b } -> std::same_as<T>;
{ a * b } -> std::same_as<T>;
{ a / b } -> std::same_as<T>;
{ -a } -> std::same_as<T>;
{ a += b } -> std::same_as<T&>;
{ a -= b } -> std::same_as<T&>;
{ a *= b } -> std::same_as<T&>;
{ a /= b } -> std::same_as<T&>;
static_cast<std::int64_t>(a);
{ to_string(a) } -> std::same_as<std::string>;
{ abs(a) } -> std::same_as<T>;
{ power(a, n) } -> std::same_as<T>;
{ root2(a) } -> std::same_as<T>;
{ T::setround(mode) } -> std::same_as<Number::RoundingMode>;
{ T::getround() } -> std::same_as<Number::RoundingMode>;
};
/**
* A NumberLike type whose exact value can be inspected digit by digit.
*/
template <class T>
concept DecomposableNumber = NumberLike<T> && requires(T a) {
{ decompose(a) } -> std::same_as<Decomposed>;
};
static_assert(DecomposableNumber<Number>);
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,250 @@
#pragma once
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <algorithm>
#include <cstddef>
#include <cstdint>
#include <limits>
#include <random>
#include <string_view>
#include <utility>
#include <vector>
namespace xrpl::number_bench {
/**
* Operand count per batch: small enough to stay in L1 for 32-byte types.
*/
constexpr std::size_t kBatchSize = 1024;
/**
* Raw operand: value = mantissa * 10^exponent.
*/
struct RawOperand
{
std::int64_t mantissa;
int exponent;
};
/**
* Operand distributions. Each one models a class of values the ledger works
* with, because cost depends on digit count and exponent alignment.
*/
enum class Dataset {
/**
* Full-width 19-digit mantissas in [10^18, 2^63-1], exponents in [-10, 10].
*/
Full,
/**
* IOU-like 16-digit mantissas in [10^15, 10^16-1], exponents in [-25, 5].
*/
Iou,
/**
* XRP drop amounts: integers with 1 to 17 digits (log-uniform), exponent 0.
*/
Drops,
/**
* MPT amounts: integers with 1 to 19 digits up to 2^63-1 (log-uniform), exponent 0.
*/
Mpt,
/**
* Rates in [1, 1.01) with full-width mantissas: the operand of interest and amortization
* math, and of long multiply/divide chains that must not overflow.
*/
Rate,
/**
* Non-integers below 10^18 (19-digit mantissas, exponents in [-18, -1]): rounding to int64.
*/
Fraction,
/**
* Exact halves, m + 0.5 with m < 10^17: the tie cases of rounding to int64.
*/
Half,
};
constexpr std::string_view
toString(Dataset d)
{
switch (d)
{
case Dataset::Full:
return "full";
case Dataset::Iou:
return "iou";
case Dataset::Drops:
return "drops";
case Dataset::Mpt:
return "mpt";
case Dataset::Rate:
return "rate";
case Dataset::Fraction:
return "fraction";
case Dataset::Half:
return "half";
}
return "unknown";
}
namespace detail {
/**
* Integer with a log-uniform digit count in [1, maxDigits], capped at `cap`.
*/
inline std::int64_t
logUniformInteger(std::mt19937_64& rng, int maxDigits, std::int64_t cap)
{
auto const digits = std::uniform_int_distribution<int>{1, maxDigits}(rng);
auto const lo = static_cast<std::int64_t>(kPowerOfTen[digits - 1]);
auto const hi =
digits >= 19 ? cap : std::min(cap, static_cast<std::int64_t>(kPowerOfTen[digits]) - 1);
return std::uniform_int_distribution<std::int64_t>{lo, hi}(rng);
}
} // namespace detail
/**
* Deterministic: the same dataset and seed always produce the same operands.
*/
inline std::vector<RawOperand>
makeRaw(Dataset d, std::uint64_t seed, std::size_t count = kBatchSize)
{
constexpr auto kMaxInt64 = std::numeric_limits<std::int64_t>::max();
constexpr auto kE18 = static_cast<std::int64_t>(kPowerOfTen[18]);
constexpr auto kE16 = static_cast<std::int64_t>(kPowerOfTen[16]);
constexpr auto kE15 = static_cast<std::int64_t>(kPowerOfTen[15]);
std::mt19937_64 rng{seed};
std::vector<RawOperand> result;
result.reserve(count);
for (std::size_t i = 0; i < count; ++i)
{
switch (d)
{
case Dataset::Full:
result.push_back(
{std::uniform_int_distribution<std::int64_t>{kE18, kMaxInt64}(rng),
std::uniform_int_distribution<int>{-10, 10}(rng)});
break;
case Dataset::Iou:
result.push_back(
{std::uniform_int_distribution<std::int64_t>{kE15, kE16 - 1}(rng),
std::uniform_int_distribution<int>{-25, 5}(rng)});
break;
case Dataset::Drops:
result.push_back({detail::logUniformInteger(rng, 17, kMaxInt64), 0});
break;
case Dataset::Mpt:
result.push_back({detail::logUniformInteger(rng, 19, kMaxInt64), 0});
break;
case Dataset::Rate:
// [10^18, 1.01 * 10^18) * 10^-18 = [1, 1.01)
result.push_back(
{std::uniform_int_distribution<std::int64_t>{
kE18, kE18 + (kE18 / 100) - 1}(rng),
-18});
break;
case Dataset::Fraction:
result.push_back(
{std::uniform_int_distribution<std::int64_t>{kE18, kMaxInt64}(rng),
std::uniform_int_distribution<int>{-18, -1}(rng)});
break;
case Dataset::Half:
result.push_back(
{(std::uniform_int_distribution<std::int64_t>{0, (kE18 / 10) - 1}(rng) * 10) +
5,
-1});
break;
}
}
return result;
}
template <NumberLike T>
std::vector<T>
materialize(std::vector<RawOperand> const& raw)
{
std::vector<T> result;
result.reserve(raw.size());
for (auto const& r : raw)
result.push_back(T{r.mantissa, r.exponent});
return result;
}
/**
* Pairs of full-width operands whose exponents differ by exactly `gap`, so
* addition must shift (and round away) `gap` digits of the smaller operand.
*/
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
makeExponentGapPairs(int gap, std::uint64_t seed, std::size_t count = kBatchSize)
{
auto lhs = makeRaw(Dataset::Full, seed, count);
auto rhs = makeRaw(Dataset::Full, seed + 1, count);
for (std::size_t i = 0; i < count; ++i)
rhs[i].exponent = lhs[i].exponent - gap;
return {std::move(lhs), std::move(rhs)};
}
/**
* Pairs of nearly equal full-width operands, so subtraction cancels most
* leading digits and the result must be renormalized by many places.
*/
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
makeCancellationPairs(std::uint64_t seed, std::size_t count = kBatchSize)
{
auto lhs = makeRaw(Dataset::Full, seed, count);
std::vector<RawOperand> rhs;
rhs.reserve(count);
std::mt19937_64 rng{seed + 1};
std::uniform_int_distribution<std::int64_t> delta{1, 1000};
for (auto const& l : lhs)
rhs.push_back({l.mantissa - delta(rng), l.exponent});
return {std::move(lhs), std::move(rhs)};
}
/**
* Pairs whose exact sum is exactly halfway between two adjacent 19-digit
* values: a = m * 10^e, b = 5 * 10^(e-1). Exercises round-half-even.
*/
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
makeTiePairs(std::uint64_t seed, std::size_t count = kBatchSize)
{
auto lhs = makeRaw(Dataset::Full, seed, count);
std::vector<RawOperand> rhs;
rhs.reserve(count);
for (auto const& l : lhs)
rhs.push_back({5, l.exponent - 1});
return {std::move(lhs), std::move(rhs)};
}
/**
* Pairs whose sum straddles 2^63-1 (Number::kMaxRep), where Number's
* representable values switch from a step of 1 to a step of 10. The addend has
* one more decimal place than the base, so most sums fall between
* representable values and must be rounded; sums are within ±200 of the cap,
* so a fair share land in (2^63-1, 2^63), the interval where LargeLegacy's
* known cusp-rounding error shows.
*/
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
makeCuspPairs(std::uint64_t seed, std::size_t count = kBatchSize)
{
constexpr auto kMaxInt64 = std::numeric_limits<std::int64_t>::max();
std::mt19937_64 rng{seed};
std::uniform_int_distribution<std::int64_t> base{kMaxInt64 - 100, kMaxInt64};
std::uniform_int_distribution<std::int64_t> addend{1, 2'000};
std::uniform_int_distribution<int> exponent{-10, 10};
std::pair<std::vector<RawOperand>, std::vector<RawOperand>> result;
result.first.reserve(count);
result.second.reserve(count);
for (std::size_t i = 0; i < count; ++i)
{
auto const e = exponent(rng);
result.first.push_back({base(rng), e});
result.second.push_back({addend(rng), e - 1});
}
return result;
}
} // namespace xrpl::number_bench

View File

@@ -0,0 +1,309 @@
#!/usr/bin/env python3
"""Render the figures in docs/NumberDecimalBenchmark.md from benchmark JSON.
Usage:
plot_results.py CORE_JSON KERNELS_JSON OUT_DIR
CORE_JSON is Google Benchmark JSON from xrpl.bench.number filtered to
'^(add|mul|div)/full/'; KERNELS_JSON from '^kernel/' with mpdecimal enabled
(for the accuracy counters). Both need --benchmark_repetitions so medians are
available. Writes self-contained SVG files with light and dark variants
(prefers-color-scheme) and no dependencies beyond the standard library.
"""
from __future__ import annotations
import json
import sys
from dataclasses import dataclass
from pathlib import Path
from typing import Callable
from xml.sax.saxutils import escape
# Categorical slots 1-3 of the reference palette (validated all-pairs, light
# and dark). Color encodes the type family; every bar is also labeled.
FAMILIES = {
"number": ("Number today (19 digits)", "#2a78d6", "#3987e5"),
"ieee64": ("16-digit types", "#eb6834", "#d95926"),
"wide": ("34- and 38-digit types", "#1baf7a", "#199e70"),
}
# (benchmark subject, row label, family)
CORE_ROWS = [
("Number.Large330", "Number (Large330)", "number"),
("BoostDecimal64", "Boost decimal64", "ieee64"),
("BoostDecimalFast64", "Boost decimal_fast64", "ieee64"),
("IntelBid64", "Intel BID64", "ieee64"),
("BoostDecimal128", "Boost decimal128", "wide"),
("BoostDecimalFast128", "Boost decimal_fast128", "wide"),
("IntelBid128", "Intel BID128", "wide"),
("MpDecimal34", "mpdecimal 34", "wide"),
("MpDecimal38", "mpdecimal 38", "wide"),
]
KERNEL_ROWS = [
("Number.Large330", "Number (Large330)", "number"),
("BoostDecimal64", "Boost decimal64", "ieee64"),
("IntelBid64", "Intel BID64", "ieee64"),
("BoostDecimal128", "Boost decimal128", "wide"),
("IntelBid128", "Intel BID128", "wide"),
("MpDecimal34", "mpdecimal 34", "wide"),
("MpDecimal38", "mpdecimal 38", "wide"),
]
FONT = "system-ui, -apple-system, sans-serif"
ROW_H = 24
BAR_H = 12
LABEL_W = 150
PANEL_W = 200
PANEL_GAP = 28
TOP = 78
BOTTOM = 30
@dataclass
class Panel:
title: str
unit: str
values: dict[str, float]
fmt: Callable[[float], str]
# Panels with the same group share an x scale; use it only for panels
# with the same unit, so bar lengths are comparable.
group: str = ""
def medians(path: str) -> dict[str, dict]:
"""Map run_name to its median aggregate."""
data = json.loads(Path(path).read_text())
return {
b["run_name"]: b
for b in data["benchmarks"]
if b.get("aggregate_name") == "median"
}
def ns_per_item(b: dict) -> float:
return 1e9 / b["items_per_second"]
def nice_max(v: float) -> tuple[float, list[float]]:
"""Axis maximum and ticks: 4-5 round steps covering v."""
for step in [
1,
2,
2.5,
5,
10,
20,
25,
50,
100,
200,
250,
500,
1000,
2000,
2500,
5000,
10000,
20000,
25000,
50000,
]:
if v / step <= 5:
top = step * -(-v // step)
return top, [step * i for i in range(int(top / step) + 1)]
return v, [0, v]
def bar_path(x: float, y: float, w: float, h: float, r: float = 4) -> str:
"""Bar anchored at the baseline (left) with a rounded data end (right)."""
r = min(r, w, h / 2)
return (
f"M{x:.1f},{y:.1f} H{x + w - r:.1f} "
f"Q{x + w:.1f},{y:.1f} {x + w:.1f},{y + r:.1f} "
f"V{y + h - r:.1f} Q{x + w:.1f},{y + h:.1f} {x + w - r:.1f},{y + h:.1f} "
f"H{x:.1f} Z"
)
def render(
title: str, rows: list[tuple[str, str, str]], panels: list[Panel], out: Path
) -> None:
"""Horizontal bar panels sharing one column of row labels."""
width = LABEL_W + len(panels) * (PANEL_W + PANEL_GAP)
height = TOP + len(rows) * ROW_H + BOTTOM
css_light = "".join(
f".f-{k}{{fill:{light}}}" for k, (_, light, _) in FAMILIES.items()
)
css_dark = "".join(f".f-{k}{{fill:{dark}}}" for k, (_, _, dark) in FAMILIES.items())
out_lines = [
f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 {width} {height}" '
f'width="{width}" height="{height}" role="img" '
f'aria-label="{escape(title)}">',
"<style>",
f"text{{font-family:{FONT};font-size:12px}}",
".bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}"
".grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}"
+ css_light,
"@media (prefers-color-scheme: dark){"
".bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}"
".grid{stroke:#383835}.base{stroke:#5e5d58}" + css_dark + "}",
"</style>",
f'<rect class="bg" width="{width}" height="{height}"/>',
f'<text class="t1" x="0" y="18" font-size="15" font-weight="600">'
f"{escape(title)}</text>",
]
# Legend: always present for more than one family.
lx = 0.0
for key, (label, _, _) in FAMILIES.items():
out_lines.append(
f'<rect class="f-{key}" x="{lx}" y="32" width="12" height="12" rx="3"/>'
)
out_lines.append(
f'<text class="t2" x="{lx + 18}" y="42">{escape(label)}</text>'
)
lx += 18 + 7.2 * len(label) + 24
for i, (_, label, _) in enumerate(rows):
y = TOP + i * ROW_H + ROW_H / 2 + 4
out_lines.append(f'<text class="t1" x="0" y="{y:.1f}">{escape(label)}</text>')
for p_index, panel in enumerate(panels):
x0 = LABEL_W + p_index * (PANEL_W + PANEL_GAP)
vmax = max(
p.values.get(s, 0)
for p in panels
if p is panel or (panel.group and p.group == panel.group)
for s, _, _ in rows
)
top, ticks = nice_max(vmax * 1.08)
scale = (PANEL_W - 40) / top
out_lines.append(
f'<text class="t1" x="{x0}" y="{TOP - 14}" font-weight="600">'
f"{escape(panel.title)}</text>"
)
y_end = TOP + len(rows) * ROW_H
for t in ticks:
gx = x0 + t * scale
out_lines.append(
f'<line class="grid" x1="{gx:.1f}" y1="{TOP}" x2="{gx:.1f}" y2="{y_end}"/>'
)
out_lines.append(
f'<text class="t2" x="{gx:.1f}" y="{y_end + 16}" text-anchor="middle" '
f'font-size="11">{t:g}</text>'
)
out_lines.append(
f'<text class="t2" x="{x0 + PANEL_W - 40:.1f}" y="{y_end + 28}" '
f'text-anchor="end" font-size="11">{escape(panel.unit)}</text>'
)
for i, (subject, label, family) in enumerate(rows):
if subject not in panel.values:
continue
v = panel.values[subject]
y = TOP + i * ROW_H + (ROW_H - BAR_H) / 2
w = max(v * scale, 1.0)
out_lines.append(
f'<path class="f-{family}" d="{bar_path(x0, y, w, BAR_H)}">'
f"<title>{escape(label)}: {panel.fmt(v)} {escape(panel.unit)}</title></path>"
)
out_lines.append(
f'<text class="t2" x="{x0 + w + 5:.1f}" y="{y + BAR_H - 2:.1f}" '
f'font-size="11">{panel.fmt(v)}</text>'
)
out_lines.append(
f'<line class="base" x1="{x0}" y1="{TOP}" x2="{x0}" y2="{y_end}"/>'
)
out_lines.append("</svg>")
out.write_text("\n".join(out_lines) + "\n")
def fmt_ns(v: float) -> str:
return f"{v:.0f}" if v >= 10 else f"{v:.1f}"
def fmt_us(v: float) -> str:
return f"{v:.0f}" if v >= 10 else f"{v:.1f}"
def fmt_digits(v: float) -> str:
return f"{v:.1f}"
def main(argv: list[str]) -> int:
if len(argv) != 4:
print(__doc__, file=sys.stderr)
return 2
core = medians(argv[1])
kernels = medians(argv[2])
out_dir = Path(argv[3])
out_dir.mkdir(parents=True, exist_ok=True)
core_panels = [
Panel(
op_title,
"ns per operation",
{
s: ns_per_item(core[f"{op}/full/{s}"])
for s, _, _ in CORE_ROWS
if f"{op}/full/{s}" in core
},
fmt_ns,
group="ns",
)
for op, op_title in [("add", "Add"), ("mul", "Multiply"), ("div", "Divide")]
]
render(
"Cost of one operation (full-width operands, lower is faster)",
CORE_ROWS,
core_panels,
out_dir / "number-core-ops.svg",
)
def kernel(name: str) -> dict[str, dict]:
prefix = f"kernel/{name}/"
return {k[len(prefix) :]: v for k, v in kernels.items() if k.startswith(prefix)}
for name, title, unit, divisor, fmt in [
("amm_swap_in", "AMM swap-in", "ns per call", 1.0, fmt_ns),
(
"loan_amortize360",
"360-payment amortization",
"µs per schedule",
1e3,
fmt_us,
),
]:
k = kernel(name)
render(
f"{title}: correct digits (higher is better) and cost (lower is faster)",
KERNEL_ROWS,
[
Panel(
"Correct digits, worst case",
"significant digits",
{s: b["digits_min"] for s, b in k.items()},
fmt_digits,
group="digits",
),
Panel(
"Correct digits, median",
"significant digits",
{s: b["digits_median"] for s, b in k.items()},
fmt_digits,
group="digits",
),
Panel(
"Cost",
unit,
{s: ns_per_item(b) / divisor for s, b in k.items()},
fmt,
),
],
out_dir / f"number-{name.replace('_', '-')}.svg",
)
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv))

View File

@@ -0,0 +1,555 @@
// number_diff: measures how far Number's results are from correctly rounded
// results. See docs/NumberDecimalBenchmark.md, "Divergence harness".
//
// For every operation, the operands are converted exactly into a 96-digit
// mpdecimal value and the operation is evaluated there (correctly rounded at
// 96 digits). That near-exact result is then rounded onto the subject type's
// own grid of representable values in the active rounding mode, and compared
// with what the subject actually returned.
//
// Rounding a 96-digit result again creates a false tie only if digits 20-96
// of a non-terminating result happen to round to exactly 5000...0, which
// will not occur at these sample sizes.
//
// Usage: xrpl.bench.number_diff [--samples N] [--csv]
#include <xrpl/basics/Number.h>
#include <benchmarks/libxrpl/number/BoostDecimal.h>
#include <benchmarks/libxrpl/number/MpDecimal.h>
#include <benchmarks/libxrpl/number/Number128.h>
#include <benchmarks/libxrpl/number/NumberLike.h>
#include <benchmarks/libxrpl/number/Operands.h>
#ifdef XRPL_BENCH_INTEL_DFP
#include <benchmarks/libxrpl/number/IntelBid.h>
#endif
#include <algorithm>
#include <array>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <exception>
#include <format>
#include <functional>
#include <iostream>
#include <limits>
#include <optional>
#include <string>
#include <string_view>
#include <utility>
#include <vector>
namespace xrpl::number_bench {
namespace {
using Exact = MpDecimal<96>;
constexpr std::uint64_t kSeedLhs = 0xd1ff'0001;
constexpr std::uint64_t kSeedRhs = 0xd1ff'0002;
/**
* The set of values a type can represent, as far as rounding is concerned.
* `int64Cap` models Number's large scales, whose mantissas above 2^63-1 must
* end in 0 (see Number.h, "External Interface").
*/
struct Grid
{
unsigned digits;
bool int64Cap = false;
};
/**
* Integers: the grid of conversions to int64.
*/
struct IntegerGrid
{
};
// ---- Exact arithmetic helpers ----
Exact
fromSubject(DecomposableNumber auto const& x)
{
return Exact::fromDecomposed(decompose(x));
}
/**
* Rounds a non-negative value to `digits` significant digits, toward (down) or away from (up)
* zero.
*/
Exact
roundMagnitude(Exact x, unsigned digits, bool up)
{
mpd_context_t ctx;
mpd_maxcontext(&ctx);
ctx.prec = digits;
ctx.round = up ? MPD_ROUND_UP : MPD_ROUND_DOWN;
ctx.traps = 0;
std::uint32_t status = 0;
mpd_qfinalize(x.get(), &ctx, &status);
return x;
}
/**
* Rounds a non-negative value to an integer, toward (down) or away from (up) zero.
*/
Exact
roundMagnitudeToInteger(Exact const& x, bool up)
{
mpd_context_t ctx;
mpd_maxcontext(&ctx);
ctx.round = up ? MPD_ROUND_UP : MPD_ROUND_DOWN;
ctx.traps = 0;
std::uint32_t status = 0;
Exact result;
mpd_qround_to_int(result.get(), x.get(), &ctx, &status);
return result;
}
bool
isZero(Exact const& x)
{
return mpd_iszero(x.get()) != 0;
}
bool
isNegative(Exact const& x)
{
return mpd_isnegative(x.get()) != 0 && !isZero(x);
}
/**
* True if q (a non-negative integer) is even.
*/
bool
isEvenInteger(Exact const& q)
{
auto const s = to_string(q);
auto const pos = s.find_first_of("Ee");
// Positive exponent in scientific notation means trailing zeros.
if (pos != std::string::npos && s[pos + 1] != '-')
return true;
auto const digits = s.substr(0, pos);
return ((digits.back() - '0') % 2) == 0;
}
// ---- Candidate representable values around an exact result ----
/**
* The representable magnitudes immediately at or below and at or above |x|.
*/
struct Bracket
{
Exact lo;
Exact hi;
/**
* Grid spacing at lo: lo / spacing is lo's integer coefficient.
*/
Exact loSpacing;
/**
* lo is 2^63-1 and hi is its successor 2^63+3: the two candidates straddle Number's cap.
*/
bool straddlesCap = false;
};
Bracket
bracket(Exact const& magnitude, Grid grid)
{
auto const adjusted = static_cast<int>(mpd_adjexp(magnitude.get()));
auto lo = roundMagnitude(magnitude, grid.digits, false);
auto hi = roundMagnitude(magnitude, grid.digits, true);
Exact loSpacing{1, adjusted - static_cast<int>(grid.digits) + 1};
bool straddlesCap = false;
if (grid.int64Cap)
{
// Above cap, only every tenth 19-digit value is representable.
Exact const cap{static_cast<std::int64_t>(Number::kMaxRep), adjusted - 18};
if (hi > cap)
hi = roundMagnitude(magnitude, 18, true);
if (lo > cap)
{
auto const lo18 = roundMagnitude(magnitude, 18, false);
if (lo18 < cap)
{
lo = cap;
straddlesCap = true;
}
else
{
lo = lo18;
loSpacing = Exact{1, adjusted - 17};
}
}
}
return {std::move(lo), std::move(hi), std::move(loSpacing), straddlesCap};
}
/**
* The correctly rounded magnitude, or nullopt if two candidates are equally valid (a cusp tie).
*/
std::optional<Exact>
choose(Exact const& magnitude, Bracket const& b, bool negative, Number::RoundingMode mode)
{
using enum Number::RoundingMode;
if (b.lo == b.hi)
return b.lo;
switch (mode)
{
case TowardsZero:
return b.lo;
case Downward:
return negative ? b.hi : b.lo;
case Upward:
return negative ? b.lo : b.hi;
case ToNearest:
break;
}
auto const below = magnitude - b.lo;
auto const above = b.hi - magnitude;
if (below < above)
return b.lo;
if (above < below)
return b.hi;
// Tie at the cap: 2^63-1 and 2^63+3 are both odd, so "even" is undefined.
if (b.straddlesCap)
return std::nullopt;
return isEvenInteger(b.lo / b.loSpacing) ? b.lo : b.hi;
}
// ---- Tallies ----
struct Tally
{
std::size_t samples = 0;
/**
* Exactly the correctly rounded result.
*/
std::size_t correct = 0;
/**
* The other neighbor of the exact result: rounded in the wrong direction.
*/
std::size_t wrongNeighbor = 0;
/**
* Neither neighbor: off by more than one step.
*/
std::size_t worse = 0;
/**
* Returned zero for a non-zero result.
*/
std::size_t flushed = 0;
/**
* Threw instead of returning.
*/
std::size_t threw = 0;
/**
* Largest |result - exact| / local grid spacing.
*/
double maxError = 0;
};
void
classify(Tally& t, Exact const& result, Exact const& exact, Grid grid, Number::RoundingMode mode)
{
if (isZero(exact))
{
++(isZero(result) ? t.correct : t.worse);
return;
}
if (isZero(result))
{
++t.flushed;
return;
}
bool const negative = isNegative(exact);
auto const magnitude = abs(exact);
auto const b = bracket(magnitude, grid);
auto const resultMagnitude = abs(result);
auto const chosen = choose(magnitude, b, negative, mode);
bool const sameSign = isNegative(result) == negative;
bool const isLo = sameSign && resultMagnitude == b.lo;
bool const isHi = sameSign && resultMagnitude == b.hi;
if (sameSign && (chosen ? resultMagnitude == *chosen : (isLo || isHi)))
++t.correct;
else if (isLo || isHi)
++t.wrongNeighbor;
else
++t.worse;
if (b.lo != b.hi)
{
auto const error = abs(result - exact) / (b.hi - b.lo);
t.maxError = std::max(t.maxError, std::stod(to_string(error)));
}
}
void
classifyInteger(Tally& t, std::int64_t result, Exact const& exact, Number::RoundingMode mode)
{
bool const negative = isNegative(exact);
auto const magnitude = abs(exact);
auto const lo = roundMagnitudeToInteger(magnitude, false);
auto const hi = roundMagnitudeToInteger(magnitude, true);
Bracket const b{lo, hi, Exact{1}};
auto const chosen = *choose(magnitude, b, negative, mode);
Exact const r{result};
auto const resultMagnitude = abs(r);
bool const sameSign = result == 0 || (result < 0) == negative;
if (sameSign && resultMagnitude == chosen)
++t.correct;
else if (sameSign && (resultMagnitude == lo || resultMagnitude == hi))
++t.wrongNeighbor;
else
++t.worse;
t.maxError = std::max(t.maxError, std::stod(to_string(abs(r - exact))));
}
// ---- Running cells ----
struct Row
{
std::string subject;
Number::RoundingMode mode;
std::string op;
std::string data;
Tally tally;
};
template <DecomposableNumber T, class BinaryOp>
Tally
runBinary(
std::vector<RawOperand> const& lhs,
std::vector<RawOperand> const& rhs,
BinaryOp op,
Grid grid,
Number::RoundingMode mode)
{
Tally t;
for (std::size_t i = 0; i < lhs.size(); ++i)
{
T const a{lhs[i].mantissa, lhs[i].exponent};
T const b{rhs[i].mantissa, rhs[i].exponent};
auto const exact = op(fromSubject(a), fromSubject(b));
++t.samples;
try
{
classify(t, fromSubject(T{op(a, b)}), exact, grid, mode);
}
catch (std::exception const&)
{
++t.threw;
}
}
return t;
}
template <DecomposableNumber T, class UnaryOp>
Tally
runUnary(std::vector<RawOperand> const& operands, UnaryOp op, Grid grid, Number::RoundingMode mode)
{
Tally t;
for (auto const& raw : operands)
{
T const a{raw.mantissa, raw.exponent};
auto const exact = op(fromSubject(a));
++t.samples;
try
{
classify(t, fromSubject(T{op(a)}), exact, grid, mode);
}
catch (std::exception const&)
{
++t.threw;
}
}
return t;
}
template <DecomposableNumber T>
Tally
runToInt64(std::vector<RawOperand> const& operands, Number::RoundingMode mode)
{
Tally t;
for (auto const& raw : operands)
{
T const a{raw.mantissa, raw.exponent};
auto const exact = fromSubject(a);
++t.samples;
try
{
classifyInteger(t, static_cast<std::int64_t>(a), exact, mode);
}
catch (std::exception const&)
{
++t.threw;
}
}
return t;
}
/**
* Runs every operation and dataset for one subject type and rounding mode.
*/
template <DecomposableNumber T>
void
runSubject(
std::string_view subject,
Grid grid,
Number::RoundingMode mode,
std::size_t samples,
std::vector<Row>& rows)
{
auto const saved = T::setround(mode);
auto const record = [&](std::string op, Dataset d, Tally t) {
rows.push_back({std::string{subject}, mode, std::move(op), std::string{toString(d)}, t});
};
auto const recordPairs = [&](std::string op, std::string data, Tally t) {
rows.push_back({std::string{subject}, mode, std::move(op), std::move(data), t});
};
auto const plus = [](auto const& x, auto const& y) { return x + y; };
auto const minus = [](auto const& x, auto const& y) { return x - y; };
auto const times = [](auto const& x, auto const& y) { return x * y; };
auto const divide = [](auto const& x, auto const& y) { return x / y; };
for (auto const d : {Dataset::Full, Dataset::Iou, Dataset::Mpt})
{
auto const lhs = makeRaw(d, kSeedLhs, samples);
auto const rhs = makeRaw(d, kSeedRhs, samples);
record("add", d, runBinary<T>(lhs, rhs, plus, grid, mode));
record("sub", d, runBinary<T>(lhs, rhs, minus, grid, mode));
record("mul", d, runBinary<T>(lhs, rhs, times, grid, mode));
record("div", d, runBinary<T>(lhs, rhs, divide, grid, mode));
record("root2", d, runUnary<T>(lhs, [](auto const& x) { return root2(x); }, grid, mode));
}
for (int const gap : {1, 5, 18})
{
auto const [lhs, rhs] = makeExponentGapPairs(gap, kSeedLhs, samples);
recordPairs("add", std::format("gap{}", gap), runBinary<T>(lhs, rhs, plus, grid, mode));
recordPairs("sub", std::format("gap{}", gap), runBinary<T>(lhs, rhs, minus, grid, mode));
}
{
auto const [lhs, rhs] = makeCancellationPairs(kSeedLhs, samples);
recordPairs("sub", "cancel", runBinary<T>(lhs, rhs, minus, grid, mode));
}
{
auto const [lhs, rhs] = makeTiePairs(kSeedLhs, samples);
recordPairs("add", "tie", runBinary<T>(lhs, rhs, plus, grid, mode));
}
{
auto const [lhs, rhs] = makeCuspPairs(kSeedLhs, samples);
recordPairs("add", "cusp", runBinary<T>(lhs, rhs, plus, grid, mode));
}
auto const rates = makeRaw(Dataset::Rate, kSeedLhs, samples);
for (unsigned const n : {12u, 360u})
{
record(
std::format("power{}", n),
Dataset::Rate,
runUnary<T>(rates, [n](auto const& x) { return power(x, n); }, grid, mode));
}
for (auto const d : {Dataset::Fraction, Dataset::Half})
record("to_int64", d, runToInt64<T>(makeRaw(d, kSeedLhs, samples), mode));
T::setround(saved);
}
void
print(std::vector<Row> const& rows, bool csv)
{
if (csv)
std::cout << "subject,mode,op,data,samples,correct,wrong_neighbor,worse,flushed,threw,max_"
"error\n";
else
std::cout << "| subject | mode | op | data | samples | correct | wrong neighbor | worse | "
"flushed | threw | max error |\n"
"|---|---|---|---|---:|---:|---:|---:|---:|---:|---:|\n";
for (auto const& r : rows)
{
auto const& t = r.tally;
auto const mode = to_string(r.mode);
auto const fmt = csv ? "{},{},{},{},{},{},{},{},{},{},{:.4g}\n"
: "| {} | {} | {} | {} | {} | {} | {} | {} | {} | {} | {:.4g} |\n";
std::cout << std::vformat(
fmt,
std::make_format_args(
r.subject,
mode,
r.op,
r.data,
t.samples,
t.correct,
t.wrongNeighbor,
t.worse,
t.flushed,
t.threw,
t.maxError));
}
}
int
run(int argc, char** argv)
{
std::size_t samples = 10'000;
bool csv = false;
for (int i = 1; i < argc; ++i)
{
std::string_view const arg = argv[i];
if (arg == "--samples" && i + 1 < argc)
samples = std::strtoull(argv[++i], nullptr, 10);
else if (arg == "--csv")
csv = true;
else
{
std::cerr << "usage: " << argv[0] << " [--samples N] [--csv]\n";
return 2;
}
}
using enum Number::RoundingMode;
using enum MantissaRange::MantissaScale;
constexpr auto kModes = std::to_array({ToNearest, TowardsZero, Downward, Upward});
std::vector<Row> rows;
for (auto const mode : kModes)
{
for (auto const scale : {Small, LargeLegacy, Large320, Large330})
{
NumberMantissaScaleGuard const guard{scale};
auto const grid =
scale == Small ? Grid{.digits = 16} : Grid{.digits = 19, .int64Cap = true};
runSubject<Number>("Number." + to_string(scale), grid, mode, samples, rows);
}
// Validate the oracle and the grid logic against IEEE implementations,
// which must be correctly rounded for + - * / and sqrt.
runSubject<BoostDecimal64>("BoostDecimal64", Grid{.digits = 16}, mode, samples, rows);
runSubject<BoostDecimalFast64>(
"BoostDecimalFast64", Grid{.digits = 16}, mode, samples, rows);
runSubject<BoostDecimal128>("BoostDecimal128", Grid{.digits = 34}, mode, samples, rows);
runSubject<BoostDecimalFast128>(
"BoostDecimalFast128", Grid{.digits = 34}, mode, samples, rows);
#ifdef XRPL_BENCH_INTEL_DFP
runSubject<IntelBid64>("IntelBid64", Grid{.digits = 16}, mode, samples, rows);
runSubject<IntelBid128>("IntelBid128", Grid{.digits = 34}, mode, samples, rows);
#endif
#ifdef XRPL_BENCH_HAS_NUMBER128
runSubject<Number128>("Number128", Grid{.digits = Number128::kDigits}, mode, samples, rows);
#endif
}
print(rows, csv);
return 0;
}
} // namespace
} // namespace xrpl::number_bench
int
main(int argc, char** argv)
{
return xrpl::number_bench::run(argc, argv);
}