mirror of
https://github.com/XRPLF/rippled.git
synced 2026-10-10 21:58:03 +00:00
Compare commits
1 Commits
pratik/ote
...
tapanito/n
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ae9cb05521 |
@@ -23,6 +23,8 @@ ignoreRegExpList:
|
||||
- /\b(XRPL|BEAST)_[A-Z_0-9]+_H+/g # include guards
|
||||
- /::[a-z:_]+/g # things from other namespaces
|
||||
- /\blib[a-z]+/g # libraries
|
||||
- /\b(mpd|MPD)_[A-Za-z_0-9]+/g # libmpdec (mpdecimal) API
|
||||
- /\bbid(64|128)_[a-z_0-9]+/g # Intel decimal library API
|
||||
- /\b[0-9]{4}-[0-9]{2}-[0-9]{2}[,:][A-Za-zÀ-ÖØ-öø-ÿ.\s]+/g # copyright dates
|
||||
- /\b[0-9]{4}[,:]?\s*[A-Za-zÀ-ÖØ-öø-ÿ.\s]+/g # copyright years
|
||||
- /\[[A-Za-z0-9-]+\]\(https:\/\/github.com\/[A-Za-z0-9-]+\)/g # Github usernames
|
||||
@@ -44,6 +46,7 @@ suggestWords:
|
||||
- synch->sync
|
||||
words:
|
||||
- abempty
|
||||
- allcr
|
||||
- AMMID
|
||||
- AMMMPT
|
||||
- AMMMPToken
|
||||
@@ -184,6 +187,7 @@ words:
|
||||
- Merkle
|
||||
- misprediction
|
||||
- missingok
|
||||
- mpdecimal
|
||||
- mptbalance
|
||||
- MPTDEX
|
||||
- mptflags
|
||||
|
||||
@@ -31,6 +31,12 @@ if(tests)
|
||||
endif()
|
||||
|
||||
option(benchmark "Build benchmarks" ON)
|
||||
# Opt-in: the Number divergence harness (xrpl.bench.number_diff) and the
|
||||
# mpdecimal benchmark subjects. Requires the `bench_mpdecimal` Conan option.
|
||||
option(bench_mpdecimal "Build benchmarks that need mpdecimal" OFF)
|
||||
# Opt-in: Intel Decimal Floating-Point Math Library benchmark subjects. The
|
||||
# library is downloaded and built at configure/build time (benchmarks only).
|
||||
option(bench_intel_dfp "Build benchmarks that need Intel's decimal library" OFF)
|
||||
|
||||
# When OFF, the crates directory is not added to the build at all: no Rust
|
||||
# toolchain is required, no cxxbridge bindings are generated, and the C++ tests
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
"openssl/3.6.3#f806de8933e3bf6f01016c6a888cee2e%1783945160.863288",
|
||||
"nudb/2.0.9#11149c73f8f2baff9a0198fe25971fc7%1782392402.297166",
|
||||
"mpt-crypto/1.0.2#b313cef0c1a493eb970ad185b2e9bab7%1784285108.866483",
|
||||
"mpdecimal/4.0.0#2fe1a7688d6c8b4bf209ea23e8b0c6ba%1736849696.519",
|
||||
"lz4/1.10.0#982d9b673900f665a1da109e09c17cab%1782392402.164188",
|
||||
"libiconv/1.17#9923bc6dc6f106646d6967e0039a5ada%1782392792.775744",
|
||||
"libbacktrace/cci.20210118#a7691bfccd8caaf66309df196790a5a1%1782392402.420732",
|
||||
|
||||
@@ -25,12 +25,15 @@ rm -f conan.lock
|
||||
conan lock create . \
|
||||
--options '&:jemalloc=True' \
|
||||
--options '&:rocksdb=True' \
|
||||
--options '&:bench_mpdecimal=True' \
|
||||
--profile:all=conan/lockfile/linux.profile
|
||||
conan lock create . \
|
||||
--options '&:jemalloc=True' \
|
||||
--options '&:rocksdb=True' \
|
||||
--options '&:bench_mpdecimal=True' \
|
||||
--profile:all=conan/lockfile/macos.profile
|
||||
conan lock create . \
|
||||
--options '&:jemalloc=True' \
|
||||
--options '&:rocksdb=True' \
|
||||
--options '&:bench_mpdecimal=True' \
|
||||
--profile:all=conan/lockfile/windows.profile
|
||||
|
||||
@@ -16,6 +16,7 @@ class Xrpl(ConanFile):
|
||||
options = {
|
||||
"assertions": [True, False],
|
||||
"benchmark": [True, False],
|
||||
"bench_mpdecimal": [True, False],
|
||||
"coverage": [True, False],
|
||||
"fPIC": [True, False],
|
||||
"jemalloc": [True, False],
|
||||
@@ -51,6 +52,7 @@ class Xrpl(ConanFile):
|
||||
default_options = {
|
||||
"assertions": False,
|
||||
"benchmark": True,
|
||||
"bench_mpdecimal": False,
|
||||
"coverage": False,
|
||||
"fPIC": True,
|
||||
"jemalloc": False,
|
||||
@@ -66,6 +68,7 @@ class Xrpl(ConanFile):
|
||||
"boost/*:without_coroutine2": False,
|
||||
"date/*:header_only": True,
|
||||
"ed25519/*:shared": False,
|
||||
"mpdecimal/*:cxx": False,
|
||||
"grpc/*:shared": False,
|
||||
"grpc/*:secure": True,
|
||||
"grpc/*:codegen": True,
|
||||
@@ -137,6 +140,8 @@ class Xrpl(ConanFile):
|
||||
def requirements(self):
|
||||
if self.options.benchmark:
|
||||
self.requires("benchmark/1.9.5")
|
||||
if self.options.bench_mpdecimal:
|
||||
self.requires("mpdecimal/4.0.0")
|
||||
self.requires("boost/1.91.0", force=True, transitive_headers=True)
|
||||
self.requires("date/3.0.4", transitive_headers=True)
|
||||
if self.options.jemalloc:
|
||||
@@ -172,6 +177,7 @@ class Xrpl(ConanFile):
|
||||
tc = CMakeToolchain(self)
|
||||
tc.variables["tests"] = self.options.tests
|
||||
tc.variables["benchmark"] = self.options.benchmark
|
||||
tc.variables["bench_mpdecimal"] = self.options.bench_mpdecimal
|
||||
tc.variables["assert"] = self.options.assertions
|
||||
tc.variables["coverage"] = self.options.coverage
|
||||
tc.variables["jemalloc"] = self.options.jemalloc
|
||||
|
||||
409
docs/NumberDecimalBenchmark.md
Normal file
409
docs/NumberDecimalBenchmark.md
Normal file
@@ -0,0 +1,409 @@
|
||||
# Number vs. decimal floating-point libraries: benchmark plan
|
||||
|
||||
Status: complete (phases 1-5). Goal: before extending `xrpl::Number` to a 128-bit
|
||||
mantissa, (1) measure how `Number` performs against established base-10
|
||||
floating-point libraries, and (2) quantify where `Number` diverges from
|
||||
IEEE 754-2008/2019 decimal arithmetic.
|
||||
|
||||
Target precisions for a future 128-bit `Number`: **both 34 digits** (matches
|
||||
IEEE decimal128) **and 38 digits** (largest `10^k - 1` that fits in a
|
||||
`uint128`, the direct analogue of today's 19-digit-in-`uint64` design).
|
||||
|
||||
## Summary
|
||||
|
||||
Measured on one x86-64 machine with GCC 15.2; see "Findings so far" for
|
||||
method, caveats, and every number.
|
||||
|
||||
**Correctness.** Today's `Number` (`Large330`) is correctly rounded for
|
||||
`+ - * /` and conversion to int64 in all four rounding modes: 9.2 million of
|
||||
9.2 million samples, including the int64-cap cases older scales got wrong.
|
||||
Its `root2` (up to ~5 ULP) and `power` (up to ~2800 ULP for n = 360) are not
|
||||
correctly rounded. No IEEE library tested has an accurate integer `power`
|
||||
either, and loan payments lose only ~4 digits to it at every precision.
|
||||
Boost.Decimal 1.91 has several rounding defects (see "Validating the
|
||||
oracle"); Intel's library had none in `+ - * /`, `sqrt`, or int64
|
||||
conversion.
|
||||
|
||||
**Divergence from IEEE 754.** `Number` is always normalized (no cohorts),
|
||||
has no infinities, NaNs, signed zero, subnormals, or status flags, throws on
|
||||
overflow and division by zero, flushes underflow to zero, has a wider
|
||||
exponent range, and only ~18.96 digits of uniform precision because the
|
||||
external mantissa is capped at int64. See "How Number diverges from IEEE 754
|
||||
decimal".
|
||||
|
||||
**Speed.** Today's `Number` is 6-25x slower than 64-bit IEEE types at `+`
|
||||
and `*`, and slower than Intel's 34-digit BID128 at everything except
|
||||
comparison: 87 vs 12 ns for `+`, 204 vs 38 ns for `*`, parity for `/`. Most
|
||||
of the cost is normalization one digit at a time (52% of instructions in
|
||||
`Guard::doDropDigit<unsigned __int128>`) plus a heap-allocated error string
|
||||
on every `+` and `*`, not the 64-bit width.
|
||||
|
||||

|
||||
|
||||
**Precision pays off in ledger formulas.** Each formula loses about the same
|
||||
number of digits at any precision, so a 34-digit type keeps ~15 more correct
|
||||
digits than today's `Number`, and 38 digits ~4 more again. The AMM swap-in is
|
||||
the clearest case: 6.5 correct digits in the worst case today, against 22
|
||||
(34 digits) or 26 (38 digits). 16-digit types are ruled out: they get 60-65%
|
||||
of vault share conversions wrong.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
**Implications for a 128-bit `Number`.**
|
||||
|
||||
- Speed is not a reason to stay at 64 bits. Intel BID128 runs every ledger
|
||||
formula 2-3x faster than today's `Number` while keeping ~15 more digits.
|
||||
Removing several digits per normalization step (one divide by 10^k, with
|
||||
the remainder feeding the guard) and passing error locations as
|
||||
`char const*` instead of `std::string` are likely the largest gains, at
|
||||
any width.
|
||||
- 34 vs 38 digits: mpdecimal runs both at the same speed, so the choice is
|
||||
~4 extra digits against matching IEEE decimal128 exactly (and being able
|
||||
to use Intel's library as a drop-in reference for testing).
|
||||
- `power` needs a better algorithm (wider intermediates or a correctly
|
||||
rounded method) regardless of width, if Lending needs it accurate to the
|
||||
last digit.
|
||||
- A prototype can be measured with no changes here: add
|
||||
`include/xrpl/basics/Number128.h` (see "Measuring a Number128 prototype").
|
||||
|
||||
## Library shortlist
|
||||
|
||||
| Library | Types | Role | Import |
|
||||
| -------------------------------------------------------- | ---------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Boost.Decimal (Boost >= 1.91) | `decimal64_t`, `decimal128_t`, `decimal_fast64_t`, `decimal_fast128_t` | Primary competitor; header-only C++ with operator overloads. The `fast` types are always normalized (no cohorts), the same design as `Number`. | Already a dependency (`boost/1.91.0`). |
|
||||
| Intel Decimal Floating-Point Math Library (libbid 2.0u4) | BID64, BID128 | IEEE reference and expected performance ceiling; backs GCC's `_Decimal*` on x86. Per-call rounding mode and status flags. | Fetched from netlib and built at build time (`-Dbench_intel_dfp=ON`); a Conan recipe if it is adopted beyond benchmarking. |
|
||||
| mpdecimal (libmpdec 4.0) | arbitrary precision (configured to 19 / 34 / 38 digits) | Correctly-rounded oracle for the divergence harness; previews 34- and 38-digit results. | ConanCenter `mpdecimal/4.0.0`. |
|
||||
| IBM decNumber (optional) | `decDouble`, `decQuad` | Second independent IEEE implementation, only if Boost and Intel disagree. | Vendored. |
|
||||
|
||||
Excluded: GCC `_Decimal*` built-ins (Intel's library underneath, not portable
|
||||
to Clang), Rust `rust_decimal` (96-bit, non-IEEE, FFI distorts timings),
|
||||
fixed-point libraries.
|
||||
|
||||
Baselines: `double`; `Number` under each `MantissaScale` (`Small`,
|
||||
`LargeLegacy`, `Large330`).
|
||||
|
||||
## How Number diverges from IEEE 754 decimal
|
||||
|
||||
Items marked _(to verify)_ are hypotheses from reading the code; the
|
||||
divergence harness below will confirm or refute them.
|
||||
|
||||
| Aspect | IEEE 754 decimal | `Number` |
|
||||
| ----------------------------- | --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Precision | 16 digits (decimal64), 34 (decimal128) | Internal mantissa `[10^18, 10^19-1]`, but the external `mantissa()` is capped at `int64` (`kMaxRep`); mantissas above it are stored with a forced trailing zero, so precision is non-uniform (~18.96 digits). |
|
||||
| Cohorts / quantum | Preserved (`1.0` and `1.00` are distinct representations) | Always normalized; one representation per value; `==` compares fields. |
|
||||
| Exponent range | ±384 (decimal64), ±6144 (decimal128) | ±32768 (`int` exponent). |
|
||||
| Underflow | Gradual (subnormals) | Flush to zero (`doNormalize`). |
|
||||
| Overflow | ±∞ plus a status flag | Throws `std::overflow_error`. |
|
||||
| Division by zero | ±∞ / NaN | Throws. |
|
||||
| NaN / ∞ | Yes | None; comparisons are a plain total order. |
|
||||
| Signed zero | Yes | `-0` collapses to `0`. |
|
||||
| Rounding modes | 5 (incl. ties-away) | 4; thread-local global mode, similar to `fenv`. (Boost.Decimal's mode is a plain process-wide global, not thread-local.) |
|
||||
| Status flags | Inexact, underflow, overflow, … | None. |
|
||||
| Correct rounding of + − × ÷ √ | Required | `×` uses a 128-bit product with guard digits (likely correct). `÷` uses a fixed 10^17 scale plus a staged correction; the legacy `Small` path may not be correctly rounded _(to verify)_. `root`/`root2` (Newton iteration) and `power` (repeated squaring) are not guaranteed correctly rounded. |
|
||||
| Amendment-gated variants | n/a | `Small`, `LargeLegacy` (keeps a known cusp-rounding error), `Large320`, `Large330`; all must stay bit-identical. |
|
||||
| Representable range | Symmetric | −2^63 is not exactly representable. |
|
||||
| Encoding | 8 / 16 bytes (BID or DPD) | 24 bytes in memory (`bool`, `uint64`, `int`, with padding); 12 bytes serialized as `STNumber` (`int64` + `int32`). |
|
||||
| Other operations | fma, quantize, sameQuantum, … | None. |
|
||||
|
||||
## Integration
|
||||
|
||||
Nothing in `libxrpl` changes; all code lives under
|
||||
`src/benchmarks/libxrpl/number/` and builds only when `benchmark=ON`.
|
||||
|
||||
- `xrpl_add_benchmark(number)` in `src/benchmarks/libxrpl/CMakeLists.txt`.
|
||||
- `number/NumberLike.h`: a concept capturing the subset of `Number`'s
|
||||
interface used by ledger code: construction from `(int64 mantissa, int
|
||||
exponent)` and from `int64`; `+ - * /`; `< ==`; explicit conversion to
|
||||
`int64`; `to_string`; `abs`; `power(x, n)`; `root2`; rounding-mode control
|
||||
mapped onto `Number::RoundingMode`; and `decompose()`, which returns
|
||||
`(sign, coefficient, exponent)` for digit-exact comparison.
|
||||
- `number/adapters/*.h`: one thin wrapper per library, holding a single
|
||||
backend value, all functions inline. `Number` satisfies the concept
|
||||
directly.
|
||||
- Optional backends are gated behind options that are off by default, so the
|
||||
default benchmark build gains no dependencies. mpdecimal: Conan option
|
||||
`bench_mpdecimal` (which also sets the CMake option of the same name).
|
||||
Intel: CMake option `bench_intel_dfp`; the tarball is downloaded (SHA-256
|
||||
pinned) and built with its own makefile via `ExternalProject`, with
|
||||
`CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0` and `-O3`.
|
||||
|
||||
## Benchmarks (each templated on the NumberLike type)
|
||||
|
||||
1. Single operations: add (equal exponents; exponents differing by 1, 5, and
|
||||
18), subtract with heavy cancellation, multiply, divide, compare,
|
||||
construct from `int64` / `(mantissa, exponent)`, convert to `int64`,
|
||||
`to_string`, `power`, `root2`.
|
||||
2. Operand sets, generated up front into arrays (never compile-time
|
||||
constants; Boost.Decimal is `constexpr`, so constants would be folded
|
||||
away): random full-width coefficients; IOU-like values (16 digits); XRP
|
||||
drop amounts; MPT integers up to 2^63−1; wide exponent spread.
|
||||
3. Throughput (independent operations) and latency (each result feeds the
|
||||
next) variants.
|
||||
4. Realistic formula kernels, as templated copies: AMM swap in/out
|
||||
(`AMMHelpers`), Vault share/asset conversion, Lending interest and
|
||||
amortization (`power`, `root`). Report speed and the size of
|
||||
end-result differences.
|
||||
5. Measurement discipline: Release, `-O3`, pinned core, fixed CPU frequency,
|
||||
`--benchmark_repetitions=10`, report medians and variation, compare runs
|
||||
with Google Benchmark's `compare.py`. Report the cost of `Number`'s
|
||||
thread-local loads (`mode`, `kRange`) separately.
|
||||
|
||||
## Divergence harness
|
||||
|
||||
A separate executable, `xrpl.bench.number_diff`, runs millions of random and
|
||||
edge-case operations against mpdecimal configured to match `Number`'s
|
||||
precision and exponent range. It tallies results that are exact, off by
|
||||
1 ULP, or worse, per operation, mantissa scale, and rounding mode. It also
|
||||
checks Boost.Decimal and Intel against mpdecimal to validate the oracle, and
|
||||
becomes the regression oracle for a 128-bit `Number` at 34 and 38 digits.
|
||||
|
||||
## Findings so far
|
||||
|
||||
Divergence results are from `xrpl.bench.number_diff --samples 100000` (all
|
||||
four rounding modes, every mantissa scale). "Correctly rounded" means identical to the exact result
|
||||
rounded onto the type's own grid of representable values.
|
||||
|
||||
### Validating the oracle
|
||||
|
||||
The harness was checked against IEEE implementations, and every disagreement
|
||||
was cross-checked against both mpdecimal at the native precision and Python's
|
||||
`decimal` module. The oracle agreed with both in every case; the
|
||||
disagreements are Boost.Decimal 1.91 defects:
|
||||
|
||||
- `llrint` returns wrong values for inputs that are already integers at full
|
||||
precision (e.g. `1.23e17` gives `12345678901235`), and divides by zero
|
||||
(SIGFPE) when the significand exponent is exactly 0. Fixed upstream; the
|
||||
adapter works around it.
|
||||
- Directed rounding is ignored when an add/sub operand is far below the
|
||||
other's last digit: `6111703207134976 - 5.857e-10` toward zero returns
|
||||
`...976` instead of `...975` (`decimal64_t` and `decimal_fast64_t`, about
|
||||
10% of IOU-dataset subtractions in `TowardsZero`).
|
||||
- `decimal128_t` division misrounds near ties (1 in 2000 full-width
|
||||
quotients).
|
||||
- `sqrt` is not correctly rounded (errors up to ~4.5 ULP), although IEEE
|
||||
requires it.
|
||||
- The `fast` types are not correctly rounded for `*` and `/` (up to ~1 ULP),
|
||||
as documented.
|
||||
|
||||
Intel's library is correctly rounded for `+ - * /`, `sqrt`, and conversion
|
||||
to int64 in all four modes (9.2 million of 9.2 million samples for both
|
||||
BID64 and BID128), including the cases Boost gets wrong. Its `pow` is not
|
||||
correctly rounded (allowed by IEEE): up to ~1500 ULP for x^360 with x in
|
||||
[1, 1.01). Example confirmed with Python's `decimal`: 1.002740590413819539^360
|
||||
is `2.678516411804073475375264650425851`, Intel BID128 returns
|
||||
`...650426258`. So no IEEE library here provides an accurate integer power;
|
||||
Number's `power` errors are of the same order as Intel's.
|
||||
|
||||
### Number
|
||||
|
||||
- **`Large330` (current):** `+ - * /` and conversion to int64 are correctly
|
||||
rounded in all four modes on every dataset, including the exponent-gap,
|
||||
cancellation, tie, and int64-cap ("cusp") cases: 9.2 million of 9.2
|
||||
million samples.
|
||||
- **Older scales:** the harness reproduces what each amendment fixed.
|
||||
`LargeLegacy` misrounds at the int64 cap in `Upward` (the documented
|
||||
`fixCleanup3_2_0` defect); `LargeLegacy` and `Large320` misround at the cap
|
||||
in the other modes and misround subtraction in `TowardsZero` (fixed in
|
||||
`Large330`); `Small` misrounds 3-5% of divisions.
|
||||
- **`root2`** is not correctly rounded in any scale: errors up to ~2.7 ULP
|
||||
to nearest and ~5 ULP in directed modes.
|
||||
- **`power`** accumulates error through repeated squaring without extra
|
||||
precision: ~5 ULP for n = 12 and ~800-2800 ULP for n = 360. Relevant to
|
||||
Lending amortization, and a design input for a 128-bit Number: wider
|
||||
intermediates would remove most of it.
|
||||
|
||||
### Performance (phase 1)
|
||||
|
||||
Median ns per operation, throughput (independent operations), from
|
||||
`xrpl.bench.number --benchmark_repetitions=3`. GCC 15.2, Release,
|
||||
single pinned core of a 16-core x86-64 machine. CPU frequency scaling was on
|
||||
(`powersave`), so treat differences under ~10% as noise; the median
|
||||
coefficient of variation across benchmarks was 1.1%.
|
||||
|
||||
N = Number, B = `BoostDecimal`, BF = `BoostDecimalFast`, Mpd = mpdecimal at
|
||||
that precision.
|
||||
|
||||
| op | data | N.Small | N.LargeLegacy | N.Large330 | B64 | BF64 | B128 | BF128 | Mpd19 | Mpd34 | Mpd38 | double |
|
||||
| --------- | ---- | ------: | ------------: | ---------: | ---: | ---: | ---: | ----: | ----: | ----: | ----: | -----: |
|
||||
| add | full | 105 | 105 | 109 | 17.6 | 21.9 | 118 | 118 | 42.1 | 26.1 | 25.7 | 0.6 |
|
||||
| sub | full | 128 | 125 | 110 | 17.7 | 21.3 | 113 | 130 | 40.1 | 26.4 | 26.3 | 0.6 |
|
||||
| mul | full | 192 | 210 | 248 | 9.2 | 18.0 | 77.4 | 139 | 30.3 | 31.9 | 15.7 | 0.6 |
|
||||
| div | full | 40.8 | 66.2 | 70.2 | 38.6 | 44.3 | 336 | 136 | 55.4 | 54.3 | 57.0 | 0.9 |
|
||||
| lt | full | 1.2 | 1.4 | 1.4 | 3.8 | 2.1 | 8.6 | 2.1 | 4.1 | 4.3 | 4.4 | 0.6 |
|
||||
| construct | full | 24.3 | 19.1 | 18.4 | 4.7 | 4.8 | 1.0 | 6.5 | 12.7 | 12.7 | 12.7 | 11.0 |
|
||||
| to_int64 | mpt | 26.4 | 30.1 | 34.8 | 11.7 | 4.4 | 86.4 | 56.7 | 9.4 | 9.5 | 9.5 | 1.5 |
|
||||
| root2 | full | 1575 | 1734 | 2077 | 120 | 159 | 856 | 906 | 929 | 1212 | 1492 | 1.3 |
|
||||
| add_gap0 | full | 48.5 | 52.1 | 34.7 | 14.4 | 20.4 | 63.7 | 56.0 | 24.3 | 18.3 | 18.3 | 0.6 |
|
||||
| add_gap18 | full | 207 | 204 | 204 | 18.6 | 21.8 | 188 | 162 | 43.6 | 44.8 | 29.1 | 0.6 |
|
||||
| power360 | rate | 1973 | 2503 | 2229 | 221 | 278 | 1421 | 1567 | 788 | 1259 | 1246 | 2.9 |
|
||||
| mul_chain | rate | 180 | 204 | 284 | 13.1 | 25.0 | 135 | 127 | 32.7 | 36.0 | 36.7 | 0.8 |
|
||||
| div_chain | rate | 47.2 | 69.9 | 103 | 42.3 | 38.3 | 350 | 130 | 54.4 | 58.0 | 60.7 | 2.7 |
|
||||
|
||||
Observations:
|
||||
|
||||
- Number's `+` and `*` are 6-25x slower than Boost `decimal64_t`, and 4-15x
|
||||
slower than mpdecimal at 38 digits, even though mpdecimal is arbitrary
|
||||
precision and heap-capable. Comparison is Number's only clear win.
|
||||
- Under callgrind, 52% of the instructions in Number's multiply-and-add loop
|
||||
are in `Guard::doDropDigit<unsigned __int128>`, which removes surplus digits
|
||||
one at a time with a 128-bit divide each. `add_gap0` vs `add_gap18` (35 vs
|
||||
204 ns) shows the same per-digit cost in addition. About another 10% is
|
||||
`malloc`/`free`: `doRoundUp`/`doRound` take the error-location message as a
|
||||
`std::string` by value, so every `*` builds one with `std::to_string`, and
|
||||
every `+` builds `"Number::addition overflow"`, which is too long for the
|
||||
small-string optimization.
|
||||
- So Number's cost today is digit-at-a-time normalization, not mantissa
|
||||
width. A 128-bit Number does not need to be slower than today's if it
|
||||
removes several digits per step (divide by 10^k with a remainder for the
|
||||
guard). mpdecimal at 38 digits (add 26 ns, mul 16 ns, div 57 ns) is a
|
||||
realistic target.
|
||||
- Boost `decimal128_t` is slower than mpdecimal at 34 digits for `+` and `/`
|
||||
(`/` 336 vs 54 ns), so Boost.Decimal is not a performance reference at 128
|
||||
bits. Intel BID128 is (next section).
|
||||
- `mpdecimal` at 34 and 38 digits costs about the same; 38 is often faster
|
||||
because 19-digit x 19-digit products fit without rounding. Precision alone
|
||||
does not separate the 34- and 38-digit options on speed.
|
||||
|
||||
### Performance with Intel (phase 3)
|
||||
|
||||
A separate run (same build, same conditions) of the Intel types with three
|
||||
anchor subjects repeated from the table above. The anchors moved by less than
|
||||
5%, except Number's `add`, which was ~15% faster in this run; read the two
|
||||
tables as consistent to within about that margin.
|
||||
|
||||
| op | data | N.Large330 | B64 | Mpd38 | Intel64 | Intel128 |
|
||||
| --------- | ---- | ---------: | ---: | ----: | ------: | -------: |
|
||||
| add | full | 89.8 | 16.3 | 25.5 | 12.3 | 11.8 |
|
||||
| sub | full | 94.2 | 16.8 | 25.9 | 12.3 | 13.0 |
|
||||
| mul | full | 206 | 9.1 | 15.7 | 18.3 | 37.6 |
|
||||
| div | full | 65.5 | 36.0 | 57.7 | 19.7 | 64.1 |
|
||||
| lt | full | 1.4 | 3.6 | 4.4 | 4.2 | 6.6 |
|
||||
| construct | full | 16.6 | 4.2 | 12.6 | 11.1 | 4.0 |
|
||||
| to_int64 | mpt | 27.8 | 9.4 | 9.6 | 4.1 | 4.8 |
|
||||
| root2 | full | 1819 | 114 | 1491 | 14.8 | 108 |
|
||||
| add_gap18 | full | 193 | 17.9 | 29.0 | 8.3 | 24.2 |
|
||||
| power360 | rate | 2196 | 193 | 1275 | 259 | 1149 |
|
||||
| mul_chain | rate | 197 | 13.1 | 38.9 | 19.2 | 49.7 |
|
||||
| div_chain | rate | 70.0 | 41.8 | 60.8 | 28.2 | 69.2 |
|
||||
|
||||
Observations:
|
||||
|
||||
- Intel BID128 (34 digits, correctly rounded) adds in ~12 ns and multiplies
|
||||
in 25-38 ns: 7-8x faster than today's 64-bit Number for `+` and 5-8x for
|
||||
`*`, with division at parity. It is the realistic speed target for a
|
||||
128-bit Number.
|
||||
- Intel's `sqrt` is two orders of magnitude faster than Number's `root2`
|
||||
(15 vs ~1800 ns at 16 digits; 108 ns at 34 digits) and correctly rounded.
|
||||
|
||||
### Ledger formulas (phase 4)
|
||||
|
||||
`number/Kernels.h` holds templated copies of production formulas, run
|
||||
identically on every type, including the rounding-mode switches:
|
||||
`swapAssetIn`/`swapAssetOut` (fixAMMv1_1 branch), `assetsToSharesDeposit`
|
||||
and `sharesToAssetsDeposit`, and `loanPeriodicRate` plus
|
||||
`loanPeriodicPayment` (fixCleanup3_2_0 branch). `loan_amortize` runs a full
|
||||
monthly schedule (12 or 360 payments) and measures the leftover balance,
|
||||
which is exactly 0 in exact arithmetic, relative to the principal. Inputs:
|
||||
AMM pools of 10^6 to 10^12 with trades of 10^-6 to 10^-2 of the pool and
|
||||
fees up to 1%; vaults of 10^9 to 10^17 integer assets with share supply up
|
||||
to 10^18; loans of 10^3 to 10^9 at 0.1% to 30% a year.
|
||||
ns per call
|
||||
|
||||
| kernel | N.Small | N.Large330 | B64 | Intel64 | double | B128 | Intel128 | Mpd34 | Mpd38 |
|
||||
| ---------------------- | ------: | ---------: | ----: | ------: | -----: | ----: | -------: | ----: | ----: |
|
||||
| amm_swap_in | 685 | 770 | 148 | 128 | 23 | 646 | 266 | 390 | 404 |
|
||||
| amm_swap_out | 508 | 598 | 174 | 127 | 27 | 590 | 289 | 385 | 415 |
|
||||
| vault_assets_to_shares | 237 | 291 | 66 | 45 | 10 | 432 | 95 | 91 | 90 |
|
||||
| vault_shares_to_assets | 219 | 266 | 48 | 39 | 0.8 | 333 | 79 | 64 | 63 |
|
||||
| loan_payment12 | 1870 | 2115 | 293 | 351 | 4.3 | 1679 | 1041 | 1158 | 1296 |
|
||||
| loan_payment360 | 3599 | 3555 | 399 | 487 | 6.7 | 2616 | 1867 | 2454 | 2200 |
|
||||
| loan_amortize12 | 5108 | 5803 | 1024 | 1055 | 16 | 5385 | 3308 | 3167 | 3091 |
|
||||
| loan_amortize360 | 98902 | 114119 | 19552 | 18359 | 888 | 96989 | 58486 | 55230 | 45190 |
|
||||
|
||||
Correct significant digits against a 96-digit evaluation of the same
|
||||
formula from the same inputs, worst case (median) over 256 cases; for
|
||||
integer results, cases that differ from the exact truncated result:
|
||||
|
||||
| kernel | N.Small | N.Large330 | B64 | Intel64 | double | B128 | Intel128 | Mpd34 | Mpd38 |
|
||||
| ---------------------- | ------------: | ----------: | ------------: | ------------: | ------------: | ----------: | ----------: | ----------: | ----------: |
|
||||
| amm_swap_in | 3.3 (10.9) | 6.5 (13.9) | 3.3 (10.8) | 3.3 (10.8) | 4.0 (11.8) | 21.9 (29.7) | 21.9 (29.8) | 21.9 (29.8) | 26.3 (33.7) |
|
||||
| amm_swap_out | 8.5 (11.3) | 11.4 (14.1) | 8.5 (11.2) | 8.5 (11.2) | 8.9 (11.4) | 26.9 (29.9) | 26.8 (30.0) | 26.8 (30.0) | 30.5 (33.9) |
|
||||
| vault_assets_to_shares | 165/256 wrong | 0/256 wrong | 165/256 wrong | 165/256 wrong | 153/256 wrong | 0/256 wrong | 0/256 wrong | 0/256 wrong | 0/256 wrong |
|
||||
| vault_shares_to_assets | 15.2 (16.0) | 18.1 (19.0) | 15.2 (16.0) | 15.2 (16.0) | 15.8 (16.3) | 33.4 (34.3) | 33.4 (34.3) | 33.4 (34.3) | 37.4 (38.2) |
|
||||
| loan_payment12 | 11.9 (13.7) | 14.8 (16.7) | 11.9 (13.7) | 11.9 (13.7) | 12.4 (14.4) | 29.4 (31.7) | 29.4 (31.7) | 29.7 (31.8) | 33.7 (35.8) |
|
||||
| loan_payment360 | 11.9 (15.1) | 15.0 (18.1) | 11.8 (15.2) | 11.8 (15.2) | 12.6 (15.7) | 30.0 (33.1) | 30.0 (33.1) | 30.1 (33.1) | 34.1 (37.1) |
|
||||
| loan_amortize12 | 11.9 (13.7) | 14.8 (16.7) | 11.9 (13.7) | 11.9 (13.7) | 12.4 (14.3) | 29.4 (31.7) | 29.4 (31.7) | 29.7 (31.7) | 33.7 (35.7) |
|
||||
| loan_amortize360 | 11.2 (12.8) | 14.1 (15.8) | 11.2 (12.8) | 11.2 (12.8) | 11.5 (13.3) | 29.1 (30.8) | 29.1 (30.8) | 29.1 (30.8) | 33.0 (34.8) |
|
||||
|
||||
Observations:
|
||||
|
||||
- Wider precision pays off directly in the results. For every formula the
|
||||
34-digit types keep about 15 more correct digits than today's Number, and
|
||||
38 digits adds about 4 more. The relative loss is roughly constant: each
|
||||
formula loses the same number of digits at any precision, so the extra
|
||||
digits carry through to the result.
|
||||
- `amm_swap_in` is where it matters most: `poolOut - ratio` cancels for
|
||||
small trades, leaving 6.5 correct digits in the worst case with today's
|
||||
Number and about 3 with 16-digit types, against 22 (34 digits) and 26
|
||||
(38 digits).
|
||||
- 16-digit types get 60-65% of vault share conversions wrong, because share
|
||||
counts above 10^16 do not fit. Today's Number and all wider types get every
|
||||
case right. Any 64-bit IEEE type is ruled out for vaults.
|
||||
- Intel BID128 runs every formula 2-3x faster than today's Number while
|
||||
keeping ~15 more digits. A 128-bit Number does not have to trade speed for
|
||||
precision.
|
||||
- Loans lose about 4 digits at every precision, even though `power` is off
|
||||
by up to thousands of ULP: the payment formula is not very sensitive to
|
||||
that error.
|
||||
- `Number.Small` and `Number.Large330` cost about the same.
|
||||
|
||||
## Phases
|
||||
|
||||
1. `Number` and Boost.Decimal adapters, plus the single-operation benchmarks
|
||||
(no new dependencies). (Done.)
|
||||
2. The divergence harness and the mpdecimal dependency, at 19, 34, and 38
|
||||
digits. (Done.)
|
||||
3. The Intel BID adapter, fetched and built at build time. (Done.)
|
||||
4. The realistic formula kernels. (Done.)
|
||||
5. Write-up with figures, and a slot for a `Number128` prototype. (Done.)
|
||||
|
||||
## Measuring a Number128 prototype
|
||||
|
||||
Add `include/xrpl/basics/Number128.h` defining `xrpl::Number128` with
|
||||
Number's interface (the `DecomposableNumber` concept in
|
||||
`src/benchmarks/libxrpl/number/NumberLike.h`) and
|
||||
`static constexpr unsigned kDigits`. On the next build CMake notices the
|
||||
header (a `CONFIGURE_DEPENDS` glob) and defines `XRPL_BENCH_NUMBER128`, and
|
||||
the type appears in every single-operation and formula benchmark and in
|
||||
`number_diff`, which checks it against a `kDigits`-digit grid. Nothing in
|
||||
the benchmark sources needs to change. See
|
||||
`src/benchmarks/libxrpl/number/Number128.h`.
|
||||
|
||||
## Running
|
||||
|
||||
```bash
|
||||
# From the build directory. bench_mpdecimal is only needed for number_diff
|
||||
# and the MpDecimal benchmark subjects; bench_intel_dfp downloads and builds
|
||||
# Intel's library (needs network access and `make`).
|
||||
conan install .. --output-folder . --build missing -s build_type=Release \
|
||||
-o '&:benchmark=True' -o '&:bench_mpdecimal=True'
|
||||
cmake -G Ninja -DCMAKE_TOOLCHAIN_FILE=build/generators/conan_toolchain.cmake \
|
||||
-DCMAKE_BUILD_TYPE=Release -Dbench_intel_dfp=ON ..
|
||||
cmake --build . --target xrpl.bench.number xrpl.bench.number_diff
|
||||
|
||||
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_repetitions=10 \
|
||||
--benchmark_report_aggregates_only=true
|
||||
./src/benchmarks/libxrpl/xrpl.bench.number_diff --samples 100000 >divergence.md
|
||||
|
||||
# Regenerate the figures (standard-library Python only).
|
||||
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_filter='^(add|mul|div)/full/' \
|
||||
--benchmark_repetitions=5 --benchmark_report_aggregates_only=true \
|
||||
--benchmark_out=core.json --benchmark_out_format=json
|
||||
./src/benchmarks/libxrpl/xrpl.bench.number --benchmark_filter='^kernel/' \
|
||||
--benchmark_repetitions=3 --benchmark_report_aggregates_only=true \
|
||||
--benchmark_out=kernels.json --benchmark_out_format=json
|
||||
../src/benchmarks/libxrpl/number/plot_results.py core.json kernels.json \
|
||||
../docs/images/number
|
||||
```
|
||||
105
docs/images/number/number-amm-swap-in.svg
Normal file
105
docs/images/number/number-amm-swap-in.svg
Normal file
@@ -0,0 +1,105 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 276" width="834" height="276" role="img" aria-label="AMM swap-in: correct digits (higher is better) and cost (lower is faster)">
|
||||
<style>
|
||||
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
|
||||
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
|
||||
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
|
||||
</style>
|
||||
<rect class="bg" width="834" height="276"/>
|
||||
<text class="t1" x="0" y="18" font-size="15" font-weight="600">AMM swap-in: correct digits (higher is better) and cost (lower is faster)</text>
|
||||
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
|
||||
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="232.8" y="42">16-digit types</text>
|
||||
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
|
||||
<text class="t1" x="0" y="94.0">Number (Large330)</text>
|
||||
<text class="t1" x="0" y="118.0">Boost decimal64</text>
|
||||
<text class="t1" x="0" y="142.0">Intel BID64</text>
|
||||
<text class="t1" x="0" y="166.0">Boost decimal128</text>
|
||||
<text class="t1" x="0" y="190.0">Intel BID128</text>
|
||||
<text class="t1" x="0" y="214.0">mpdecimal 34</text>
|
||||
<text class="t1" x="0" y="238.0">mpdecimal 38</text>
|
||||
<text class="t1" x="150" y="64" font-weight="600">Correct digits, worst case</text>
|
||||
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="246"/>
|
||||
<text class="t2" x="150.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="246"/>
|
||||
<text class="t2" x="190.0" y="262" text-anchor="middle" font-size="11">10</text>
|
||||
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="246"/>
|
||||
<text class="t2" x="230.0" y="262" text-anchor="middle" font-size="11">20</text>
|
||||
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="246"/>
|
||||
<text class="t2" x="270.0" y="262" text-anchor="middle" font-size="11">30</text>
|
||||
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="246"/>
|
||||
<text class="t2" x="310.0" y="262" text-anchor="middle" font-size="11">40</text>
|
||||
<text class="t2" x="310.0" y="274" text-anchor="end" font-size="11">significant digits</text>
|
||||
<path class="f-number" d="M150.0,84.0 H172.0 Q176.0,84.0 176.0,88.0 V92.0 Q176.0,96.0 172.0,96.0 H150.0 Z"><title>Number (Large330): 6.5 significant digits</title></path>
|
||||
<text class="t2" x="181.0" y="94.0" font-size="11">6.5</text>
|
||||
<path class="f-ieee64" d="M150.0,108.0 H159.2 Q163.2,108.0 163.2,112.0 V116.0 Q163.2,120.0 159.2,120.0 H150.0 Z"><title>Boost decimal64: 3.3 significant digits</title></path>
|
||||
<text class="t2" x="168.2" y="118.0" font-size="11">3.3</text>
|
||||
<path class="f-ieee64" d="M150.0,132.0 H159.2 Q163.2,132.0 163.2,136.0 V140.0 Q163.2,144.0 159.2,144.0 H150.0 Z"><title>Intel BID64: 3.3 significant digits</title></path>
|
||||
<text class="t2" x="168.2" y="142.0" font-size="11">3.3</text>
|
||||
<path class="f-wide" d="M150.0,156.0 H233.7 Q237.7,156.0 237.7,160.0 V164.0 Q237.7,168.0 233.7,168.0 H150.0 Z"><title>Boost decimal128: 21.9 significant digits</title></path>
|
||||
<text class="t2" x="242.7" y="166.0" font-size="11">21.9</text>
|
||||
<path class="f-wide" d="M150.0,180.0 H233.7 Q237.7,180.0 237.7,184.0 V188.0 Q237.7,192.0 233.7,192.0 H150.0 Z"><title>Intel BID128: 21.9 significant digits</title></path>
|
||||
<text class="t2" x="242.7" y="190.0" font-size="11">21.9</text>
|
||||
<path class="f-wide" d="M150.0,204.0 H233.7 Q237.7,204.0 237.7,208.0 V212.0 Q237.7,216.0 233.7,216.0 H150.0 Z"><title>mpdecimal 34: 21.9 significant digits</title></path>
|
||||
<text class="t2" x="242.7" y="214.0" font-size="11">21.9</text>
|
||||
<path class="f-wide" d="M150.0,228.0 H251.2 Q255.2,228.0 255.2,232.0 V236.0 Q255.2,240.0 251.2,240.0 H150.0 Z"><title>mpdecimal 38: 26.3 significant digits</title></path>
|
||||
<text class="t2" x="260.2" y="238.0" font-size="11">26.3</text>
|
||||
<line class="base" x1="150" y1="78" x2="150" y2="246"/>
|
||||
<text class="t1" x="378" y="64" font-weight="600">Correct digits, median</text>
|
||||
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="246"/>
|
||||
<text class="t2" x="378.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="246"/>
|
||||
<text class="t2" x="418.0" y="262" text-anchor="middle" font-size="11">10</text>
|
||||
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="246"/>
|
||||
<text class="t2" x="458.0" y="262" text-anchor="middle" font-size="11">20</text>
|
||||
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="246"/>
|
||||
<text class="t2" x="498.0" y="262" text-anchor="middle" font-size="11">30</text>
|
||||
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="246"/>
|
||||
<text class="t2" x="538.0" y="262" text-anchor="middle" font-size="11">40</text>
|
||||
<text class="t2" x="538.0" y="274" text-anchor="end" font-size="11">significant digits</text>
|
||||
<path class="f-number" d="M378.0,84.0 H429.6 Q433.6,84.0 433.6,88.0 V92.0 Q433.6,96.0 429.6,96.0 H378.0 Z"><title>Number (Large330): 13.9 significant digits</title></path>
|
||||
<text class="t2" x="438.6" y="94.0" font-size="11">13.9</text>
|
||||
<path class="f-ieee64" d="M378.0,108.0 H417.3 Q421.3,108.0 421.3,112.0 V116.0 Q421.3,120.0 417.3,120.0 H378.0 Z"><title>Boost decimal64: 10.8 significant digits</title></path>
|
||||
<text class="t2" x="426.3" y="118.0" font-size="11">10.8</text>
|
||||
<path class="f-ieee64" d="M378.0,132.0 H417.3 Q421.3,132.0 421.3,136.0 V140.0 Q421.3,144.0 417.3,144.0 H378.0 Z"><title>Intel BID64: 10.8 significant digits</title></path>
|
||||
<text class="t2" x="426.3" y="142.0" font-size="11">10.8</text>
|
||||
<path class="f-wide" d="M378.0,156.0 H492.6 Q496.6,156.0 496.6,160.0 V164.0 Q496.6,168.0 492.6,168.0 H378.0 Z"><title>Boost decimal128: 29.7 significant digits</title></path>
|
||||
<text class="t2" x="501.6" y="166.0" font-size="11">29.7</text>
|
||||
<path class="f-wide" d="M378.0,180.0 H493.3 Q497.3,180.0 497.3,184.0 V188.0 Q497.3,192.0 493.3,192.0 H378.0 Z"><title>Intel BID128: 29.8 significant digits</title></path>
|
||||
<text class="t2" x="502.3" y="190.0" font-size="11">29.8</text>
|
||||
<path class="f-wide" d="M378.0,204.0 H493.3 Q497.3,204.0 497.3,208.0 V212.0 Q497.3,216.0 493.3,216.0 H378.0 Z"><title>mpdecimal 34: 29.8 significant digits</title></path>
|
||||
<text class="t2" x="502.3" y="214.0" font-size="11">29.8</text>
|
||||
<path class="f-wide" d="M378.0,228.0 H508.7 Q512.7,228.0 512.7,232.0 V236.0 Q512.7,240.0 508.7,240.0 H378.0 Z"><title>mpdecimal 38: 33.7 significant digits</title></path>
|
||||
<text class="t2" x="517.7" y="238.0" font-size="11">33.7</text>
|
||||
<line class="base" x1="378" y1="78" x2="378" y2="246"/>
|
||||
<text class="t1" x="606" y="64" font-weight="600">Cost</text>
|
||||
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="246"/>
|
||||
<text class="t2" x="606.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="638.0" y1="78" x2="638.0" y2="246"/>
|
||||
<text class="t2" x="638.0" y="262" text-anchor="middle" font-size="11">200</text>
|
||||
<line class="grid" x1="670.0" y1="78" x2="670.0" y2="246"/>
|
||||
<text class="t2" x="670.0" y="262" text-anchor="middle" font-size="11">400</text>
|
||||
<line class="grid" x1="702.0" y1="78" x2="702.0" y2="246"/>
|
||||
<text class="t2" x="702.0" y="262" text-anchor="middle" font-size="11">600</text>
|
||||
<line class="grid" x1="734.0" y1="78" x2="734.0" y2="246"/>
|
||||
<text class="t2" x="734.0" y="262" text-anchor="middle" font-size="11">800</text>
|
||||
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="246"/>
|
||||
<text class="t2" x="766.0" y="262" text-anchor="middle" font-size="11">1000</text>
|
||||
<text class="t2" x="766.0" y="274" text-anchor="end" font-size="11">ns per call</text>
|
||||
<path class="f-number" d="M606.0,84.0 H725.2 Q729.2,84.0 729.2,88.0 V92.0 Q729.2,96.0 725.2,96.0 H606.0 Z"><title>Number (Large330): 770 ns per call</title></path>
|
||||
<text class="t2" x="734.2" y="94.0" font-size="11">770</text>
|
||||
<path class="f-ieee64" d="M606.0,108.0 H625.7 Q629.7,108.0 629.7,112.0 V116.0 Q629.7,120.0 625.7,120.0 H606.0 Z"><title>Boost decimal64: 148 ns per call</title></path>
|
||||
<text class="t2" x="634.7" y="118.0" font-size="11">148</text>
|
||||
<path class="f-ieee64" d="M606.0,132.0 H622.4 Q626.4,132.0 626.4,136.0 V140.0 Q626.4,144.0 622.4,144.0 H606.0 Z"><title>Intel BID64: 128 ns per call</title></path>
|
||||
<text class="t2" x="631.4" y="142.0" font-size="11">128</text>
|
||||
<path class="f-wide" d="M606.0,156.0 H705.3 Q709.3,156.0 709.3,160.0 V164.0 Q709.3,168.0 705.3,168.0 H606.0 Z"><title>Boost decimal128: 646 ns per call</title></path>
|
||||
<text class="t2" x="714.3" y="166.0" font-size="11">646</text>
|
||||
<path class="f-wide" d="M606.0,180.0 H644.6 Q648.6,180.0 648.6,184.0 V188.0 Q648.6,192.0 644.6,192.0 H606.0 Z"><title>Intel BID128: 266 ns per call</title></path>
|
||||
<text class="t2" x="653.6" y="190.0" font-size="11">266</text>
|
||||
<path class="f-wide" d="M606.0,204.0 H664.4 Q668.4,204.0 668.4,208.0 V212.0 Q668.4,216.0 664.4,216.0 H606.0 Z"><title>mpdecimal 34: 390 ns per call</title></path>
|
||||
<text class="t2" x="673.4" y="214.0" font-size="11">390</text>
|
||||
<path class="f-wide" d="M606.0,228.0 H666.7 Q670.7,228.0 670.7,232.0 V236.0 Q670.7,240.0 666.7,240.0 H606.0 Z"><title>mpdecimal 38: 404 ns per call</title></path>
|
||||
<text class="t2" x="675.7" y="238.0" font-size="11">404</text>
|
||||
<line class="base" x1="606" y1="78" x2="606" y2="246"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 9.2 KiB |
117
docs/images/number/number-core-ops.svg
Normal file
117
docs/images/number/number-core-ops.svg
Normal file
@@ -0,0 +1,117 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 324" width="834" height="324" role="img" aria-label="Cost of one operation (full-width operands, lower is faster)">
|
||||
<style>
|
||||
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
|
||||
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
|
||||
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
|
||||
</style>
|
||||
<rect class="bg" width="834" height="324"/>
|
||||
<text class="t1" x="0" y="18" font-size="15" font-weight="600">Cost of one operation (full-width operands, lower is faster)</text>
|
||||
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
|
||||
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="232.8" y="42">16-digit types</text>
|
||||
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
|
||||
<text class="t1" x="0" y="94.0">Number (Large330)</text>
|
||||
<text class="t1" x="0" y="118.0">Boost decimal64</text>
|
||||
<text class="t1" x="0" y="142.0">Boost decimal_fast64</text>
|
||||
<text class="t1" x="0" y="166.0">Intel BID64</text>
|
||||
<text class="t1" x="0" y="190.0">Boost decimal128</text>
|
||||
<text class="t1" x="0" y="214.0">Boost decimal_fast128</text>
|
||||
<text class="t1" x="0" y="238.0">Intel BID128</text>
|
||||
<text class="t1" x="0" y="262.0">mpdecimal 34</text>
|
||||
<text class="t1" x="0" y="286.0">mpdecimal 38</text>
|
||||
<text class="t1" x="150" y="64" font-weight="600">Add</text>
|
||||
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="294"/>
|
||||
<text class="t2" x="150.0" y="310" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="294"/>
|
||||
<text class="t2" x="190.0" y="310" text-anchor="middle" font-size="11">100</text>
|
||||
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="294"/>
|
||||
<text class="t2" x="230.0" y="310" text-anchor="middle" font-size="11">200</text>
|
||||
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="294"/>
|
||||
<text class="t2" x="270.0" y="310" text-anchor="middle" font-size="11">300</text>
|
||||
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="294"/>
|
||||
<text class="t2" x="310.0" y="310" text-anchor="middle" font-size="11">400</text>
|
||||
<text class="t2" x="310.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
|
||||
<path class="f-number" d="M150.0,84.0 H180.9 Q184.9,84.0 184.9,88.0 V92.0 Q184.9,96.0 180.9,96.0 H150.0 Z"><title>Number (Large330): 87 ns per operation</title></path>
|
||||
<text class="t2" x="189.9" y="94.0" font-size="11">87</text>
|
||||
<path class="f-ieee64" d="M150.0,108.0 H152.3 Q156.3,108.0 156.3,112.0 V116.0 Q156.3,120.0 152.3,120.0 H150.0 Z"><title>Boost decimal64: 16 ns per operation</title></path>
|
||||
<text class="t2" x="161.3" y="118.0" font-size="11">16</text>
|
||||
<path class="f-ieee64" d="M150.0,132.0 H154.2 Q158.2,132.0 158.2,136.0 V140.0 Q158.2,144.0 154.2,144.0 H150.0 Z"><title>Boost decimal_fast64: 21 ns per operation</title></path>
|
||||
<text class="t2" x="163.2" y="142.0" font-size="11">21</text>
|
||||
<path class="f-ieee64" d="M150.0,156.0 H150.6 Q154.6,156.0 154.6,160.0 V164.0 Q154.6,168.0 150.6,168.0 H150.0 Z"><title>Intel BID64: 11 ns per operation</title></path>
|
||||
<text class="t2" x="159.6" y="166.0" font-size="11">11</text>
|
||||
<path class="f-wide" d="M150.0,180.0 H190.4 Q194.4,180.0 194.4,184.0 V188.0 Q194.4,192.0 190.4,192.0 H150.0 Z"><title>Boost decimal128: 111 ns per operation</title></path>
|
||||
<text class="t2" x="199.4" y="190.0" font-size="11">111</text>
|
||||
<path class="f-wide" d="M150.0,204.0 H187.2 Q191.2,204.0 191.2,208.0 V212.0 Q191.2,216.0 187.2,216.0 H150.0 Z"><title>Boost decimal_fast128: 103 ns per operation</title></path>
|
||||
<text class="t2" x="196.2" y="214.0" font-size="11">103</text>
|
||||
<path class="f-wide" d="M150.0,228.0 H150.7 Q154.7,228.0 154.7,232.0 V236.0 Q154.7,240.0 150.7,240.0 H150.0 Z"><title>Intel BID128: 12 ns per operation</title></path>
|
||||
<text class="t2" x="159.7" y="238.0" font-size="11">12</text>
|
||||
<path class="f-wide" d="M150.0,252.0 H156.2 Q160.2,252.0 160.2,256.0 V260.0 Q160.2,264.0 156.2,264.0 H150.0 Z"><title>mpdecimal 34: 26 ns per operation</title></path>
|
||||
<text class="t2" x="165.2" y="262.0" font-size="11">26</text>
|
||||
<path class="f-wide" d="M150.0,276.0 H156.0 Q160.0,276.0 160.0,280.0 V284.0 Q160.0,288.0 156.0,288.0 H150.0 Z"><title>mpdecimal 38: 25 ns per operation</title></path>
|
||||
<text class="t2" x="165.0" y="286.0" font-size="11">25</text>
|
||||
<line class="base" x1="150" y1="78" x2="150" y2="294"/>
|
||||
<text class="t1" x="378" y="64" font-weight="600">Multiply</text>
|
||||
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="294"/>
|
||||
<text class="t2" x="378.0" y="310" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="294"/>
|
||||
<text class="t2" x="418.0" y="310" text-anchor="middle" font-size="11">100</text>
|
||||
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="294"/>
|
||||
<text class="t2" x="458.0" y="310" text-anchor="middle" font-size="11">200</text>
|
||||
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="294"/>
|
||||
<text class="t2" x="498.0" y="310" text-anchor="middle" font-size="11">300</text>
|
||||
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="294"/>
|
||||
<text class="t2" x="538.0" y="310" text-anchor="middle" font-size="11">400</text>
|
||||
<text class="t2" x="538.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
|
||||
<path class="f-number" d="M378.0,84.0 H455.6 Q459.6,84.0 459.6,88.0 V92.0 Q459.6,96.0 455.6,96.0 H378.0 Z"><title>Number (Large330): 204 ns per operation</title></path>
|
||||
<text class="t2" x="464.6" y="94.0" font-size="11">204</text>
|
||||
<path class="f-ieee64" d="M378.0,108.0 H378.0 Q381.4,108.0 381.4,111.4 V116.6 Q381.4,120.0 378.0,120.0 H378.0 Z"><title>Boost decimal64: 8.5 ns per operation</title></path>
|
||||
<text class="t2" x="386.4" y="118.0" font-size="11">8.5</text>
|
||||
<path class="f-ieee64" d="M378.0,132.0 H380.6 Q384.6,132.0 384.6,136.0 V140.0 Q384.6,144.0 380.6,144.0 H378.0 Z"><title>Boost decimal_fast64: 17 ns per operation</title></path>
|
||||
<text class="t2" x="389.6" y="142.0" font-size="11">17</text>
|
||||
<path class="f-ieee64" d="M378.0,156.0 H381.3 Q385.3,156.0 385.3,160.0 V164.0 Q385.3,168.0 381.3,168.0 H378.0 Z"><title>Intel BID64: 18 ns per operation</title></path>
|
||||
<text class="t2" x="390.3" y="166.0" font-size="11">18</text>
|
||||
<path class="f-wide" d="M378.0,180.0 H403.1 Q407.1,180.0 407.1,184.0 V188.0 Q407.1,192.0 403.1,192.0 H378.0 Z"><title>Boost decimal128: 73 ns per operation</title></path>
|
||||
<text class="t2" x="412.1" y="190.0" font-size="11">73</text>
|
||||
<path class="f-wide" d="M378.0,204.0 H421.7 Q425.7,204.0 425.7,208.0 V212.0 Q425.7,216.0 421.7,216.0 H378.0 Z"><title>Boost decimal_fast128: 119 ns per operation</title></path>
|
||||
<text class="t2" x="430.7" y="214.0" font-size="11">119</text>
|
||||
<path class="f-wide" d="M378.0,228.0 H389.3 Q393.3,228.0 393.3,232.0 V236.0 Q393.3,240.0 389.3,240.0 H378.0 Z"><title>Intel BID128: 38 ns per operation</title></path>
|
||||
<text class="t2" x="398.3" y="238.0" font-size="11">38</text>
|
||||
<path class="f-wide" d="M378.0,252.0 H386.4 Q390.4,252.0 390.4,256.0 V260.0 Q390.4,264.0 386.4,264.0 H378.0 Z"><title>mpdecimal 34: 31 ns per operation</title></path>
|
||||
<text class="t2" x="395.4" y="262.0" font-size="11">31</text>
|
||||
<path class="f-wide" d="M378.0,276.0 H380.1 Q384.1,276.0 384.1,280.0 V284.0 Q384.1,288.0 380.1,288.0 H378.0 Z"><title>mpdecimal 38: 15 ns per operation</title></path>
|
||||
<text class="t2" x="389.1" y="286.0" font-size="11">15</text>
|
||||
<line class="base" x1="378" y1="78" x2="378" y2="294"/>
|
||||
<text class="t1" x="606" y="64" font-weight="600">Divide</text>
|
||||
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="294"/>
|
||||
<text class="t2" x="606.0" y="310" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="646.0" y1="78" x2="646.0" y2="294"/>
|
||||
<text class="t2" x="646.0" y="310" text-anchor="middle" font-size="11">100</text>
|
||||
<line class="grid" x1="686.0" y1="78" x2="686.0" y2="294"/>
|
||||
<text class="t2" x="686.0" y="310" text-anchor="middle" font-size="11">200</text>
|
||||
<line class="grid" x1="726.0" y1="78" x2="726.0" y2="294"/>
|
||||
<text class="t2" x="726.0" y="310" text-anchor="middle" font-size="11">300</text>
|
||||
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="294"/>
|
||||
<text class="t2" x="766.0" y="310" text-anchor="middle" font-size="11">400</text>
|
||||
<text class="t2" x="766.0" y="322" text-anchor="end" font-size="11">ns per operation</text>
|
||||
<path class="f-number" d="M606.0,84.0 H627.7 Q631.7,84.0 631.7,88.0 V92.0 Q631.7,96.0 627.7,96.0 H606.0 Z"><title>Number (Large330): 64 ns per operation</title></path>
|
||||
<text class="t2" x="636.7" y="94.0" font-size="11">64</text>
|
||||
<path class="f-ieee64" d="M606.0,108.0 H615.7 Q619.7,108.0 619.7,112.0 V116.0 Q619.7,120.0 615.7,120.0 H606.0 Z"><title>Boost decimal64: 34 ns per operation</title></path>
|
||||
<text class="t2" x="624.7" y="118.0" font-size="11">34</text>
|
||||
<path class="f-ieee64" d="M606.0,132.0 H616.8 Q620.8,132.0 620.8,136.0 V140.0 Q620.8,144.0 616.8,144.0 H606.0 Z"><title>Boost decimal_fast64: 37 ns per operation</title></path>
|
||||
<text class="t2" x="625.8" y="142.0" font-size="11">37</text>
|
||||
<path class="f-ieee64" d="M606.0,156.0 H610.0 Q614.0,156.0 614.0,160.0 V164.0 Q614.0,168.0 610.0,168.0 H606.0 Z"><title>Intel BID64: 20 ns per operation</title></path>
|
||||
<text class="t2" x="619.0" y="166.0" font-size="11">20</text>
|
||||
<path class="f-wide" d="M606.0,180.0 H726.2 Q730.2,180.0 730.2,184.0 V188.0 Q730.2,192.0 726.2,192.0 H606.0 Z"><title>Boost decimal128: 310 ns per operation</title></path>
|
||||
<text class="t2" x="735.2" y="190.0" font-size="11">310</text>
|
||||
<path class="f-wide" d="M606.0,204.0 H648.7 Q652.7,204.0 652.7,208.0 V212.0 Q652.7,216.0 648.7,216.0 H606.0 Z"><title>Boost decimal_fast128: 117 ns per operation</title></path>
|
||||
<text class="t2" x="657.7" y="214.0" font-size="11">117</text>
|
||||
<path class="f-wide" d="M606.0,228.0 H627.4 Q631.4,228.0 631.4,232.0 V236.0 Q631.4,240.0 627.4,240.0 H606.0 Z"><title>Intel BID128: 63 ns per operation</title></path>
|
||||
<text class="t2" x="636.4" y="238.0" font-size="11">63</text>
|
||||
<path class="f-wide" d="M606.0,252.0 H623.2 Q627.2,252.0 627.2,256.0 V260.0 Q627.2,264.0 623.2,264.0 H606.0 Z"><title>mpdecimal 34: 53 ns per operation</title></path>
|
||||
<text class="t2" x="632.2" y="262.0" font-size="11">53</text>
|
||||
<path class="f-wide" d="M606.0,276.0 H624.5 Q628.5,276.0 628.5,280.0 V284.0 Q628.5,288.0 624.5,288.0 H606.0 Z"><title>mpdecimal 38: 56 ns per operation</title></path>
|
||||
<text class="t2" x="633.5" y="286.0" font-size="11">56</text>
|
||||
<line class="base" x1="606" y1="78" x2="606" y2="294"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 10 KiB |
105
docs/images/number/number-loan-amortize360.svg
Normal file
105
docs/images/number/number-loan-amortize360.svg
Normal file
@@ -0,0 +1,105 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 834 276" width="834" height="276" role="img" aria-label="360-payment amortization: correct digits (higher is better) and cost (lower is faster)">
|
||||
<style>
|
||||
text{font-family:system-ui, -apple-system, sans-serif;font-size:12px}
|
||||
.bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}.grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}.f-number{fill:#2a78d6}.f-ieee64{fill:#eb6834}.f-wide{fill:#1baf7a}
|
||||
@media (prefers-color-scheme: dark){.bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}.grid{stroke:#383835}.base{stroke:#5e5d58}.f-number{fill:#3987e5}.f-ieee64{fill:#d95926}.f-wide{fill:#199e70}}
|
||||
</style>
|
||||
<rect class="bg" width="834" height="276"/>
|
||||
<text class="t1" x="0" y="18" font-size="15" font-weight="600">360-payment amortization: correct digits (higher is better) and cost (lower is faster)</text>
|
||||
<rect class="f-number" x="0.0" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="18.0" y="42">Number today (19 digits)</text>
|
||||
<rect class="f-ieee64" x="214.8" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="232.8" y="42">16-digit types</text>
|
||||
<rect class="f-wide" x="357.6" y="32" width="12" height="12" rx="3"/>
|
||||
<text class="t2" x="375.6" y="42">34- and 38-digit types</text>
|
||||
<text class="t1" x="0" y="94.0">Number (Large330)</text>
|
||||
<text class="t1" x="0" y="118.0">Boost decimal64</text>
|
||||
<text class="t1" x="0" y="142.0">Intel BID64</text>
|
||||
<text class="t1" x="0" y="166.0">Boost decimal128</text>
|
||||
<text class="t1" x="0" y="190.0">Intel BID128</text>
|
||||
<text class="t1" x="0" y="214.0">mpdecimal 34</text>
|
||||
<text class="t1" x="0" y="238.0">mpdecimal 38</text>
|
||||
<text class="t1" x="150" y="64" font-weight="600">Correct digits, worst case</text>
|
||||
<line class="grid" x1="150.0" y1="78" x2="150.0" y2="246"/>
|
||||
<text class="t2" x="150.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="190.0" y1="78" x2="190.0" y2="246"/>
|
||||
<text class="t2" x="190.0" y="262" text-anchor="middle" font-size="11">10</text>
|
||||
<line class="grid" x1="230.0" y1="78" x2="230.0" y2="246"/>
|
||||
<text class="t2" x="230.0" y="262" text-anchor="middle" font-size="11">20</text>
|
||||
<line class="grid" x1="270.0" y1="78" x2="270.0" y2="246"/>
|
||||
<text class="t2" x="270.0" y="262" text-anchor="middle" font-size="11">30</text>
|
||||
<line class="grid" x1="310.0" y1="78" x2="310.0" y2="246"/>
|
||||
<text class="t2" x="310.0" y="262" text-anchor="middle" font-size="11">40</text>
|
||||
<text class="t2" x="310.0" y="274" text-anchor="end" font-size="11">significant digits</text>
|
||||
<path class="f-number" d="M150.0,84.0 H202.3 Q206.3,84.0 206.3,88.0 V92.0 Q206.3,96.0 202.3,96.0 H150.0 Z"><title>Number (Large330): 14.1 significant digits</title></path>
|
||||
<text class="t2" x="211.3" y="94.0" font-size="11">14.1</text>
|
||||
<path class="f-ieee64" d="M150.0,108.0 H190.8 Q194.8,108.0 194.8,112.0 V116.0 Q194.8,120.0 190.8,120.0 H150.0 Z"><title>Boost decimal64: 11.2 significant digits</title></path>
|
||||
<text class="t2" x="199.8" y="118.0" font-size="11">11.2</text>
|
||||
<path class="f-ieee64" d="M150.0,132.0 H190.8 Q194.8,132.0 194.8,136.0 V140.0 Q194.8,144.0 190.8,144.0 H150.0 Z"><title>Intel BID64: 11.2 significant digits</title></path>
|
||||
<text class="t2" x="199.8" y="142.0" font-size="11">11.2</text>
|
||||
<path class="f-wide" d="M150.0,156.0 H262.5 Q266.5,156.0 266.5,160.0 V164.0 Q266.5,168.0 262.5,168.0 H150.0 Z"><title>Boost decimal128: 29.1 significant digits</title></path>
|
||||
<text class="t2" x="271.5" y="166.0" font-size="11">29.1</text>
|
||||
<path class="f-wide" d="M150.0,180.0 H262.5 Q266.5,180.0 266.5,184.0 V188.0 Q266.5,192.0 262.5,192.0 H150.0 Z"><title>Intel BID128: 29.1 significant digits</title></path>
|
||||
<text class="t2" x="271.5" y="190.0" font-size="11">29.1</text>
|
||||
<path class="f-wide" d="M150.0,204.0 H262.6 Q266.6,204.0 266.6,208.0 V212.0 Q266.6,216.0 262.6,216.0 H150.0 Z"><title>mpdecimal 34: 29.1 significant digits</title></path>
|
||||
<text class="t2" x="271.6" y="214.0" font-size="11">29.1</text>
|
||||
<path class="f-wide" d="M150.0,228.0 H277.9 Q281.9,228.0 281.9,232.0 V236.0 Q281.9,240.0 277.9,240.0 H150.0 Z"><title>mpdecimal 38: 33.0 significant digits</title></path>
|
||||
<text class="t2" x="286.9" y="238.0" font-size="11">33.0</text>
|
||||
<line class="base" x1="150" y1="78" x2="150" y2="246"/>
|
||||
<text class="t1" x="378" y="64" font-weight="600">Correct digits, median</text>
|
||||
<line class="grid" x1="378.0" y1="78" x2="378.0" y2="246"/>
|
||||
<text class="t2" x="378.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="418.0" y1="78" x2="418.0" y2="246"/>
|
||||
<text class="t2" x="418.0" y="262" text-anchor="middle" font-size="11">10</text>
|
||||
<line class="grid" x1="458.0" y1="78" x2="458.0" y2="246"/>
|
||||
<text class="t2" x="458.0" y="262" text-anchor="middle" font-size="11">20</text>
|
||||
<line class="grid" x1="498.0" y1="78" x2="498.0" y2="246"/>
|
||||
<text class="t2" x="498.0" y="262" text-anchor="middle" font-size="11">30</text>
|
||||
<line class="grid" x1="538.0" y1="78" x2="538.0" y2="246"/>
|
||||
<text class="t2" x="538.0" y="262" text-anchor="middle" font-size="11">40</text>
|
||||
<text class="t2" x="538.0" y="274" text-anchor="end" font-size="11">significant digits</text>
|
||||
<path class="f-number" d="M378.0,84.0 H437.1 Q441.1,84.0 441.1,88.0 V92.0 Q441.1,96.0 437.1,96.0 H378.0 Z"><title>Number (Large330): 15.8 significant digits</title></path>
|
||||
<text class="t2" x="446.1" y="94.0" font-size="11">15.8</text>
|
||||
<path class="f-ieee64" d="M378.0,108.0 H425.4 Q429.4,108.0 429.4,112.0 V116.0 Q429.4,120.0 425.4,120.0 H378.0 Z"><title>Boost decimal64: 12.8 significant digits</title></path>
|
||||
<text class="t2" x="434.4" y="118.0" font-size="11">12.8</text>
|
||||
<path class="f-ieee64" d="M378.0,132.0 H425.4 Q429.4,132.0 429.4,136.0 V140.0 Q429.4,144.0 425.4,144.0 H378.0 Z"><title>Intel BID64: 12.8 significant digits</title></path>
|
||||
<text class="t2" x="434.4" y="142.0" font-size="11">12.8</text>
|
||||
<path class="f-wide" d="M378.0,156.0 H497.2 Q501.2,156.0 501.2,160.0 V164.0 Q501.2,168.0 497.2,168.0 H378.0 Z"><title>Boost decimal128: 30.8 significant digits</title></path>
|
||||
<text class="t2" x="506.2" y="166.0" font-size="11">30.8</text>
|
||||
<path class="f-wide" d="M378.0,180.0 H497.2 Q501.2,180.0 501.2,184.0 V188.0 Q501.2,192.0 497.2,192.0 H378.0 Z"><title>Intel BID128: 30.8 significant digits</title></path>
|
||||
<text class="t2" x="506.2" y="190.0" font-size="11">30.8</text>
|
||||
<path class="f-wide" d="M378.0,204.0 H497.2 Q501.2,204.0 501.2,208.0 V212.0 Q501.2,216.0 497.2,216.0 H378.0 Z"><title>mpdecimal 34: 30.8 significant digits</title></path>
|
||||
<text class="t2" x="506.2" y="214.0" font-size="11">30.8</text>
|
||||
<path class="f-wide" d="M378.0,228.0 H513.3 Q517.3,228.0 517.3,232.0 V236.0 Q517.3,240.0 513.3,240.0 H378.0 Z"><title>mpdecimal 38: 34.8 significant digits</title></path>
|
||||
<text class="t2" x="522.3" y="238.0" font-size="11">34.8</text>
|
||||
<line class="base" x1="378" y1="78" x2="378" y2="246"/>
|
||||
<text class="t1" x="606" y="64" font-weight="600">Cost</text>
|
||||
<line class="grid" x1="606.0" y1="78" x2="606.0" y2="246"/>
|
||||
<text class="t2" x="606.0" y="262" text-anchor="middle" font-size="11">0</text>
|
||||
<line class="grid" x1="638.0" y1="78" x2="638.0" y2="246"/>
|
||||
<text class="t2" x="638.0" y="262" text-anchor="middle" font-size="11">25</text>
|
||||
<line class="grid" x1="670.0" y1="78" x2="670.0" y2="246"/>
|
||||
<text class="t2" x="670.0" y="262" text-anchor="middle" font-size="11">50</text>
|
||||
<line class="grid" x1="702.0" y1="78" x2="702.0" y2="246"/>
|
||||
<text class="t2" x="702.0" y="262" text-anchor="middle" font-size="11">75</text>
|
||||
<line class="grid" x1="734.0" y1="78" x2="734.0" y2="246"/>
|
||||
<text class="t2" x="734.0" y="262" text-anchor="middle" font-size="11">100</text>
|
||||
<line class="grid" x1="766.0" y1="78" x2="766.0" y2="246"/>
|
||||
<text class="t2" x="766.0" y="262" text-anchor="middle" font-size="11">125</text>
|
||||
<text class="t2" x="766.0" y="274" text-anchor="end" font-size="11">µs per schedule</text>
|
||||
<path class="f-number" d="M606.0,84.0 H748.1 Q752.1,84.0 752.1,88.0 V92.0 Q752.1,96.0 748.1,96.0 H606.0 Z"><title>Number (Large330): 114 µs per schedule</title></path>
|
||||
<text class="t2" x="757.1" y="94.0" font-size="11">114</text>
|
||||
<path class="f-ieee64" d="M606.0,108.0 H627.0 Q631.0,108.0 631.0,112.0 V116.0 Q631.0,120.0 627.0,120.0 H606.0 Z"><title>Boost decimal64: 20 µs per schedule</title></path>
|
||||
<text class="t2" x="636.0" y="118.0" font-size="11">20</text>
|
||||
<path class="f-ieee64" d="M606.0,132.0 H625.5 Q629.5,132.0 629.5,136.0 V140.0 Q629.5,144.0 625.5,144.0 H606.0 Z"><title>Intel BID64: 18 µs per schedule</title></path>
|
||||
<text class="t2" x="634.5" y="142.0" font-size="11">18</text>
|
||||
<path class="f-wide" d="M606.0,156.0 H726.1 Q730.1,156.0 730.1,160.0 V164.0 Q730.1,168.0 726.1,168.0 H606.0 Z"><title>Boost decimal128: 97 µs per schedule</title></path>
|
||||
<text class="t2" x="735.1" y="166.0" font-size="11">97</text>
|
||||
<path class="f-wide" d="M606.0,180.0 H676.9 Q680.9,180.0 680.9,184.0 V188.0 Q680.9,192.0 676.9,192.0 H606.0 Z"><title>Intel BID128: 58 µs per schedule</title></path>
|
||||
<text class="t2" x="685.9" y="190.0" font-size="11">58</text>
|
||||
<path class="f-wide" d="M606.0,204.0 H672.7 Q676.7,204.0 676.7,208.0 V212.0 Q676.7,216.0 672.7,216.0 H606.0 Z"><title>mpdecimal 34: 55 µs per schedule</title></path>
|
||||
<text class="t2" x="681.7" y="214.0" font-size="11">55</text>
|
||||
<path class="f-wide" d="M606.0,228.0 H659.8 Q663.8,228.0 663.8,232.0 V236.0 Q663.8,240.0 659.8,240.0 H606.0 Z"><title>mpdecimal 38: 45 µs per schedule</title></path>
|
||||
<text class="t2" x="668.8" y="238.0" font-size="11">45</text>
|
||||
<line class="base" x1="606" y1="78" x2="606" y2="246"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 9.3 KiB |
@@ -20,3 +20,103 @@ target_link_libraries(
|
||||
xrpl_add_benchmark(nodestore)
|
||||
target_link_libraries(xrpl.bench.nodestore PRIVATE xrpl.imports.bench)
|
||||
add_dependencies(xrpl.benchmarks xrpl.bench.nodestore)
|
||||
|
||||
# Number vs. decimal floating-point libraries. See docs/NumberDecimalBenchmark.md.
|
||||
# Boost.Decimal is header-only and ships with Boost >= 1.91.
|
||||
xrpl_add_benchmark(number)
|
||||
target_link_libraries(xrpl.bench.number PRIVATE xrpl.imports.bench)
|
||||
add_dependencies(xrpl.benchmarks xrpl.bench.number)
|
||||
|
||||
# Benchmark a 128-bit Number prototype if one exists (see number/Number128.h).
|
||||
# CONFIGURE_DEPENDS re-checks the glob on every build, so adding or removing
|
||||
# the header is enough to reconfigure and rebuild.
|
||||
file(
|
||||
GLOB number128_header
|
||||
CONFIGURE_DEPENDS
|
||||
"${CMAKE_SOURCE_DIR}/include/xrpl/basics/Number128.h"
|
||||
)
|
||||
if(number128_header)
|
||||
set(XRPL_BENCH_NUMBER128 ON)
|
||||
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_NUMBER128)
|
||||
endif()
|
||||
|
||||
if(bench_intel_dfp)
|
||||
# Intel Decimal Floating-Point Math Library, fetched and built with its own
|
||||
# makefile for local benchmarking only. If it is ever used beyond that, it
|
||||
# should become a Conan recipe instead.
|
||||
#
|
||||
# CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0: arguments by value, rounding
|
||||
# mode and status flags passed to every call. number/IntelBid.h defines the
|
||||
# matching DECIMAL_* macros. The upstream makefile adds no optimization
|
||||
# flags of its own, so pass -O3 to match the Release build.
|
||||
include(ExternalProject)
|
||||
find_program(XRPL_MAKE_EXECUTABLE NAMES gmake make REQUIRED)
|
||||
set(intel_dfp_dir "${CMAKE_CURRENT_BINARY_DIR}/intel_dfp")
|
||||
ExternalProject_Add(
|
||||
intel_dfp
|
||||
URL https://netlib.org/misc/intel/IntelRDFPMathLib20U4.tar.gz
|
||||
URL_HASH
|
||||
SHA256=1df86132e7a31fd74d784fee1c679b21a088f73a8ec979cfaf784c200392e125
|
||||
DOWNLOAD_EXTRACT_TIMESTAMP TRUE
|
||||
SOURCE_DIR "${intel_dfp_dir}"
|
||||
BUILD_IN_SOURCE TRUE
|
||||
CONFIGURE_COMMAND ""
|
||||
BUILD_COMMAND
|
||||
${XRPL_MAKE_EXECUTABLE} -C LIBRARY CC=${CMAKE_C_COMPILER}
|
||||
CALL_BY_REF=0 GLOBAL_RND=0 GLOBAL_FLAGS=0 UNCHANGED_BINARY_FLAGS=0
|
||||
"CFLAGS_OPT=-O3 -fPIC"
|
||||
INSTALL_COMMAND ""
|
||||
BUILD_BYPRODUCTS "${intel_dfp_dir}/LIBRARY/libbid.a"
|
||||
)
|
||||
# Imported targets require their include directory to exist at generate time.
|
||||
file(MAKE_DIRECTORY "${intel_dfp_dir}/LIBRARY/src")
|
||||
add_library(xrpl.imports.intel_dfp STATIC IMPORTED)
|
||||
set_target_properties(
|
||||
xrpl.imports.intel_dfp
|
||||
PROPERTIES
|
||||
IMPORTED_LOCATION "${intel_dfp_dir}/LIBRARY/libbid.a"
|
||||
INTERFACE_INCLUDE_DIRECTORIES "${intel_dfp_dir}/LIBRARY/src"
|
||||
)
|
||||
add_dependencies(xrpl.imports.intel_dfp intel_dfp)
|
||||
|
||||
target_link_libraries(xrpl.bench.number PRIVATE xrpl.imports.intel_dfp)
|
||||
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_INTEL_DFP)
|
||||
endif()
|
||||
|
||||
if(bench_mpdecimal)
|
||||
find_package(mpdecimal REQUIRED)
|
||||
target_link_libraries(xrpl.bench.number PRIVATE mpdecimal::libmpdecimal)
|
||||
target_compile_definitions(xrpl.bench.number PRIVATE XRPL_BENCH_MPDECIMAL)
|
||||
|
||||
# The divergence harness is a plain executable, not a Google Benchmark, so
|
||||
# it is defined here rather than with xrpl_add_benchmark. It shares the
|
||||
# adapters in number/.
|
||||
add_executable(xrpl.bench.number_diff number_diff/main.cpp)
|
||||
isolate_headers(
|
||||
xrpl.bench.number_diff
|
||||
"${CMAKE_SOURCE_DIR}/src"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/number"
|
||||
PRIVATE
|
||||
)
|
||||
target_link_libraries(
|
||||
xrpl.bench.number_diff
|
||||
PRIVATE xrpl.libxrpl mpdecimal::libmpdecimal
|
||||
)
|
||||
if(bench_intel_dfp)
|
||||
target_link_libraries(
|
||||
xrpl.bench.number_diff
|
||||
PRIVATE xrpl.imports.intel_dfp
|
||||
)
|
||||
target_compile_definitions(
|
||||
xrpl.bench.number_diff
|
||||
PRIVATE XRPL_BENCH_INTEL_DFP
|
||||
)
|
||||
endif()
|
||||
if(XRPL_BENCH_NUMBER128)
|
||||
target_compile_definitions(
|
||||
xrpl.bench.number_diff
|
||||
PRIVATE XRPL_BENCH_NUMBER128
|
||||
)
|
||||
endif()
|
||||
add_dependencies(xrpl.benchmarks xrpl.bench.number_diff)
|
||||
endif()
|
||||
|
||||
263
src/benchmarks/libxrpl/number/Arithmetic.cpp
Normal file
263
src/benchmarks/libxrpl/number/Arithmetic.cpp
Normal file
@@ -0,0 +1,263 @@
|
||||
// Single-operation benchmarks for xrpl::Number and the decimal floating-point
|
||||
// types it is being compared against. See docs/NumberDecimalBenchmark.md.
|
||||
//
|
||||
// Benchmark names are "<operation>/<dataset>/<type>", e.g.
|
||||
// "mul/full/Number.Large330", so they can be filtered with
|
||||
// --benchmark_filter='^mul/.*/(Number|BoostDecimalFast128)'.
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
#include <benchmarks/libxrpl/number/BoostDecimal.h>
|
||||
#include <benchmarks/libxrpl/number/Double.h>
|
||||
#include <benchmarks/libxrpl/number/Number128.h>
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
#include <benchmarks/libxrpl/number/Operands.h>
|
||||
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
#include <benchmarks/libxrpl/number/MpDecimal.h>
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
#include <benchmarks/libxrpl/number/IntelBid.h>
|
||||
#endif
|
||||
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <format>
|
||||
#include <functional>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
namespace {
|
||||
|
||||
constexpr std::uint64_t kSeedLhs = 0x5eed'0001;
|
||||
constexpr std::uint64_t kSeedRhs = 0x5eed'0002;
|
||||
|
||||
/**
|
||||
* Environment for types that need no per-benchmark setup.
|
||||
*/
|
||||
struct NoEnv
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* Runs Number under a specific mantissa scale. Operands are materialized after
|
||||
* the guard is in place, so they are normalized for that scale.
|
||||
*/
|
||||
template <MantissaRange::MantissaScale Scale>
|
||||
struct NumberScaleEnv
|
||||
{
|
||||
NumberMantissaScaleGuard guard{Scale};
|
||||
};
|
||||
|
||||
/**
|
||||
* Independent operations over a batch: measures throughput.
|
||||
*/
|
||||
template <NumberLike T, class Env, class Op>
|
||||
auto
|
||||
binaryThroughput(std::vector<RawOperand> lhs, std::vector<RawOperand> rhs, Op op)
|
||||
{
|
||||
return [lhs = std::move(lhs), rhs = std::move(rhs), op](benchmark::State& state) {
|
||||
[[maybe_unused]] Env const env{};
|
||||
auto const a = materialize<T>(lhs);
|
||||
auto const b = materialize<T>(rhs);
|
||||
for (auto _ : state)
|
||||
{
|
||||
for (std::size_t i = 0; i < a.size(); ++i)
|
||||
{
|
||||
auto r = op(a[i], b[i]);
|
||||
benchmark::DoNotOptimize(r);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
|
||||
};
|
||||
}
|
||||
|
||||
template <NumberLike T, class Env, class Op>
|
||||
auto
|
||||
unaryThroughput(std::vector<RawOperand> operands, Op op)
|
||||
{
|
||||
return [operands = std::move(operands), op](benchmark::State& state) {
|
||||
[[maybe_unused]] Env const env{};
|
||||
auto const a = materialize<T>(operands);
|
||||
for (auto _ : state)
|
||||
{
|
||||
for (auto const& x : a)
|
||||
{
|
||||
auto r = op(x);
|
||||
benchmark::DoNotOptimize(r);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Construction from (mantissa, exponent), the path every STNumber/STAmount read takes.
|
||||
*/
|
||||
template <NumberLike T, class Env>
|
||||
auto
|
||||
constructThroughput(std::vector<RawOperand> operands)
|
||||
{
|
||||
return [operands = std::move(operands)](benchmark::State& state) {
|
||||
[[maybe_unused]] Env const env{};
|
||||
for (auto _ : state)
|
||||
{
|
||||
for (auto const& raw : operands)
|
||||
{
|
||||
T x{raw.mantissa, raw.exponent};
|
||||
benchmark::DoNotOptimize(x);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * operands.size()));
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Each result feeds the next operation: measures latency.
|
||||
*/
|
||||
template <NumberLike T, class Env, class Op>
|
||||
auto
|
||||
chainLatency(std::vector<RawOperand> operands, Op op)
|
||||
{
|
||||
return [operands = std::move(operands), op](benchmark::State& state) {
|
||||
[[maybe_unused]] Env const env{};
|
||||
auto const a = materialize<T>(operands);
|
||||
for (auto _ : state)
|
||||
{
|
||||
T acc{std::int64_t{1}};
|
||||
for (auto const& x : a)
|
||||
acc = op(acc, x);
|
||||
benchmark::DoNotOptimize(acc);
|
||||
}
|
||||
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * a.size()));
|
||||
};
|
||||
}
|
||||
|
||||
template <class T, class Fn>
|
||||
void
|
||||
add(std::string_view op, std::string_view data, std::string_view subject, Fn fn)
|
||||
{
|
||||
benchmark::RegisterBenchmark(
|
||||
std::format("{}/{}/{}", op, data, subject), [fn = std::move(fn)](benchmark::State& state) {
|
||||
fn(state);
|
||||
state.counters["sizeof"] = sizeof(T);
|
||||
});
|
||||
}
|
||||
|
||||
template <NumberLike T, class Env>
|
||||
void
|
||||
registerSubject(std::string_view subject)
|
||||
{
|
||||
constexpr auto kGeneral =
|
||||
std::to_array({Dataset::Full, Dataset::Iou, Dataset::Drops, Dataset::Mpt});
|
||||
|
||||
for (auto const d : kGeneral)
|
||||
{
|
||||
auto const name = toString(d);
|
||||
auto const lhs = makeRaw(d, kSeedLhs);
|
||||
auto const rhs = makeRaw(d, kSeedRhs);
|
||||
|
||||
add<T>("add", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::plus<>{}));
|
||||
add<T>("sub", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::minus<>{}));
|
||||
add<T>("mul", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::multiplies<>{}));
|
||||
add<T>("div", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::divides<>{}));
|
||||
add<T>("lt", name, subject, binaryThroughput<T, Env>(lhs, rhs, std::less<>{}));
|
||||
add<T>("construct", name, subject, constructThroughput<T, Env>(lhs));
|
||||
add<T>("to_string", name, subject, unaryThroughput<T, Env>(lhs, [](T const& x) {
|
||||
return to_string(x);
|
||||
}));
|
||||
add<T>("root2", name, subject, unaryThroughput<T, Env>(lhs, [](T const& x) {
|
||||
return root2(x);
|
||||
}));
|
||||
}
|
||||
|
||||
// Only integer-valued datasets fit in int64 without overflow.
|
||||
for (auto const d : {Dataset::Drops, Dataset::Mpt})
|
||||
{
|
||||
add<T>(
|
||||
"to_int64",
|
||||
toString(d),
|
||||
subject,
|
||||
unaryThroughput<T, Env>(
|
||||
makeRaw(d, kSeedLhs), [](T const& x) { return static_cast<std::int64_t>(x); }));
|
||||
}
|
||||
|
||||
for (int const gap : {0, 1, 5, 18})
|
||||
{
|
||||
auto [lhs, rhs] = makeExponentGapPairs(gap, kSeedLhs);
|
||||
add<T>(
|
||||
std::format("add_gap{}", gap),
|
||||
toString(Dataset::Full),
|
||||
subject,
|
||||
binaryThroughput<T, Env>(std::move(lhs), std::move(rhs), std::plus<>{}));
|
||||
}
|
||||
|
||||
{
|
||||
auto [lhs, rhs] = makeCancellationPairs(kSeedLhs);
|
||||
add<T>(
|
||||
"sub_cancel",
|
||||
toString(Dataset::Full),
|
||||
subject,
|
||||
binaryThroughput<T, Env>(std::move(lhs), std::move(rhs), std::minus<>{}));
|
||||
}
|
||||
|
||||
// 12 and 360: monthly payments for one year and for thirty years.
|
||||
for (unsigned const n : {12u, 360u})
|
||||
{
|
||||
add<T>(
|
||||
std::format("power{}", n),
|
||||
toString(Dataset::Rate),
|
||||
subject,
|
||||
unaryThroughput<T, Env>(
|
||||
makeRaw(Dataset::Rate, kSeedLhs), [n](T const& x) { return power(x, n); }));
|
||||
}
|
||||
|
||||
add<T>(
|
||||
"add_chain",
|
||||
toString(Dataset::Full),
|
||||
subject,
|
||||
chainLatency<T, Env>(makeRaw(Dataset::Full, kSeedLhs), std::plus<>{}));
|
||||
add<T>(
|
||||
"mul_chain",
|
||||
toString(Dataset::Rate),
|
||||
subject,
|
||||
chainLatency<T, Env>(makeRaw(Dataset::Rate, kSeedLhs), std::multiplies<>{}));
|
||||
add<T>(
|
||||
"div_chain",
|
||||
toString(Dataset::Rate),
|
||||
subject,
|
||||
chainLatency<T, Env>(makeRaw(Dataset::Rate, kSeedLhs), std::divides<>{}));
|
||||
}
|
||||
|
||||
[[maybe_unused]] bool const kRegistered = [] {
|
||||
using enum MantissaRange::MantissaScale;
|
||||
registerSubject<Number, NumberScaleEnv<Small>>("Number.Small");
|
||||
registerSubject<Number, NumberScaleEnv<LargeLegacy>>("Number.LargeLegacy");
|
||||
registerSubject<Number, NumberScaleEnv<Large330>>("Number.Large330");
|
||||
registerSubject<BoostDecimal64, NoEnv>("BoostDecimal64");
|
||||
registerSubject<BoostDecimalFast64, NoEnv>("BoostDecimalFast64");
|
||||
registerSubject<BoostDecimal128, NoEnv>("BoostDecimal128");
|
||||
registerSubject<BoostDecimalFast128, NoEnv>("BoostDecimalFast128");
|
||||
registerSubject<Double, NoEnv>("Double");
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
registerSubject<MpDecimal19, NoEnv>("MpDecimal19");
|
||||
registerSubject<MpDecimal34, NoEnv>("MpDecimal34");
|
||||
registerSubject<MpDecimal38, NoEnv>("MpDecimal38");
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
registerSubject<IntelBid64, NoEnv>("IntelBid64");
|
||||
registerSubject<IntelBid128, NoEnv>("IntelBid128");
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_HAS_NUMBER128
|
||||
registerSubject<Number128, NoEnv>("Number128");
|
||||
#endif
|
||||
return true;
|
||||
}();
|
||||
|
||||
} // namespace
|
||||
} // namespace xrpl::number_bench
|
||||
269
src/benchmarks/libxrpl/number/BoostDecimal.h
Normal file
269
src/benchmarks/libxrpl/number/BoostDecimal.h
Normal file
@@ -0,0 +1,269 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <boost/decimal.hpp>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
namespace detail {
|
||||
|
||||
constexpr boost::decimal::rounding_mode
|
||||
toBoost(Number::RoundingMode mode)
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
using boost::decimal::rounding_mode;
|
||||
switch (mode)
|
||||
{
|
||||
case ToNearest:
|
||||
return rounding_mode::fe_dec_to_nearest;
|
||||
case TowardsZero:
|
||||
return rounding_mode::fe_dec_toward_zero;
|
||||
case Downward:
|
||||
return rounding_mode::fe_dec_downward;
|
||||
case Upward:
|
||||
return rounding_mode::fe_dec_upward;
|
||||
}
|
||||
return rounding_mode::fe_dec_to_nearest;
|
||||
}
|
||||
|
||||
constexpr Number::RoundingMode
|
||||
fromBoost(boost::decimal::rounding_mode mode)
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
using boost::decimal::rounding_mode;
|
||||
switch (mode)
|
||||
{
|
||||
case rounding_mode::fe_dec_toward_zero:
|
||||
return TowardsZero;
|
||||
case rounding_mode::fe_dec_downward:
|
||||
return Downward;
|
||||
case rounding_mode::fe_dec_upward:
|
||||
return Upward;
|
||||
default:
|
||||
return ToNearest;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
/**
|
||||
* Wraps a Boost.Decimal type behind Number's interface.
|
||||
*
|
||||
* Every member is a one-line inline forward, so at -O2 and above the wrapper
|
||||
* adds no cost over using the Boost type directly.
|
||||
*
|
||||
* Semantic differences from Number that the wrapper deliberately does NOT
|
||||
* hide (they are part of what we are measuring):
|
||||
* - Overflow produces infinity and division by zero produces inf/NaN
|
||||
* instead of throwing.
|
||||
* - The rounding mode is a process-wide global in Boost.Decimal, not
|
||||
* thread-local like Number's.
|
||||
* - decimal64_t / decimal_fast64_t carry 16 digits, so constructing from a
|
||||
* 19-digit int64 mantissa rounds.
|
||||
*/
|
||||
template <class D>
|
||||
class BoostDecimal final
|
||||
{
|
||||
D value_{};
|
||||
|
||||
explicit constexpr BoostDecimal(D value) noexcept : value_{value}
|
||||
{
|
||||
}
|
||||
|
||||
public:
|
||||
using rep = std::int64_t;
|
||||
using value_type = D;
|
||||
|
||||
constexpr BoostDecimal() = default;
|
||||
|
||||
// Implicit, like Number(rep).
|
||||
constexpr BoostDecimal(rep mantissa) noexcept : value_{mantissa, 0} // NOLINT
|
||||
{
|
||||
}
|
||||
|
||||
explicit constexpr BoostDecimal(rep mantissa, int exponent) noexcept
|
||||
: value_{mantissa, exponent}
|
||||
{
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr D
|
||||
value() const noexcept
|
||||
{
|
||||
return value_;
|
||||
}
|
||||
|
||||
constexpr BoostDecimal
|
||||
operator-() const noexcept
|
||||
{
|
||||
return BoostDecimal{-value_};
|
||||
}
|
||||
|
||||
constexpr BoostDecimal&
|
||||
operator+=(BoostDecimal const& x) noexcept
|
||||
{
|
||||
value_ += x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr BoostDecimal&
|
||||
operator-=(BoostDecimal const& x) noexcept
|
||||
{
|
||||
value_ -= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr BoostDecimal&
|
||||
operator*=(BoostDecimal const& x) noexcept
|
||||
{
|
||||
value_ *= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr BoostDecimal&
|
||||
operator/=(BoostDecimal const& x) noexcept
|
||||
{
|
||||
value_ /= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
operator+(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return BoostDecimal{x.value_ + y.value_};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
operator-(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return BoostDecimal{x.value_ - y.value_};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
operator*(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return BoostDecimal{x.value_ * y.value_};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
operator/(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return BoostDecimal{x.value_ / y.value_};
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator==(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return x.value_ == y.value_;
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator<(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return x.value_ < y.value_;
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator>(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return y < x;
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator<=(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return !(y < x);
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator>=(BoostDecimal const& x, BoostDecimal const& y) noexcept
|
||||
{
|
||||
return !(x < y);
|
||||
}
|
||||
|
||||
// Round to an integer using the current rounding mode, like Number.
|
||||
explicit
|
||||
operator rep() const noexcept
|
||||
{
|
||||
// Boost 1.91's llrint mishandles values that are already integers at full
|
||||
// precision (significand exponent >= 0): it returns a wrong result, or
|
||||
// divides by zero when the exponent is exactly 0. Fixed upstream (the
|
||||
// same early return as here); remove this once Boost includes the fix.
|
||||
int exponent = 0;
|
||||
boost::decimal::frexp10(value_, &exponent);
|
||||
if (exponent >= 0)
|
||||
return static_cast<rep>(value_);
|
||||
return static_cast<rep>(boost::decimal::llrint(value_));
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
getround() noexcept
|
||||
{
|
||||
return detail::fromBoost(boost::decimal::fegetround());
|
||||
}
|
||||
|
||||
// Returns the previous mode, like Number::setround.
|
||||
static Number::RoundingMode
|
||||
setround(Number::RoundingMode mode) noexcept
|
||||
{
|
||||
auto const old = getround();
|
||||
boost::decimal::fesetround(detail::toBoost(mode));
|
||||
return old;
|
||||
}
|
||||
|
||||
friend std::string
|
||||
to_string(BoostDecimal const& x)
|
||||
{
|
||||
std::array<char, 64> buffer{};
|
||||
auto const result =
|
||||
boost::decimal::to_chars(buffer.data(), buffer.data() + buffer.size(), x.value_);
|
||||
return {buffer.data(), result.ptr};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
abs(BoostDecimal const& x) noexcept
|
||||
{
|
||||
return BoostDecimal{boost::decimal::abs(x.value_)};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
power(BoostDecimal const& x, unsigned n) noexcept
|
||||
{
|
||||
return BoostDecimal{boost::decimal::pow(x.value_, n)};
|
||||
}
|
||||
|
||||
friend constexpr BoostDecimal
|
||||
root2(BoostDecimal const& x) noexcept
|
||||
{
|
||||
return BoostDecimal{boost::decimal::sqrt(x.value_)};
|
||||
}
|
||||
|
||||
friend constexpr Decomposed
|
||||
decompose(BoostDecimal const& x) noexcept
|
||||
{
|
||||
int exponent = 0;
|
||||
auto const coefficient = boost::decimal::frexp10(x.value_, &exponent);
|
||||
return {
|
||||
.negative = boost::decimal::signbit(x.value_),
|
||||
.coefficient = static_cast<unsigned __int128>(coefficient),
|
||||
.exponent = exponent};
|
||||
}
|
||||
};
|
||||
|
||||
using BoostDecimal64 = BoostDecimal<boost::decimal::decimal64_t>;
|
||||
using BoostDecimal128 = BoostDecimal<boost::decimal::decimal128_t>;
|
||||
using BoostDecimalFast64 = BoostDecimal<boost::decimal::decimal_fast64_t>;
|
||||
using BoostDecimalFast128 = BoostDecimal<boost::decimal::decimal_fast128_t>;
|
||||
|
||||
static_assert(DecomposableNumber<BoostDecimal64>);
|
||||
static_assert(DecomposableNumber<BoostDecimal128>);
|
||||
static_assert(DecomposableNumber<BoostDecimalFast64>);
|
||||
static_assert(DecomposableNumber<BoostDecimalFast128>);
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
200
src/benchmarks/libxrpl/number/Double.h
Normal file
200
src/benchmarks/libxrpl/number/Double.h
Normal file
@@ -0,0 +1,200 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <cfenv>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
/**
|
||||
* Binary `double` behind Number's interface: a hardware speed-of-light
|
||||
* baseline, not a candidate (base 2, 53-bit significand).
|
||||
*
|
||||
* It is not DecomposableNumber because a binary value has no exact short
|
||||
* decimal coefficient.
|
||||
*/
|
||||
class Double final
|
||||
{
|
||||
double value_ = 0.0;
|
||||
|
||||
explicit constexpr Double(double value) noexcept : value_{value}
|
||||
{
|
||||
}
|
||||
|
||||
public:
|
||||
using rep = std::int64_t;
|
||||
|
||||
constexpr Double() = default;
|
||||
|
||||
// Implicit, like Number(rep).
|
||||
Double(rep mantissa) noexcept : value_{static_cast<double>(mantissa)} // NOLINT
|
||||
{
|
||||
}
|
||||
|
||||
explicit Double(rep mantissa, int exponent) noexcept
|
||||
: value_{static_cast<double>(mantissa) * std::pow(10.0, exponent)}
|
||||
{
|
||||
}
|
||||
|
||||
constexpr Double
|
||||
operator-() const noexcept
|
||||
{
|
||||
return Double{-value_};
|
||||
}
|
||||
|
||||
constexpr Double&
|
||||
operator+=(Double const& x) noexcept
|
||||
{
|
||||
value_ += x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr Double&
|
||||
operator-=(Double const& x) noexcept
|
||||
{
|
||||
value_ -= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr Double&
|
||||
operator*=(Double const& x) noexcept
|
||||
{
|
||||
value_ *= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
constexpr Double&
|
||||
operator/=(Double const& x) noexcept
|
||||
{
|
||||
value_ /= x.value_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
friend constexpr Double
|
||||
operator+(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return Double{x.value_ + y.value_};
|
||||
}
|
||||
|
||||
friend constexpr Double
|
||||
operator-(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return Double{x.value_ - y.value_};
|
||||
}
|
||||
|
||||
friend constexpr Double
|
||||
operator*(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return Double{x.value_ * y.value_};
|
||||
}
|
||||
|
||||
friend constexpr Double
|
||||
operator/(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return Double{x.value_ / y.value_};
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator==(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return x.value_ == y.value_;
|
||||
}
|
||||
|
||||
friend constexpr auto
|
||||
operator<=>(Double const& x, Double const& y) noexcept
|
||||
{
|
||||
return x.value_ <=> y.value_;
|
||||
}
|
||||
|
||||
explicit
|
||||
operator rep() const noexcept
|
||||
{
|
||||
return static_cast<rep>(std::llrint(value_));
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
getround() noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
switch (std::fegetround())
|
||||
{
|
||||
case FE_TOWARDZERO:
|
||||
return TowardsZero;
|
||||
case FE_DOWNWARD:
|
||||
return Downward;
|
||||
case FE_UPWARD:
|
||||
return Upward;
|
||||
default:
|
||||
return ToNearest;
|
||||
}
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
setround(Number::RoundingMode mode) noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
auto const old = getround();
|
||||
switch (mode)
|
||||
{
|
||||
case ToNearest:
|
||||
std::fesetround(FE_TONEAREST);
|
||||
break;
|
||||
case TowardsZero:
|
||||
std::fesetround(FE_TOWARDZERO);
|
||||
break;
|
||||
case Downward:
|
||||
std::fesetround(FE_DOWNWARD);
|
||||
break;
|
||||
case Upward:
|
||||
std::fesetround(FE_UPWARD);
|
||||
break;
|
||||
}
|
||||
return old;
|
||||
}
|
||||
|
||||
friend std::string
|
||||
to_string(Double const& x)
|
||||
{
|
||||
std::ostringstream os;
|
||||
os.precision(17);
|
||||
os << x.value_;
|
||||
return os.str();
|
||||
}
|
||||
|
||||
friend Double
|
||||
abs(Double const& x) noexcept
|
||||
{
|
||||
return Double{std::fabs(x.value_)};
|
||||
}
|
||||
|
||||
// Repeated squaring, the same algorithm as xrpl::power(Number, unsigned).
|
||||
friend constexpr Double
|
||||
power(Double const& x, unsigned n) noexcept
|
||||
{
|
||||
double result = 1.0;
|
||||
double base = x.value_;
|
||||
for (; n != 0; n >>= 1)
|
||||
{
|
||||
if ((n & 1u) != 0)
|
||||
result *= base;
|
||||
base *= base;
|
||||
}
|
||||
return Double{result};
|
||||
}
|
||||
|
||||
friend Double
|
||||
root2(Double const& x) noexcept
|
||||
{
|
||||
return Double{std::sqrt(x.value_)};
|
||||
}
|
||||
};
|
||||
|
||||
static_assert(NumberLike<Double>);
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
377
src/benchmarks/libxrpl/number/IntelBid.h
Normal file
377
src/benchmarks/libxrpl/number/IntelBid.h
Normal file
@@ -0,0 +1,377 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
// The library must be built with the same three settings (see the
|
||||
// bench_intel_dfp block in src/benchmarks/libxrpl/CMakeLists.txt): arguments
|
||||
// by value, rounding mode passed to each call, status flags passed to each
|
||||
// call. That is the configuration closest to Number, whose rounding mode is a
|
||||
// per-thread setting rather than process-global state.
|
||||
#define DECIMAL_CALL_BY_REFERENCE 0 // NOLINT(cppcoreguidelines-macro-usage)
|
||||
#define DECIMAL_GLOBAL_ROUNDING 0 // NOLINT(cppcoreguidelines-macro-usage)
|
||||
#define DECIMAL_GLOBAL_EXCEPTION_FLAGS 0 // NOLINT(cppcoreguidelines-macro-usage)
|
||||
#include <bid_conf.h>
|
||||
#include <bid_functions.h>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
namespace detail {
|
||||
|
||||
/**
|
||||
* One function table per interchange width, so IntelBid is written once.
|
||||
*/
|
||||
struct Bid64Ops
|
||||
{
|
||||
using Value = BID_UINT64;
|
||||
static constexpr unsigned kDigits = 16;
|
||||
static constexpr int kBias = 398;
|
||||
|
||||
// clang-format off
|
||||
static Value add(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_add(x, y, r, f); }
|
||||
static Value sub(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_sub(x, y, r, f); }
|
||||
static Value mul(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_mul(x, y, r, f); }
|
||||
static Value div(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_div(x, y, r, f); }
|
||||
static Value sqrt(Value x, _IDEC_round r, _IDEC_flags* f) { return bid64_sqrt(x, r, f); }
|
||||
static Value pow(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid64_pow(x, y, r, f); }
|
||||
static Value fromInt64(std::int64_t x, _IDEC_round r, _IDEC_flags* f) { return bid64_from_int64(x, r, f); }
|
||||
static Value scalbn(Value x, int n, _IDEC_round r, _IDEC_flags* f) { return bid64_scalbn(x, n, r, f); }
|
||||
static Value negate(Value x) { return bid64_negate(x); }
|
||||
static Value abs(Value x) { return bid64_abs(x); }
|
||||
static bool less(Value x, Value y, _IDEC_flags* f) { return bid64_quiet_less(x, y, f) != 0; }
|
||||
static bool equal(Value x, Value y, _IDEC_flags* f) { return bid64_quiet_equal(x, y, f) != 0; }
|
||||
static std::int64_t toInt64Nearest(Value x, _IDEC_flags* f) { return bid64_to_int64_rnint(x, f); }
|
||||
static std::int64_t toInt64TowardZero(Value x, _IDEC_flags* f) { return bid64_to_int64_int(x, f); }
|
||||
static std::int64_t toInt64Floor(Value x, _IDEC_flags* f) { return bid64_to_int64_floor(x, f); }
|
||||
static std::int64_t toInt64Ceil(Value x, _IDEC_flags* f) { return bid64_to_int64_ceil(x, f); }
|
||||
static void toString(char* s, Value x, _IDEC_flags* f) { bid64_to_string(s, x, f); }
|
||||
// clang-format on
|
||||
|
||||
/**
|
||||
* Decodes the BID64 layout. Precondition: finite and canonical.
|
||||
*/
|
||||
static Decomposed
|
||||
decompose(Value x) noexcept
|
||||
{
|
||||
bool const negative = (x >> 63) != 0;
|
||||
bool const largeForm = ((x >> 61) & 3) == 3;
|
||||
auto const exponent = static_cast<int>(largeForm ? (x >> 51) & 0x3ff : (x >> 53) & 0x3ff);
|
||||
auto const coefficient = largeForm
|
||||
? (x & 0x0007'ffff'ffff'ffffULL) | 0x0020'0000'0000'0000ULL
|
||||
: x & 0x001f'ffff'ffff'ffffULL;
|
||||
return {.negative = negative, .coefficient = coefficient, .exponent = exponent - kBias};
|
||||
}
|
||||
};
|
||||
|
||||
struct Bid128Ops
|
||||
{
|
||||
using Value = BID_UINT128;
|
||||
static constexpr unsigned kDigits = 34;
|
||||
static constexpr int kBias = 6176;
|
||||
|
||||
// clang-format off
|
||||
static Value add(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_add(x, y, r, f); }
|
||||
static Value sub(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_sub(x, y, r, f); }
|
||||
static Value mul(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_mul(x, y, r, f); }
|
||||
static Value div(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_div(x, y, r, f); }
|
||||
static Value sqrt(Value x, _IDEC_round r, _IDEC_flags* f) { return bid128_sqrt(x, r, f); }
|
||||
static Value pow(Value x, Value y, _IDEC_round r, _IDEC_flags* f) { return bid128_pow(x, y, r, f); }
|
||||
static Value fromInt64(std::int64_t x, _IDEC_round, _IDEC_flags*) { return bid128_from_int64(x); }
|
||||
static Value scalbn(Value x, int n, _IDEC_round r, _IDEC_flags* f) { return bid128_scalbn(x, n, r, f); }
|
||||
static Value negate(Value x) { return bid128_negate(x); }
|
||||
static Value abs(Value x) { return bid128_abs(x); }
|
||||
static bool less(Value x, Value y, _IDEC_flags* f) { return bid128_quiet_less(x, y, f) != 0; }
|
||||
static bool equal(Value x, Value y, _IDEC_flags* f) { return bid128_quiet_equal(x, y, f) != 0; }
|
||||
static std::int64_t toInt64Nearest(Value x, _IDEC_flags* f) { return bid128_to_int64_rnint(x, f); }
|
||||
static std::int64_t toInt64TowardZero(Value x, _IDEC_flags* f) { return bid128_to_int64_int(x, f); }
|
||||
static std::int64_t toInt64Floor(Value x, _IDEC_flags* f) { return bid128_to_int64_floor(x, f); }
|
||||
static std::int64_t toInt64Ceil(Value x, _IDEC_flags* f) { return bid128_to_int64_ceil(x, f); }
|
||||
static void toString(char* s, Value x, _IDEC_flags* f) { bid128_to_string(s, x, f); }
|
||||
// clang-format on
|
||||
|
||||
/**
|
||||
* Decodes the BID128 layout. Precondition: finite and canonical.
|
||||
*/
|
||||
static Decomposed
|
||||
decompose(Value x) noexcept
|
||||
{
|
||||
auto const hi = x.w[BID_HIGH_128W];
|
||||
auto const lo = x.w[BID_LOW_128W];
|
||||
bool const negative = (hi >> 63) != 0;
|
||||
auto const exponent = static_cast<int>((hi >> 49) & 0x3fff);
|
||||
auto const coefficient =
|
||||
(static_cast<unsigned __int128>(hi & 0x0001'ffff'ffff'ffffULL) << 64) | lo;
|
||||
return {.negative = negative, .coefficient = coefficient, .exponent = exponent - kBias};
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
/**
|
||||
* Intel's Decimal Floating-Point Math Library (BID encoding) behind Number's
|
||||
* interface.
|
||||
*
|
||||
* The rounding mode is thread-local, like Number's, and passed to each call.
|
||||
* Status flags from every call accumulate in a thread-local; read and clear
|
||||
* them with `takeStatus()`.
|
||||
*
|
||||
* Like the other IEEE types, overflow produces infinity and division by zero
|
||||
* produces inf/NaN instead of throwing; decimal64 rounds 19-digit int64
|
||||
* mantissas to 16 digits on construction.
|
||||
*/
|
||||
template <class Ops>
|
||||
class IntelBid final
|
||||
{
|
||||
using Value = typename Ops::Value;
|
||||
|
||||
Value value_{};
|
||||
|
||||
struct State
|
||||
{
|
||||
_IDEC_round round = BID_ROUNDING_TO_NEAREST;
|
||||
_IDEC_flags flags = 0;
|
||||
};
|
||||
|
||||
static State&
|
||||
state() noexcept
|
||||
{
|
||||
static thread_local State s;
|
||||
return s;
|
||||
}
|
||||
|
||||
static IntelBid
|
||||
wrap(Value v) noexcept
|
||||
{
|
||||
IntelBid result;
|
||||
result.value_ = v;
|
||||
return result;
|
||||
}
|
||||
|
||||
public:
|
||||
using rep = std::int64_t;
|
||||
|
||||
IntelBid() : value_{Ops::fromInt64(0, BID_ROUNDING_TO_NEAREST, &state().flags)}
|
||||
{
|
||||
}
|
||||
|
||||
// Implicit, like Number(rep).
|
||||
IntelBid(rep mantissa) : IntelBid(mantissa, 0) // NOLINT
|
||||
{
|
||||
}
|
||||
|
||||
explicit IntelBid(rep mantissa, int exponent)
|
||||
{
|
||||
auto& s = state();
|
||||
value_ = Ops::fromInt64(mantissa, s.round, &s.flags);
|
||||
if (exponent != 0)
|
||||
value_ = Ops::scalbn(value_, exponent, s.round, &s.flags);
|
||||
}
|
||||
|
||||
IntelBid
|
||||
operator-() const noexcept
|
||||
{
|
||||
return wrap(Ops::negate(value_));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
operator+(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
return wrap(Ops::add(x.value_, y.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
operator-(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
return wrap(Ops::sub(x.value_, y.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
operator*(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
return wrap(Ops::mul(x.value_, y.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
operator/(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
return wrap(Ops::div(x.value_, y.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
IntelBid&
|
||||
operator+=(IntelBid const& x) noexcept
|
||||
{
|
||||
return *this = *this + x;
|
||||
}
|
||||
|
||||
IntelBid&
|
||||
operator-=(IntelBid const& x) noexcept
|
||||
{
|
||||
return *this = *this - x;
|
||||
}
|
||||
|
||||
IntelBid&
|
||||
operator*=(IntelBid const& x) noexcept
|
||||
{
|
||||
return *this = *this * x;
|
||||
}
|
||||
|
||||
IntelBid&
|
||||
operator/=(IntelBid const& x) noexcept
|
||||
{
|
||||
return *this = *this / x;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator==(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
return Ops::equal(x.value_, y.value_, &state().flags);
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator<(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
return Ops::less(x.value_, y.value_, &state().flags);
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator>(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
return y < x;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator<=(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
return !(y < x);
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator>=(IntelBid const& x, IntelBid const& y) noexcept
|
||||
{
|
||||
return !(x < y);
|
||||
}
|
||||
|
||||
// Round to an integer using the current rounding mode, like Number.
|
||||
explicit
|
||||
operator rep() const noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
switch (s.round)
|
||||
{
|
||||
case BID_ROUNDING_TO_ZERO:
|
||||
return Ops::toInt64TowardZero(value_, &s.flags);
|
||||
case BID_ROUNDING_DOWN:
|
||||
return Ops::toInt64Floor(value_, &s.flags);
|
||||
case BID_ROUNDING_UP:
|
||||
return Ops::toInt64Ceil(value_, &s.flags);
|
||||
default:
|
||||
return Ops::toInt64Nearest(value_, &s.flags);
|
||||
}
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
getround() noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
switch (state().round)
|
||||
{
|
||||
case BID_ROUNDING_TO_ZERO:
|
||||
return TowardsZero;
|
||||
case BID_ROUNDING_DOWN:
|
||||
return Downward;
|
||||
case BID_ROUNDING_UP:
|
||||
return Upward;
|
||||
default:
|
||||
return ToNearest;
|
||||
}
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
setround(Number::RoundingMode mode) noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
auto const old = getround();
|
||||
auto& round = state().round;
|
||||
switch (mode)
|
||||
{
|
||||
case ToNearest:
|
||||
round = BID_ROUNDING_TO_NEAREST;
|
||||
break;
|
||||
case TowardsZero:
|
||||
round = BID_ROUNDING_TO_ZERO;
|
||||
break;
|
||||
case Downward:
|
||||
round = BID_ROUNDING_DOWN;
|
||||
break;
|
||||
case Upward:
|
||||
round = BID_ROUNDING_UP;
|
||||
break;
|
||||
}
|
||||
return old;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the status flags raised since the last call, and clears them.
|
||||
*/
|
||||
static _IDEC_flags
|
||||
takeStatus() noexcept
|
||||
{
|
||||
return std::exchange(state().flags, 0);
|
||||
}
|
||||
|
||||
friend std::string
|
||||
to_string(IntelBid const& x)
|
||||
{
|
||||
std::array<char, 64> buffer{};
|
||||
Ops::toString(buffer.data(), x.value_, &state().flags);
|
||||
return buffer.data();
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
abs(IntelBid const& x) noexcept
|
||||
{
|
||||
return wrap(Ops::abs(x.value_));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
power(IntelBid const& x, unsigned n)
|
||||
{
|
||||
auto& s = state();
|
||||
IntelBid const exponent{static_cast<rep>(n)};
|
||||
return wrap(Ops::pow(x.value_, exponent.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
friend IntelBid
|
||||
root2(IntelBid const& x) noexcept
|
||||
{
|
||||
auto& s = state();
|
||||
return wrap(Ops::sqrt(x.value_, s.round, &s.flags));
|
||||
}
|
||||
|
||||
/**
|
||||
* Precondition: x is finite.
|
||||
*/
|
||||
friend Decomposed
|
||||
decompose(IntelBid const& x) noexcept
|
||||
{
|
||||
return Ops::decompose(x.value_);
|
||||
}
|
||||
};
|
||||
|
||||
using IntelBid64 = IntelBid<detail::Bid64Ops>;
|
||||
using IntelBid128 = IntelBid<detail::Bid128Ops>;
|
||||
|
||||
static_assert(DecomposableNumber<IntelBid64>);
|
||||
static_assert(DecomposableNumber<IntelBid128>);
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
355
src/benchmarks/libxrpl/number/Kernels.cpp
Normal file
355
src/benchmarks/libxrpl/number/Kernels.cpp
Normal file
@@ -0,0 +1,355 @@
|
||||
// Ledger-formula benchmarks: AMM swaps, vault share conversion, and loan
|
||||
// amortization (see Kernels.h), run identically on every NumberLike type.
|
||||
// See docs/NumberDecimalBenchmark.md.
|
||||
//
|
||||
// Benchmark names are "kernel/<kernel>/<type>". Each reports throughput and,
|
||||
// when built with mpdecimal (bench_mpdecimal), accuracy counters comparing the
|
||||
// type's result with the same formula evaluated at 96 digits from the type's
|
||||
// own inputs, so construction rounding is excluded:
|
||||
//
|
||||
// digits_min / digits_median correct significant digits, -log10(relative
|
||||
// error), worst case and median over the
|
||||
// cases; 40 means exact. For loan_amortize the
|
||||
// error is the final balance relative to the
|
||||
// principal.
|
||||
// mismatches for integer results: cases that differ from
|
||||
// the exact truncated result.
|
||||
|
||||
#include <benchmarks/libxrpl/number/Kernels.h>
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
#include <benchmarks/libxrpl/number/BoostDecimal.h>
|
||||
#include <benchmarks/libxrpl/number/Double.h>
|
||||
#include <benchmarks/libxrpl/number/Number128.h>
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
#include <benchmarks/libxrpl/number/Operands.h>
|
||||
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
#include <benchmarks/libxrpl/number/MpDecimal.h>
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
#include <benchmarks/libxrpl/number/IntelBid.h>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <format>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
namespace {
|
||||
|
||||
constexpr std::size_t kCases = 256;
|
||||
constexpr std::uint64_t kSeed = 0x6b65'726e;
|
||||
|
||||
/**
|
||||
* Inputs for one kernel evaluation: up to three decimal values and two
|
||||
* integer parameters (fee, rate, interval, payment count).
|
||||
*/
|
||||
struct KernelCase
|
||||
{
|
||||
std::array<RawOperand, 3> values;
|
||||
std::array<std::uint32_t, 3> integers;
|
||||
};
|
||||
|
||||
template <class T>
|
||||
using Values = std::array<T, 3>;
|
||||
|
||||
// ---- Case generation ----
|
||||
|
||||
/**
|
||||
* A positive value with a 16-digit mantissa and magnitude 10^lo to 10^hi.
|
||||
*/
|
||||
RawOperand
|
||||
amount(std::mt19937_64& rng, int lo, int hi)
|
||||
{
|
||||
constexpr auto kE15 = static_cast<std::int64_t>(kPowerOfTen[15]);
|
||||
auto const mantissa = std::uniform_int_distribution<std::int64_t>{kE15, (kE15 * 10) - 1}(rng);
|
||||
return {mantissa, std::uniform_int_distribution<int>{lo, hi}(rng)-15};
|
||||
}
|
||||
|
||||
/**
|
||||
* `base` scaled by a factor 10^lo to 10^hi, with a fresh mantissa.
|
||||
*/
|
||||
RawOperand
|
||||
fractionOf(std::mt19937_64& rng, RawOperand base, int lo, int hi)
|
||||
{
|
||||
auto result = amount(rng, 0, 0);
|
||||
result.exponent = base.exponent + std::uniform_int_distribution<int>{lo, hi}(rng);
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pools of 10^6 to 10^12 per side, trades of 10^-6 to 10^-1 of the pool,
|
||||
* fees of 0 to 1%.
|
||||
*/
|
||||
std::vector<KernelCase>
|
||||
ammCases()
|
||||
{
|
||||
std::mt19937_64 rng{kSeed};
|
||||
std::vector<KernelCase> cases;
|
||||
for (std::size_t i = 0; i < kCases; ++i)
|
||||
{
|
||||
auto const poolIn = amount(rng, 6, 12);
|
||||
auto const poolOut = amount(rng, 6, 12);
|
||||
// Trades are a fraction of the out pool for swap-out to stay solvent.
|
||||
auto const trade = fractionOf(rng, poolOut, -6, -2);
|
||||
auto const fee = std::uniform_int_distribution<std::uint32_t>{0, 1000}(rng);
|
||||
cases.push_back({{poolIn, poolOut, trade}, {fee, 0, 0}});
|
||||
}
|
||||
return cases;
|
||||
}
|
||||
|
||||
/**
|
||||
* Vaults holding 10^9 to 10^17 drops-like integer assets, share supply 10^6
|
||||
* times larger (vault scale 6) within a factor of 2, but at most 10^18 since
|
||||
* MPT supply is capped at 2^63-1; deposits of 10^-6 to 10^-1 of the assets.
|
||||
*/
|
||||
std::vector<KernelCase>
|
||||
vaultCases()
|
||||
{
|
||||
std::mt19937_64 rng{kSeed + 1};
|
||||
std::vector<KernelCase> cases;
|
||||
for (std::size_t i = 0; i < kCases; ++i)
|
||||
{
|
||||
auto const assetDigits = std::uniform_int_distribution<int>{10, 17}(rng);
|
||||
auto const assetTotal = std::uniform_int_distribution<std::int64_t>{
|
||||
static_cast<std::int64_t>(kPowerOfTen[assetDigits - 1]),
|
||||
static_cast<std::int64_t>(kPowerOfTen[assetDigits]) - 1}(rng);
|
||||
auto const ratio = std::uniform_real_distribution<double>{0.5, 2.0}(rng);
|
||||
auto const shareMantissa =
|
||||
static_cast<std::int64_t>(static_cast<double>(assetTotal) * ratio) | 1;
|
||||
auto const shareExponent = std::min(6, 18 - assetDigits);
|
||||
auto const deposit = std::uniform_int_distribution<std::int64_t>{
|
||||
std::max<std::int64_t>(1, assetTotal / 1'000'000), assetTotal / 10}(rng);
|
||||
cases.push_back(
|
||||
{{RawOperand{shareMantissa, shareExponent},
|
||||
RawOperand{assetTotal, 0},
|
||||
RawOperand{deposit, 0}},
|
||||
{}});
|
||||
}
|
||||
return cases;
|
||||
}
|
||||
|
||||
/**
|
||||
* Principal of 10^3 to 10^9, annual rates of 0.1% to 30% (in 1/10 bips),
|
||||
* monthly payments.
|
||||
*/
|
||||
std::vector<KernelCase>
|
||||
loanCases(std::uint32_t payments)
|
||||
{
|
||||
constexpr std::uint32_t kMonth = 30 * 24 * 60 * 60;
|
||||
std::mt19937_64 rng{kSeed + 2 + payments};
|
||||
std::vector<KernelCase> cases;
|
||||
for (std::size_t i = 0; i < kCases; ++i)
|
||||
{
|
||||
auto const principal = amount(rng, 3, 9);
|
||||
auto const rate = std::uniform_int_distribution<std::uint32_t>{100, 30'000}(rng);
|
||||
cases.push_back(
|
||||
{{principal, RawOperand{0, 0}, RawOperand{0, 0}}, {rate, kMonth, payments}});
|
||||
}
|
||||
return cases;
|
||||
}
|
||||
|
||||
// ---- Kernels, adapted to the common case shape ----
|
||||
|
||||
auto const kAmmSwapIn = [](auto const& v, auto const& i) {
|
||||
return ammSwapIn(v[0], v[1], v[2], static_cast<std::uint16_t>(i[0]));
|
||||
};
|
||||
auto const kAmmSwapOut = [](auto const& v, auto const& i) {
|
||||
return ammSwapOut(v[0], v[1], v[2], static_cast<std::uint16_t>(i[0]));
|
||||
};
|
||||
auto const kVaultAssetsToShares = [](auto const& v, auto const&) {
|
||||
return vaultAssetsToShares(v[0], v[1], v[2]);
|
||||
};
|
||||
auto const kVaultSharesToAssets = [](auto const& v, auto const&) {
|
||||
// Redeem the same magnitude of shares as the deposit's asset amount.
|
||||
return vaultSharesToAssets(v[1], v[0], v[2]);
|
||||
};
|
||||
auto const kLoanPayment = [](auto const& v, auto const& i) {
|
||||
using T = std::decay_t<decltype(v[0])>;
|
||||
return loanPeriodicPayment(v[0], loanPeriodicRate<T>(i[0], i[1]), i[2]);
|
||||
};
|
||||
auto const kLoanAmortize = [](auto const& v, auto const& i) {
|
||||
using T = std::decay_t<decltype(v[0])>;
|
||||
return loanAmortize(v[0], loanPeriodicRate<T>(i[0], i[1]), i[2]);
|
||||
};
|
||||
|
||||
// ---- Accuracy against a 96-digit evaluation ----
|
||||
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
using Exact = MpDecimal<96>;
|
||||
|
||||
template <NumberLike T>
|
||||
Exact
|
||||
toExact(T const& x)
|
||||
{
|
||||
if constexpr (DecomposableNumber<T>)
|
||||
{
|
||||
return Exact::fromDecomposed(decompose(x));
|
||||
}
|
||||
else
|
||||
{
|
||||
Exact result;
|
||||
mpd_context_t ctx;
|
||||
mpd_maxcontext(&ctx);
|
||||
std::uint32_t status = 0;
|
||||
mpd_qset_string(result.get(), to_string(x).c_str(), &ctx, &status);
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
double
|
||||
correctDigits(Exact const& error, Exact const& scale)
|
||||
{
|
||||
constexpr double kExact = 40;
|
||||
if (mpd_iszero(error.get()) != 0)
|
||||
return kExact;
|
||||
auto const relative = std::stod(to_string(abs(error) / abs(scale)));
|
||||
return std::min(kExact, -std::log10(relative));
|
||||
}
|
||||
|
||||
/**
|
||||
* Sets digits_min/digits_median (or mismatches) on `state`.
|
||||
*/
|
||||
template <NumberLike T, class Fn>
|
||||
void
|
||||
measureAccuracy(
|
||||
benchmark::State& state,
|
||||
std::vector<Values<T>> const& inputs,
|
||||
std::vector<KernelCase> const& cases,
|
||||
Fn fn,
|
||||
bool relativeToFirstInput)
|
||||
{
|
||||
std::vector<double> digits;
|
||||
std::size_t mismatches = 0;
|
||||
for (std::size_t c = 0; c < inputs.size(); ++c)
|
||||
{
|
||||
Values<Exact> const exactInputs{
|
||||
toExact(inputs[c][0]), toExact(inputs[c][1]), toExact(inputs[c][2])};
|
||||
auto const result = fn(inputs[c], cases[c].integers);
|
||||
auto const exact = fn(exactInputs, cases[c].integers);
|
||||
if constexpr (std::is_integral_v<std::decay_t<decltype(result)>>)
|
||||
{
|
||||
mismatches += result == exact ? 0 : 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
auto const& scale = relativeToFirstInput ? exactInputs[0] : exact;
|
||||
digits.push_back(correctDigits(toExact(result) - exact, scale));
|
||||
}
|
||||
}
|
||||
if (digits.empty())
|
||||
{
|
||||
state.counters["mismatches"] = static_cast<double>(mismatches);
|
||||
return;
|
||||
}
|
||||
std::ranges::sort(digits);
|
||||
state.counters["digits_min"] = digits.front();
|
||||
state.counters["digits_median"] = digits[digits.size() / 2];
|
||||
}
|
||||
#endif
|
||||
|
||||
// ---- Benchmarks ----
|
||||
|
||||
struct NoEnv
|
||||
{
|
||||
};
|
||||
|
||||
template <MantissaRange::MantissaScale Scale>
|
||||
struct NumberScaleEnv
|
||||
{
|
||||
NumberMantissaScaleGuard guard{Scale};
|
||||
};
|
||||
|
||||
template <NumberLike T, class Env, class Fn>
|
||||
void
|
||||
registerKernel(
|
||||
std::string_view kernel,
|
||||
std::string_view subject,
|
||||
std::vector<KernelCase> const& cases,
|
||||
Fn fn,
|
||||
bool relativeToFirstInput = false)
|
||||
{
|
||||
benchmark::RegisterBenchmark(
|
||||
std::format("kernel/{}/{}", kernel, subject),
|
||||
[cases, fn, relativeToFirstInput](benchmark::State& state) {
|
||||
[[maybe_unused]] Env const env{};
|
||||
std::vector<Values<T>> inputs;
|
||||
inputs.reserve(cases.size());
|
||||
for (auto const& c : cases)
|
||||
{
|
||||
inputs.push_back(
|
||||
{T{c.values[0].mantissa, c.values[0].exponent},
|
||||
T{c.values[1].mantissa, c.values[1].exponent},
|
||||
T{c.values[2].mantissa, c.values[2].exponent}});
|
||||
}
|
||||
for (auto _ : state)
|
||||
{
|
||||
for (std::size_t c = 0; c < inputs.size(); ++c)
|
||||
{
|
||||
auto r = fn(inputs[c], cases[c].integers);
|
||||
benchmark::DoNotOptimize(r);
|
||||
}
|
||||
}
|
||||
state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * inputs.size()));
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
measureAccuracy<T>(state, inputs, cases, fn, relativeToFirstInput);
|
||||
#else
|
||||
(void)relativeToFirstInput;
|
||||
#endif
|
||||
});
|
||||
}
|
||||
|
||||
template <NumberLike T, class Env>
|
||||
void
|
||||
registerSubject(std::string_view subject)
|
||||
{
|
||||
static auto const kAmm = ammCases();
|
||||
static auto const kVault = vaultCases();
|
||||
static auto const kLoan12 = loanCases(12);
|
||||
static auto const kLoan360 = loanCases(360);
|
||||
|
||||
registerKernel<T, Env>("amm_swap_in", subject, kAmm, kAmmSwapIn);
|
||||
registerKernel<T, Env>("amm_swap_out", subject, kAmm, kAmmSwapOut);
|
||||
registerKernel<T, Env>("vault_assets_to_shares", subject, kVault, kVaultAssetsToShares);
|
||||
registerKernel<T, Env>("vault_shares_to_assets", subject, kVault, kVaultSharesToAssets);
|
||||
registerKernel<T, Env>("loan_payment12", subject, kLoan12, kLoanPayment);
|
||||
registerKernel<T, Env>("loan_payment360", subject, kLoan360, kLoanPayment);
|
||||
registerKernel<T, Env>("loan_amortize12", subject, kLoan12, kLoanAmortize, true);
|
||||
registerKernel<T, Env>("loan_amortize360", subject, kLoan360, kLoanAmortize, true);
|
||||
}
|
||||
|
||||
[[maybe_unused]] bool const kRegistered = [] {
|
||||
using enum MantissaRange::MantissaScale;
|
||||
registerSubject<Number, NumberScaleEnv<Small>>("Number.Small");
|
||||
registerSubject<Number, NumberScaleEnv<Large330>>("Number.Large330");
|
||||
registerSubject<BoostDecimal64, NoEnv>("BoostDecimal64");
|
||||
registerSubject<BoostDecimal128, NoEnv>("BoostDecimal128");
|
||||
registerSubject<Double, NoEnv>("Double");
|
||||
#ifdef XRPL_BENCH_MPDECIMAL
|
||||
registerSubject<MpDecimal34, NoEnv>("MpDecimal34");
|
||||
registerSubject<MpDecimal38, NoEnv>("MpDecimal38");
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
registerSubject<IntelBid64, NoEnv>("IntelBid64");
|
||||
registerSubject<IntelBid128, NoEnv>("IntelBid128");
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_HAS_NUMBER128
|
||||
registerSubject<Number128, NoEnv>("Number128");
|
||||
#endif
|
||||
return true;
|
||||
}();
|
||||
|
||||
} // namespace
|
||||
} // namespace xrpl::number_bench
|
||||
220
src/benchmarks/libxrpl/number/Kernels.h
Normal file
220
src/benchmarks/libxrpl/number/Kernels.h
Normal file
@@ -0,0 +1,220 @@
|
||||
#pragma once
|
||||
|
||||
// Ledger formulas, written once against NumberLike so every library runs the
|
||||
// identical computation. Each kernel mirrors a production function; keep them
|
||||
// in sync if those change:
|
||||
//
|
||||
// ammSwapIn / ammSwapOut swapAssetIn / swapAssetOut, fixAMMv1_1 branch
|
||||
// (include/xrpl/ledger/helpers/AMMHelpers.h)
|
||||
// vaultAssetsToShares assetsToSharesDeposit
|
||||
// vaultSharesToAssets sharesToAssetsDeposit
|
||||
// (src/libxrpl/ledger/helpers/VaultHelpers.cpp)
|
||||
// loanPeriodicRate loanPeriodicRate
|
||||
// loanPowerMinusOne computePowerMinusOneHybrid
|
||||
// loanPeriodicPayment loanPeriodicPayment, fixCleanup3_2_0 branch
|
||||
// (src/libxrpl/ledger/helpers/LendingHelpers.cpp)
|
||||
//
|
||||
// loanAmortize is not a production function: it runs a full payment schedule
|
||||
// (interest = balance * rate; balance -= payment - interest) so error
|
||||
// accumulates the way it does over a loan's life. The exact remaining balance
|
||||
// after the last payment is 0.
|
||||
//
|
||||
// The kernels operate on bare values, not STAmount or ledger entries, and
|
||||
// omit the final rounding to the asset's scale, so they measure arithmetic
|
||||
// alone.
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
namespace detail {
|
||||
|
||||
template <NumberLike T>
|
||||
T
|
||||
integer(std::int64_t value)
|
||||
{
|
||||
return T{value};
|
||||
}
|
||||
|
||||
/**
|
||||
* Sets a rounding mode for its lifetime, then restores the previous one.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
class RoundGuard
|
||||
{
|
||||
Number::RoundingMode saved_;
|
||||
|
||||
public:
|
||||
explicit RoundGuard(Number::RoundingMode mode) : saved_{T::setround(mode)}
|
||||
{
|
||||
}
|
||||
|
||||
~RoundGuard()
|
||||
{
|
||||
T::setround(saved_);
|
||||
}
|
||||
|
||||
RoundGuard(RoundGuard const&) = delete;
|
||||
RoundGuard&
|
||||
operator=(RoundGuard const&) = delete;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
/**
|
||||
* Amount of the out asset received for swapping `assetIn` into the pool.
|
||||
* Rounding at each step favors the AMM.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
ammSwapIn(T const& poolIn, T const& poolOut, T const& assetIn, std::uint16_t tfee)
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
auto const saved = T::getround();
|
||||
|
||||
T::setround(Upward);
|
||||
auto const numerator = poolIn * poolOut;
|
||||
auto const fee = detail::integer<T>(tfee) / detail::integer<T>(100'000);
|
||||
|
||||
T::setround(Downward);
|
||||
auto const denom = poolIn + (assetIn * (detail::integer<T>(1) - fee));
|
||||
|
||||
T::setround(Upward);
|
||||
auto const ratio = numerator / denom;
|
||||
|
||||
T::setround(Downward);
|
||||
auto const swapOut = poolOut - ratio;
|
||||
|
||||
T::setround(saved);
|
||||
return swapOut;
|
||||
}
|
||||
|
||||
/**
|
||||
* Amount of the in asset required to take `assetOut` out of the pool.
|
||||
* Rounding at each step favors the AMM.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
ammSwapOut(T const& poolIn, T const& poolOut, T const& assetOut, std::uint16_t tfee)
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
auto const saved = T::getround();
|
||||
|
||||
T::setround(Upward);
|
||||
auto const numerator = poolIn * poolOut;
|
||||
|
||||
T::setround(Downward);
|
||||
auto const denom = poolOut - assetOut;
|
||||
|
||||
T::setround(Upward);
|
||||
auto const ratio = numerator / denom;
|
||||
auto const numerator2 = ratio - poolIn;
|
||||
auto const fee = detail::integer<T>(tfee) / detail::integer<T>(100'000);
|
||||
|
||||
T::setround(Downward);
|
||||
auto const feeMult = detail::integer<T>(1) - fee;
|
||||
|
||||
T::setround(Upward);
|
||||
auto const swapIn = numerator2 / feeMult;
|
||||
|
||||
T::setround(saved);
|
||||
return swapIn;
|
||||
}
|
||||
|
||||
/**
|
||||
* Shares minted for depositing `assets` into a non-empty vault, truncated to
|
||||
* an integer (vault shares are MPTs).
|
||||
*/
|
||||
template <NumberLike T>
|
||||
std::int64_t
|
||||
vaultAssetsToShares(T const& shareTotal, T const& assetTotal, T const& assets)
|
||||
{
|
||||
auto const shares = (shareTotal * assets) / assetTotal;
|
||||
detail::RoundGuard<T> const guard{Number::RoundingMode::TowardsZero};
|
||||
return static_cast<std::int64_t>(shares);
|
||||
}
|
||||
|
||||
/**
|
||||
* Assets corresponding to `shares` of a non-empty vault.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
vaultSharesToAssets(T const& assetTotal, T const& shareTotal, T const& shares)
|
||||
{
|
||||
return (assetTotal * shares) / shareTotal;
|
||||
}
|
||||
|
||||
constexpr std::uint32_t kSecondsInYear = 365 * 24 * 60 * 60;
|
||||
|
||||
/**
|
||||
* Interest rate per payment interval, from an annual rate in 1/10 bips.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
loanPeriodicRate(std::uint32_t interestRateTenthBips, std::uint32_t paymentInterval)
|
||||
{
|
||||
auto const interval = detail::integer<T>(paymentInterval);
|
||||
return interval * detail::integer<T>(interestRateTenthBips) / detail::integer<T>(100'000) /
|
||||
detail::integer<T>(kSecondsInYear);
|
||||
}
|
||||
|
||||
/**
|
||||
* (1 + r)^n - 1: closed form, or a binomial series when r * n is tiny and the
|
||||
* closed form would cancel catastrophically.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
loanPowerMinusOne(T const& rate, std::uint32_t payments)
|
||||
{
|
||||
auto const n = detail::integer<T>(payments);
|
||||
if (n * rate >= T{1, -9})
|
||||
return power(detail::integer<T>(1) + rate, payments) - detail::integer<T>(1);
|
||||
|
||||
T term = n * rate;
|
||||
T sum = term;
|
||||
for (std::uint32_t k = 1; k < payments; ++k)
|
||||
{
|
||||
term = term * rate * detail::integer<T>(payments - k) / detail::integer<T>(k + 1);
|
||||
T const next = sum + term;
|
||||
if (next == sum)
|
||||
break;
|
||||
sum = next;
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
/**
|
||||
* Level payment that amortizes `principal` over `payments` periods.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
loanPeriodicPayment(T const& principal, T const& rate, std::uint32_t payments)
|
||||
{
|
||||
auto const raisedMinusOne = loanPowerMinusOne(rate, payments);
|
||||
auto const raised = detail::integer<T>(1) + raisedMinusOne;
|
||||
return principal * ((rate * raised) / raisedMinusOne);
|
||||
}
|
||||
|
||||
/**
|
||||
* Remaining balance after making every scheduled payment. Exactly 0 in exact
|
||||
* arithmetic; anything else is accumulated rounding error.
|
||||
*/
|
||||
template <NumberLike T>
|
||||
T
|
||||
loanAmortize(T const& principal, T const& rate, std::uint32_t payments)
|
||||
{
|
||||
auto const payment = loanPeriodicPayment(principal, rate, payments);
|
||||
T balance = principal;
|
||||
for (std::uint32_t i = 0; i < payments; ++i)
|
||||
{
|
||||
auto const interest = balance * rate;
|
||||
balance -= payment - interest;
|
||||
}
|
||||
return balance;
|
||||
}
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
395
src/benchmarks/libxrpl/number/MpDecimal.h
Normal file
395
src/benchmarks/libxrpl/number/MpDecimal.h
Normal file
@@ -0,0 +1,395 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <mpdecimal.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
/**
|
||||
* mpdecimal (libmpdec) at a fixed precision, behind Number's interface.
|
||||
*
|
||||
* This is the correctly-rounded reference for the divergence harness: with
|
||||
* `allcr` set, every operation, including `power`, is correctly rounded in
|
||||
* the active rounding mode.
|
||||
*
|
||||
* `Digits` is the coefficient precision: 19 to mirror today's Number, 34 for
|
||||
* IEEE decimal128, 38 for a full-uint128 Number. The exponent range follows
|
||||
* Number's convention: value = coefficient * 10^e with e in
|
||||
* [Number::kMinExponent, Number::kMaxExponent] for a `Digits`-digit
|
||||
* coefficient. mpdecimal uses adjusted exponents (the exponent of the leading
|
||||
* digit), hence the `Digits - 1` offset.
|
||||
*
|
||||
* Coefficients live in inline storage sized for `Digits`, so copying never
|
||||
* allocates. Library temporaries may still allocate; that is part of
|
||||
* mpdecimal's cost. Precisions above 38 digits exist for the divergence
|
||||
* harness, which needs a near-exact result to round from; `decompose` is only
|
||||
* available up to 38 digits.
|
||||
*
|
||||
* Unlike Number, results that overflow become infinity and underflow becomes
|
||||
* subnormal or zero. Each operation ORs its mpdecimal status into a
|
||||
* thread-local accumulator; read and clear it with `takeStatus()`.
|
||||
*/
|
||||
template <unsigned Digits>
|
||||
class MpDecimal final
|
||||
{
|
||||
static_assert(Digits >= 16 && Digits <= 114);
|
||||
|
||||
// Enough 19-digit limbs for a Digits-digit coefficient, and never fewer
|
||||
// than MPD_MINALLOC_MIN.
|
||||
static constexpr mpd_ssize_t kLimbs = (Digits / MPD_RDIGITS) + 2;
|
||||
|
||||
mpd_uint_t data_[kLimbs]{};
|
||||
// Zero: one limb holding 0.
|
||||
mpd_t dec_{MPD_STATIC | MPD_STATIC_DATA, 0, 1, 1, kLimbs, data_};
|
||||
|
||||
struct Context
|
||||
{
|
||||
mpd_context_t ctx{};
|
||||
uint32_t status = 0;
|
||||
|
||||
Context()
|
||||
{
|
||||
mpd_defaultcontext(&ctx);
|
||||
ctx.prec = Digits;
|
||||
ctx.emax = Number::kMaxExponent + static_cast<int>(Digits) - 1;
|
||||
ctx.emin = Number::kMinExponent + static_cast<int>(Digits) - 1;
|
||||
ctx.round = MPD_ROUND_HALF_EVEN;
|
||||
ctx.traps = 0;
|
||||
ctx.clamp = 0;
|
||||
ctx.allcr = 1;
|
||||
}
|
||||
};
|
||||
|
||||
static Context&
|
||||
context() noexcept
|
||||
{
|
||||
static thread_local Context c;
|
||||
return c;
|
||||
}
|
||||
|
||||
template <class F>
|
||||
static MpDecimal
|
||||
compute(F f)
|
||||
{
|
||||
MpDecimal result;
|
||||
auto& c = context();
|
||||
f(&result.dec_, &c.ctx, &c.status);
|
||||
return result;
|
||||
}
|
||||
|
||||
public:
|
||||
using rep = std::int64_t;
|
||||
|
||||
static constexpr unsigned kDigits = Digits;
|
||||
|
||||
MpDecimal() noexcept = default;
|
||||
|
||||
~MpDecimal()
|
||||
{
|
||||
// Frees the coefficient only if an operation outgrew inline storage.
|
||||
mpd_del(&dec_);
|
||||
}
|
||||
|
||||
MpDecimal(MpDecimal const& other) noexcept
|
||||
{
|
||||
mpd_qcopy(&dec_, &other.dec_, &context().status);
|
||||
}
|
||||
|
||||
MpDecimal&
|
||||
operator=(MpDecimal const& other) noexcept
|
||||
{
|
||||
if (this != &other)
|
||||
mpd_qcopy(&dec_, &other.dec_, &context().status);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Implicit, like Number(rep).
|
||||
MpDecimal(rep mantissa) : MpDecimal(mantissa, 0) // NOLINT
|
||||
{
|
||||
}
|
||||
|
||||
explicit MpDecimal(rep mantissa, int exponent)
|
||||
{
|
||||
auto& c = context();
|
||||
mpd_qset_i64_exact(&dec_, mantissa, &c.status);
|
||||
dec_.exp += exponent;
|
||||
mpd_qfinalize(&dec_, &c.ctx, &c.status);
|
||||
}
|
||||
|
||||
MpDecimal
|
||||
operator-() const
|
||||
{
|
||||
return compute(
|
||||
[this](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qminus(r, &dec_, ctx, s); });
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
operator+(MpDecimal const& x, MpDecimal const& y)
|
||||
{
|
||||
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qadd(r, &x.dec_, &y.dec_, ctx, s);
|
||||
});
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
operator-(MpDecimal const& x, MpDecimal const& y)
|
||||
{
|
||||
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qsub(r, &x.dec_, &y.dec_, ctx, s);
|
||||
});
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
operator*(MpDecimal const& x, MpDecimal const& y)
|
||||
{
|
||||
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qmul(r, &x.dec_, &y.dec_, ctx, s);
|
||||
});
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
operator/(MpDecimal const& x, MpDecimal const& y)
|
||||
{
|
||||
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qdiv(r, &x.dec_, &y.dec_, ctx, s);
|
||||
});
|
||||
}
|
||||
|
||||
MpDecimal&
|
||||
operator+=(MpDecimal const& x)
|
||||
{
|
||||
return *this = *this + x;
|
||||
}
|
||||
|
||||
MpDecimal&
|
||||
operator-=(MpDecimal const& x)
|
||||
{
|
||||
return *this = *this - x;
|
||||
}
|
||||
|
||||
MpDecimal&
|
||||
operator*=(MpDecimal const& x)
|
||||
{
|
||||
return *this = *this * x;
|
||||
}
|
||||
|
||||
MpDecimal&
|
||||
operator/=(MpDecimal const& x)
|
||||
{
|
||||
return *this = *this / x;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator==(MpDecimal const& x, MpDecimal const& y) noexcept
|
||||
{
|
||||
return mpd_qcmp(&x.dec_, &y.dec_, &context().status) == 0;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator<(MpDecimal const& x, MpDecimal const& y) noexcept
|
||||
{
|
||||
return mpd_qcmp(&x.dec_, &y.dec_, &context().status) == -1;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator>(MpDecimal const& x, MpDecimal const& y) noexcept
|
||||
{
|
||||
return y < x;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator<=(MpDecimal const& x, MpDecimal const& y) noexcept
|
||||
{
|
||||
return x < y || x == y;
|
||||
}
|
||||
|
||||
friend bool
|
||||
operator>=(MpDecimal const& x, MpDecimal const& y) noexcept
|
||||
{
|
||||
return y <= x;
|
||||
}
|
||||
|
||||
// Round to an integer using the current rounding mode, like Number.
|
||||
explicit
|
||||
operator rep() const
|
||||
{
|
||||
auto const rounded = compute([this](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qround_to_int(r, &dec_, ctx, s);
|
||||
});
|
||||
return mpd_qget_i64(&rounded.dec_, &context().status);
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
getround() noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
switch (context().ctx.round)
|
||||
{
|
||||
case MPD_ROUND_DOWN:
|
||||
return TowardsZero;
|
||||
case MPD_ROUND_FLOOR:
|
||||
return Downward;
|
||||
case MPD_ROUND_CEILING:
|
||||
return Upward;
|
||||
default:
|
||||
return ToNearest;
|
||||
}
|
||||
}
|
||||
|
||||
static Number::RoundingMode
|
||||
setround(Number::RoundingMode mode) noexcept
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
auto const old = getround();
|
||||
auto& round = context().ctx.round;
|
||||
switch (mode)
|
||||
{
|
||||
case ToNearest:
|
||||
round = MPD_ROUND_HALF_EVEN;
|
||||
break;
|
||||
case TowardsZero:
|
||||
round = MPD_ROUND_DOWN;
|
||||
break;
|
||||
case Downward:
|
||||
round = MPD_ROUND_FLOOR;
|
||||
break;
|
||||
case Upward:
|
||||
round = MPD_ROUND_CEILING;
|
||||
break;
|
||||
}
|
||||
return old;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the mpdecimal status flags raised since the last call, and clears them.
|
||||
*/
|
||||
static uint32_t
|
||||
takeStatus() noexcept
|
||||
{
|
||||
return std::exchange(context().status, 0);
|
||||
}
|
||||
|
||||
friend std::string
|
||||
to_string(MpDecimal const& x)
|
||||
{
|
||||
char* raw = nullptr;
|
||||
auto const size = mpd_to_sci_size(&raw, &x.dec_, 0);
|
||||
std::unique_ptr<char, void (*)(void*)> const owned{raw, mpd_free};
|
||||
return size < 0 ? std::string{} : std::string{raw, static_cast<std::size_t>(size)};
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
abs(MpDecimal const& x)
|
||||
{
|
||||
return compute(
|
||||
[&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qabs(r, &x.dec_, ctx, s); });
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
power(MpDecimal const& x, unsigned n)
|
||||
{
|
||||
MpDecimal const exponent{static_cast<rep>(n)};
|
||||
return compute([&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) {
|
||||
mpd_qpow(r, &x.dec_, &exponent.dec_, ctx, s);
|
||||
});
|
||||
}
|
||||
|
||||
friend MpDecimal
|
||||
root2(MpDecimal const& x)
|
||||
{
|
||||
return compute(
|
||||
[&](mpd_t* r, mpd_context_t* ctx, uint32_t* s) { mpd_qsqrt(r, &x.dec_, ctx, s); });
|
||||
}
|
||||
|
||||
/**
|
||||
* The underlying libmpdec value, for the divergence harness.
|
||||
*/
|
||||
[[nodiscard]] mpd_t const*
|
||||
get() const noexcept
|
||||
{
|
||||
return &dec_;
|
||||
}
|
||||
|
||||
[[nodiscard]] mpd_t*
|
||||
get() noexcept
|
||||
{
|
||||
return &dec_;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rounds `x` to this type's precision in the current rounding mode.
|
||||
*/
|
||||
template <unsigned OtherDigits>
|
||||
static MpDecimal
|
||||
roundFrom(MpDecimal<OtherDigits> const& x)
|
||||
{
|
||||
MpDecimal result;
|
||||
auto& c = context();
|
||||
mpd_qcopy(&result.dec_, x.get(), &c.status);
|
||||
mpd_qfinalize(&result.dec_, &c.ctx, &c.status);
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Exact conversion; rounds only if `d` has more than Digits digits.
|
||||
*/
|
||||
static MpDecimal
|
||||
fromDecomposed(Decomposed const& d)
|
||||
{
|
||||
auto& c = context();
|
||||
mpd_context_t exact;
|
||||
mpd_maxcontext(&exact);
|
||||
exact.traps = 0;
|
||||
|
||||
// coefficient = hi * 2^64 + lo, assembled at unlimited precision.
|
||||
MpDecimal result;
|
||||
MpDecimal lo;
|
||||
MpDecimal shift;
|
||||
mpd_qset_u64_exact(
|
||||
&result.dec_, static_cast<std::uint64_t>(d.coefficient >> 64), &c.status);
|
||||
mpd_qset_u64_exact(&lo.dec_, static_cast<std::uint64_t>(d.coefficient), &c.status);
|
||||
mpd_qset_u64_exact(&shift.dec_, std::uint64_t{1} << 32, &c.status);
|
||||
mpd_qmul(&shift.dec_, &shift.dec_, &shift.dec_, &exact, &c.status);
|
||||
mpd_qmul(&result.dec_, &result.dec_, &shift.dec_, &exact, &c.status);
|
||||
mpd_qadd(&result.dec_, &result.dec_, &lo.dec_, &exact, &c.status);
|
||||
if (d.negative)
|
||||
mpd_set_negative(&result.dec_);
|
||||
result.dec_.exp += d.exponent;
|
||||
mpd_qfinalize(&result.dec_, &c.ctx, &c.status);
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Precondition: x is finite (check takeStatus() for overflow/invalid first).
|
||||
*/
|
||||
friend Decomposed
|
||||
decompose(MpDecimal const& x) noexcept
|
||||
requires(Digits <= 38)
|
||||
{
|
||||
unsigned __int128 coefficient = 0;
|
||||
for (auto i = x.dec_.len; i-- > 0;)
|
||||
coefficient = (coefficient * MPD_RADIX) + x.dec_.data[i];
|
||||
return {
|
||||
.negative = (x.dec_.flags & MPD_NEG) != 0,
|
||||
.coefficient = coefficient,
|
||||
.exponent = static_cast<int>(x.dec_.exp)};
|
||||
}
|
||||
};
|
||||
|
||||
using MpDecimal19 = MpDecimal<19>;
|
||||
using MpDecimal34 = MpDecimal<34>;
|
||||
using MpDecimal38 = MpDecimal<38>;
|
||||
|
||||
static_assert(DecomposableNumber<MpDecimal19>);
|
||||
static_assert(DecomposableNumber<MpDecimal34>);
|
||||
static_assert(DecomposableNumber<MpDecimal38>);
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
37
src/benchmarks/libxrpl/number/Number128.h
Normal file
37
src/benchmarks/libxrpl/number/Number128.h
Normal file
@@ -0,0 +1,37 @@
|
||||
#pragma once
|
||||
|
||||
// Slot for a 128-bit Number prototype.
|
||||
//
|
||||
// When <xrpl/basics/Number128.h> exists, CMake defines XRPL_BENCH_NUMBER128
|
||||
// (see src/benchmarks/libxrpl/CMakeLists.txt) and the type is benchmarked and
|
||||
// checked by every tool here with no further changes: xrpl.bench.number (single
|
||||
// operations and ledger formulas) and xrpl.bench.number_diff (rounding
|
||||
// correctness in all four modes). The header must provide `xrpl::Number128`
|
||||
// satisfying DecomposableNumber (see NumberLike.h), plus:
|
||||
//
|
||||
// static constexpr unsigned kDigits; // mantissa precision, e.g. 34 or 38
|
||||
//
|
||||
// number_diff rounds exact results onto a plain kDigits-digit grid. Up to 38
|
||||
// digits that is exact: 10^38 - 1 is below the signed 128-bit maximum, so
|
||||
// there is no capped region like today's Number has above 2^63 - 1. If the
|
||||
// prototype's grid differs, extend `Grid` in number_diff/main.cpp.
|
||||
|
||||
#ifdef XRPL_BENCH_NUMBER128
|
||||
|
||||
#include <xrpl/basics/Number128.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-macro-usage)
|
||||
#define XRPL_BENCH_HAS_NUMBER128 1
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
static_assert(
|
||||
DecomposableNumber<Number128>,
|
||||
"Number128 must provide Number's interface; see NumberLike.h");
|
||||
static_assert(Number128::kDigits <= 38, "Decomposed holds at most 38 digits");
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
|
||||
#endif
|
||||
91
src/benchmarks/libxrpl/number/NumberLike.h
Normal file
91
src/benchmarks/libxrpl/number/NumberLike.h
Normal file
@@ -0,0 +1,91 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <concepts>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
/**
|
||||
* Sign, coefficient, and exponent of a finite decimal value:
|
||||
* value = (negative ? -1 : 1) * coefficient * 10^exponent.
|
||||
*
|
||||
* The coefficient is 128 bits wide so the same struct can describe 34- and
|
||||
* 38-digit types. Different libraries normalize to different precisions, so
|
||||
* compare decompositions with `canonical()`, which strips trailing zeros.
|
||||
*/
|
||||
struct Decomposed
|
||||
{
|
||||
bool negative = false;
|
||||
unsigned __int128 coefficient = 0;
|
||||
int exponent = 0;
|
||||
|
||||
[[nodiscard]] constexpr Decomposed
|
||||
canonical() const noexcept
|
||||
{
|
||||
if (coefficient == 0)
|
||||
return {};
|
||||
Decomposed result = *this;
|
||||
while (result.coefficient % 10 == 0)
|
||||
{
|
||||
result.coefficient /= 10;
|
||||
++result.exponent;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
friend constexpr bool
|
||||
operator==(Decomposed const&, Decomposed const&) = default;
|
||||
};
|
||||
|
||||
[[nodiscard]] inline Decomposed
|
||||
decompose(Number const& x) noexcept
|
||||
{
|
||||
auto const m = x.mantissa();
|
||||
// Negating INT64_MIN is UB, but Number never produces it (see Number.h).
|
||||
auto const magnitude = static_cast<std::uint64_t>(m < 0 ? -m : m);
|
||||
return {.negative = m < 0, .coefficient = magnitude, .exponent = x.exponent()};
|
||||
}
|
||||
|
||||
/**
|
||||
* The subset of Number's interface that ledger code relies on. Benchmarks and
|
||||
* the divergence harness are written against this concept, so any type
|
||||
* satisfying it, including a future 128-bit Number, can be dropped in.
|
||||
*/
|
||||
template <class T>
|
||||
concept NumberLike = std::copyable<T> && std::totally_ordered<T> &&
|
||||
requires(T a, T b, std::int64_t mantissa, int exponent, unsigned n, Number::RoundingMode mode) {
|
||||
T{mantissa};
|
||||
T{mantissa, exponent};
|
||||
{ a + b } -> std::same_as<T>;
|
||||
{ a - b } -> std::same_as<T>;
|
||||
{ a * b } -> std::same_as<T>;
|
||||
{ a / b } -> std::same_as<T>;
|
||||
{ -a } -> std::same_as<T>;
|
||||
{ a += b } -> std::same_as<T&>;
|
||||
{ a -= b } -> std::same_as<T&>;
|
||||
{ a *= b } -> std::same_as<T&>;
|
||||
{ a /= b } -> std::same_as<T&>;
|
||||
static_cast<std::int64_t>(a);
|
||||
{ to_string(a) } -> std::same_as<std::string>;
|
||||
{ abs(a) } -> std::same_as<T>;
|
||||
{ power(a, n) } -> std::same_as<T>;
|
||||
{ root2(a) } -> std::same_as<T>;
|
||||
{ T::setround(mode) } -> std::same_as<Number::RoundingMode>;
|
||||
{ T::getround() } -> std::same_as<Number::RoundingMode>;
|
||||
};
|
||||
|
||||
/**
|
||||
* A NumberLike type whose exact value can be inspected digit by digit.
|
||||
*/
|
||||
template <class T>
|
||||
concept DecomposableNumber = NumberLike<T> && requires(T a) {
|
||||
{ decompose(a) } -> std::same_as<Decomposed>;
|
||||
};
|
||||
|
||||
static_assert(DecomposableNumber<Number>);
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
250
src/benchmarks/libxrpl/number/Operands.h
Normal file
250
src/benchmarks/libxrpl/number/Operands.h
Normal file
@@ -0,0 +1,250 @@
|
||||
#pragma once
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <limits>
|
||||
#include <random>
|
||||
#include <string_view>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
|
||||
/**
|
||||
* Operand count per batch: small enough to stay in L1 for 32-byte types.
|
||||
*/
|
||||
constexpr std::size_t kBatchSize = 1024;
|
||||
|
||||
/**
|
||||
* Raw operand: value = mantissa * 10^exponent.
|
||||
*/
|
||||
struct RawOperand
|
||||
{
|
||||
std::int64_t mantissa;
|
||||
int exponent;
|
||||
};
|
||||
|
||||
/**
|
||||
* Operand distributions. Each one models a class of values the ledger works
|
||||
* with, because cost depends on digit count and exponent alignment.
|
||||
*/
|
||||
enum class Dataset {
|
||||
/**
|
||||
* Full-width 19-digit mantissas in [10^18, 2^63-1], exponents in [-10, 10].
|
||||
*/
|
||||
Full,
|
||||
/**
|
||||
* IOU-like 16-digit mantissas in [10^15, 10^16-1], exponents in [-25, 5].
|
||||
*/
|
||||
Iou,
|
||||
/**
|
||||
* XRP drop amounts: integers with 1 to 17 digits (log-uniform), exponent 0.
|
||||
*/
|
||||
Drops,
|
||||
/**
|
||||
* MPT amounts: integers with 1 to 19 digits up to 2^63-1 (log-uniform), exponent 0.
|
||||
*/
|
||||
Mpt,
|
||||
/**
|
||||
* Rates in [1, 1.01) with full-width mantissas: the operand of interest and amortization
|
||||
* math, and of long multiply/divide chains that must not overflow.
|
||||
*/
|
||||
Rate,
|
||||
/**
|
||||
* Non-integers below 10^18 (19-digit mantissas, exponents in [-18, -1]): rounding to int64.
|
||||
*/
|
||||
Fraction,
|
||||
/**
|
||||
* Exact halves, m + 0.5 with m < 10^17: the tie cases of rounding to int64.
|
||||
*/
|
||||
Half,
|
||||
};
|
||||
|
||||
constexpr std::string_view
|
||||
toString(Dataset d)
|
||||
{
|
||||
switch (d)
|
||||
{
|
||||
case Dataset::Full:
|
||||
return "full";
|
||||
case Dataset::Iou:
|
||||
return "iou";
|
||||
case Dataset::Drops:
|
||||
return "drops";
|
||||
case Dataset::Mpt:
|
||||
return "mpt";
|
||||
case Dataset::Rate:
|
||||
return "rate";
|
||||
case Dataset::Fraction:
|
||||
return "fraction";
|
||||
case Dataset::Half:
|
||||
return "half";
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
namespace detail {
|
||||
|
||||
/**
|
||||
* Integer with a log-uniform digit count in [1, maxDigits], capped at `cap`.
|
||||
*/
|
||||
inline std::int64_t
|
||||
logUniformInteger(std::mt19937_64& rng, int maxDigits, std::int64_t cap)
|
||||
{
|
||||
auto const digits = std::uniform_int_distribution<int>{1, maxDigits}(rng);
|
||||
auto const lo = static_cast<std::int64_t>(kPowerOfTen[digits - 1]);
|
||||
auto const hi =
|
||||
digits >= 19 ? cap : std::min(cap, static_cast<std::int64_t>(kPowerOfTen[digits]) - 1);
|
||||
return std::uniform_int_distribution<std::int64_t>{lo, hi}(rng);
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
/**
|
||||
* Deterministic: the same dataset and seed always produce the same operands.
|
||||
*/
|
||||
inline std::vector<RawOperand>
|
||||
makeRaw(Dataset d, std::uint64_t seed, std::size_t count = kBatchSize)
|
||||
{
|
||||
constexpr auto kMaxInt64 = std::numeric_limits<std::int64_t>::max();
|
||||
constexpr auto kE18 = static_cast<std::int64_t>(kPowerOfTen[18]);
|
||||
constexpr auto kE16 = static_cast<std::int64_t>(kPowerOfTen[16]);
|
||||
constexpr auto kE15 = static_cast<std::int64_t>(kPowerOfTen[15]);
|
||||
|
||||
std::mt19937_64 rng{seed};
|
||||
std::vector<RawOperand> result;
|
||||
result.reserve(count);
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
switch (d)
|
||||
{
|
||||
case Dataset::Full:
|
||||
result.push_back(
|
||||
{std::uniform_int_distribution<std::int64_t>{kE18, kMaxInt64}(rng),
|
||||
std::uniform_int_distribution<int>{-10, 10}(rng)});
|
||||
break;
|
||||
case Dataset::Iou:
|
||||
result.push_back(
|
||||
{std::uniform_int_distribution<std::int64_t>{kE15, kE16 - 1}(rng),
|
||||
std::uniform_int_distribution<int>{-25, 5}(rng)});
|
||||
break;
|
||||
case Dataset::Drops:
|
||||
result.push_back({detail::logUniformInteger(rng, 17, kMaxInt64), 0});
|
||||
break;
|
||||
case Dataset::Mpt:
|
||||
result.push_back({detail::logUniformInteger(rng, 19, kMaxInt64), 0});
|
||||
break;
|
||||
case Dataset::Rate:
|
||||
// [10^18, 1.01 * 10^18) * 10^-18 = [1, 1.01)
|
||||
result.push_back(
|
||||
{std::uniform_int_distribution<std::int64_t>{
|
||||
kE18, kE18 + (kE18 / 100) - 1}(rng),
|
||||
-18});
|
||||
break;
|
||||
case Dataset::Fraction:
|
||||
result.push_back(
|
||||
{std::uniform_int_distribution<std::int64_t>{kE18, kMaxInt64}(rng),
|
||||
std::uniform_int_distribution<int>{-18, -1}(rng)});
|
||||
break;
|
||||
case Dataset::Half:
|
||||
result.push_back(
|
||||
{(std::uniform_int_distribution<std::int64_t>{0, (kE18 / 10) - 1}(rng) * 10) +
|
||||
5,
|
||||
-1});
|
||||
break;
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
template <NumberLike T>
|
||||
std::vector<T>
|
||||
materialize(std::vector<RawOperand> const& raw)
|
||||
{
|
||||
std::vector<T> result;
|
||||
result.reserve(raw.size());
|
||||
for (auto const& r : raw)
|
||||
result.push_back(T{r.mantissa, r.exponent});
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pairs of full-width operands whose exponents differ by exactly `gap`, so
|
||||
* addition must shift (and round away) `gap` digits of the smaller operand.
|
||||
*/
|
||||
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
|
||||
makeExponentGapPairs(int gap, std::uint64_t seed, std::size_t count = kBatchSize)
|
||||
{
|
||||
auto lhs = makeRaw(Dataset::Full, seed, count);
|
||||
auto rhs = makeRaw(Dataset::Full, seed + 1, count);
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
rhs[i].exponent = lhs[i].exponent - gap;
|
||||
return {std::move(lhs), std::move(rhs)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Pairs of nearly equal full-width operands, so subtraction cancels most
|
||||
* leading digits and the result must be renormalized by many places.
|
||||
*/
|
||||
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
|
||||
makeCancellationPairs(std::uint64_t seed, std::size_t count = kBatchSize)
|
||||
{
|
||||
auto lhs = makeRaw(Dataset::Full, seed, count);
|
||||
std::vector<RawOperand> rhs;
|
||||
rhs.reserve(count);
|
||||
std::mt19937_64 rng{seed + 1};
|
||||
std::uniform_int_distribution<std::int64_t> delta{1, 1000};
|
||||
for (auto const& l : lhs)
|
||||
rhs.push_back({l.mantissa - delta(rng), l.exponent});
|
||||
return {std::move(lhs), std::move(rhs)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Pairs whose exact sum is exactly halfway between two adjacent 19-digit
|
||||
* values: a = m * 10^e, b = 5 * 10^(e-1). Exercises round-half-even.
|
||||
*/
|
||||
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
|
||||
makeTiePairs(std::uint64_t seed, std::size_t count = kBatchSize)
|
||||
{
|
||||
auto lhs = makeRaw(Dataset::Full, seed, count);
|
||||
std::vector<RawOperand> rhs;
|
||||
rhs.reserve(count);
|
||||
for (auto const& l : lhs)
|
||||
rhs.push_back({5, l.exponent - 1});
|
||||
return {std::move(lhs), std::move(rhs)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Pairs whose sum straddles 2^63-1 (Number::kMaxRep), where Number's
|
||||
* representable values switch from a step of 1 to a step of 10. The addend has
|
||||
* one more decimal place than the base, so most sums fall between
|
||||
* representable values and must be rounded; sums are within ±200 of the cap,
|
||||
* so a fair share land in (2^63-1, 2^63), the interval where LargeLegacy's
|
||||
* known cusp-rounding error shows.
|
||||
*/
|
||||
inline std::pair<std::vector<RawOperand>, std::vector<RawOperand>>
|
||||
makeCuspPairs(std::uint64_t seed, std::size_t count = kBatchSize)
|
||||
{
|
||||
constexpr auto kMaxInt64 = std::numeric_limits<std::int64_t>::max();
|
||||
std::mt19937_64 rng{seed};
|
||||
std::uniform_int_distribution<std::int64_t> base{kMaxInt64 - 100, kMaxInt64};
|
||||
std::uniform_int_distribution<std::int64_t> addend{1, 2'000};
|
||||
std::uniform_int_distribution<int> exponent{-10, 10};
|
||||
std::pair<std::vector<RawOperand>, std::vector<RawOperand>> result;
|
||||
result.first.reserve(count);
|
||||
result.second.reserve(count);
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
auto const e = exponent(rng);
|
||||
result.first.push_back({base(rng), e});
|
||||
result.second.push_back({addend(rng), e - 1});
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
} // namespace xrpl::number_bench
|
||||
309
src/benchmarks/libxrpl/number/plot_results.py
Executable file
309
src/benchmarks/libxrpl/number/plot_results.py
Executable file
@@ -0,0 +1,309 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Render the figures in docs/NumberDecimalBenchmark.md from benchmark JSON.
|
||||
|
||||
Usage:
|
||||
plot_results.py CORE_JSON KERNELS_JSON OUT_DIR
|
||||
|
||||
CORE_JSON is Google Benchmark JSON from xrpl.bench.number filtered to
|
||||
'^(add|mul|div)/full/'; KERNELS_JSON from '^kernel/' with mpdecimal enabled
|
||||
(for the accuracy counters). Both need --benchmark_repetitions so medians are
|
||||
available. Writes self-contained SVG files with light and dark variants
|
||||
(prefers-color-scheme) and no dependencies beyond the standard library.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
from xml.sax.saxutils import escape
|
||||
|
||||
# Categorical slots 1-3 of the reference palette (validated all-pairs, light
|
||||
# and dark). Color encodes the type family; every bar is also labeled.
|
||||
FAMILIES = {
|
||||
"number": ("Number today (19 digits)", "#2a78d6", "#3987e5"),
|
||||
"ieee64": ("16-digit types", "#eb6834", "#d95926"),
|
||||
"wide": ("34- and 38-digit types", "#1baf7a", "#199e70"),
|
||||
}
|
||||
|
||||
# (benchmark subject, row label, family)
|
||||
CORE_ROWS = [
|
||||
("Number.Large330", "Number (Large330)", "number"),
|
||||
("BoostDecimal64", "Boost decimal64", "ieee64"),
|
||||
("BoostDecimalFast64", "Boost decimal_fast64", "ieee64"),
|
||||
("IntelBid64", "Intel BID64", "ieee64"),
|
||||
("BoostDecimal128", "Boost decimal128", "wide"),
|
||||
("BoostDecimalFast128", "Boost decimal_fast128", "wide"),
|
||||
("IntelBid128", "Intel BID128", "wide"),
|
||||
("MpDecimal34", "mpdecimal 34", "wide"),
|
||||
("MpDecimal38", "mpdecimal 38", "wide"),
|
||||
]
|
||||
|
||||
KERNEL_ROWS = [
|
||||
("Number.Large330", "Number (Large330)", "number"),
|
||||
("BoostDecimal64", "Boost decimal64", "ieee64"),
|
||||
("IntelBid64", "Intel BID64", "ieee64"),
|
||||
("BoostDecimal128", "Boost decimal128", "wide"),
|
||||
("IntelBid128", "Intel BID128", "wide"),
|
||||
("MpDecimal34", "mpdecimal 34", "wide"),
|
||||
("MpDecimal38", "mpdecimal 38", "wide"),
|
||||
]
|
||||
|
||||
FONT = "system-ui, -apple-system, sans-serif"
|
||||
ROW_H = 24
|
||||
BAR_H = 12
|
||||
LABEL_W = 150
|
||||
PANEL_W = 200
|
||||
PANEL_GAP = 28
|
||||
TOP = 78
|
||||
BOTTOM = 30
|
||||
|
||||
|
||||
@dataclass
|
||||
class Panel:
|
||||
title: str
|
||||
unit: str
|
||||
values: dict[str, float]
|
||||
fmt: Callable[[float], str]
|
||||
# Panels with the same group share an x scale; use it only for panels
|
||||
# with the same unit, so bar lengths are comparable.
|
||||
group: str = ""
|
||||
|
||||
|
||||
def medians(path: str) -> dict[str, dict]:
|
||||
"""Map run_name to its median aggregate."""
|
||||
data = json.loads(Path(path).read_text())
|
||||
return {
|
||||
b["run_name"]: b
|
||||
for b in data["benchmarks"]
|
||||
if b.get("aggregate_name") == "median"
|
||||
}
|
||||
|
||||
|
||||
def ns_per_item(b: dict) -> float:
|
||||
return 1e9 / b["items_per_second"]
|
||||
|
||||
|
||||
def nice_max(v: float) -> tuple[float, list[float]]:
|
||||
"""Axis maximum and ticks: 4-5 round steps covering v."""
|
||||
for step in [
|
||||
1,
|
||||
2,
|
||||
2.5,
|
||||
5,
|
||||
10,
|
||||
20,
|
||||
25,
|
||||
50,
|
||||
100,
|
||||
200,
|
||||
250,
|
||||
500,
|
||||
1000,
|
||||
2000,
|
||||
2500,
|
||||
5000,
|
||||
10000,
|
||||
20000,
|
||||
25000,
|
||||
50000,
|
||||
]:
|
||||
if v / step <= 5:
|
||||
top = step * -(-v // step)
|
||||
return top, [step * i for i in range(int(top / step) + 1)]
|
||||
return v, [0, v]
|
||||
|
||||
|
||||
def bar_path(x: float, y: float, w: float, h: float, r: float = 4) -> str:
|
||||
"""Bar anchored at the baseline (left) with a rounded data end (right)."""
|
||||
r = min(r, w, h / 2)
|
||||
return (
|
||||
f"M{x:.1f},{y:.1f} H{x + w - r:.1f} "
|
||||
f"Q{x + w:.1f},{y:.1f} {x + w:.1f},{y + r:.1f} "
|
||||
f"V{y + h - r:.1f} Q{x + w:.1f},{y + h:.1f} {x + w - r:.1f},{y + h:.1f} "
|
||||
f"H{x:.1f} Z"
|
||||
)
|
||||
|
||||
|
||||
def render(
|
||||
title: str, rows: list[tuple[str, str, str]], panels: list[Panel], out: Path
|
||||
) -> None:
|
||||
"""Horizontal bar panels sharing one column of row labels."""
|
||||
width = LABEL_W + len(panels) * (PANEL_W + PANEL_GAP)
|
||||
height = TOP + len(rows) * ROW_H + BOTTOM
|
||||
css_light = "".join(
|
||||
f".f-{k}{{fill:{light}}}" for k, (_, light, _) in FAMILIES.items()
|
||||
)
|
||||
css_dark = "".join(f".f-{k}{{fill:{dark}}}" for k, (_, _, dark) in FAMILIES.items())
|
||||
out_lines = [
|
||||
f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 {width} {height}" '
|
||||
f'width="{width}" height="{height}" role="img" '
|
||||
f'aria-label="{escape(title)}">',
|
||||
"<style>",
|
||||
f"text{{font-family:{FONT};font-size:12px}}",
|
||||
".bg{fill:#fcfcfb}.t1{fill:#0b0b0b}.t2{fill:#52514e}"
|
||||
".grid{stroke:#e4e3df;stroke-width:1}.base{stroke:#b9b8b2;stroke-width:1}"
|
||||
+ css_light,
|
||||
"@media (prefers-color-scheme: dark){"
|
||||
".bg{fill:#1a1a19}.t1{fill:#ffffff}.t2{fill:#c3c2b7}"
|
||||
".grid{stroke:#383835}.base{stroke:#5e5d58}" + css_dark + "}",
|
||||
"</style>",
|
||||
f'<rect class="bg" width="{width}" height="{height}"/>',
|
||||
f'<text class="t1" x="0" y="18" font-size="15" font-weight="600">'
|
||||
f"{escape(title)}</text>",
|
||||
]
|
||||
# Legend: always present for more than one family.
|
||||
lx = 0.0
|
||||
for key, (label, _, _) in FAMILIES.items():
|
||||
out_lines.append(
|
||||
f'<rect class="f-{key}" x="{lx}" y="32" width="12" height="12" rx="3"/>'
|
||||
)
|
||||
out_lines.append(
|
||||
f'<text class="t2" x="{lx + 18}" y="42">{escape(label)}</text>'
|
||||
)
|
||||
lx += 18 + 7.2 * len(label) + 24
|
||||
|
||||
for i, (_, label, _) in enumerate(rows):
|
||||
y = TOP + i * ROW_H + ROW_H / 2 + 4
|
||||
out_lines.append(f'<text class="t1" x="0" y="{y:.1f}">{escape(label)}</text>')
|
||||
|
||||
for p_index, panel in enumerate(panels):
|
||||
x0 = LABEL_W + p_index * (PANEL_W + PANEL_GAP)
|
||||
vmax = max(
|
||||
p.values.get(s, 0)
|
||||
for p in panels
|
||||
if p is panel or (panel.group and p.group == panel.group)
|
||||
for s, _, _ in rows
|
||||
)
|
||||
top, ticks = nice_max(vmax * 1.08)
|
||||
scale = (PANEL_W - 40) / top
|
||||
out_lines.append(
|
||||
f'<text class="t1" x="{x0}" y="{TOP - 14}" font-weight="600">'
|
||||
f"{escape(panel.title)}</text>"
|
||||
)
|
||||
y_end = TOP + len(rows) * ROW_H
|
||||
for t in ticks:
|
||||
gx = x0 + t * scale
|
||||
out_lines.append(
|
||||
f'<line class="grid" x1="{gx:.1f}" y1="{TOP}" x2="{gx:.1f}" y2="{y_end}"/>'
|
||||
)
|
||||
out_lines.append(
|
||||
f'<text class="t2" x="{gx:.1f}" y="{y_end + 16}" text-anchor="middle" '
|
||||
f'font-size="11">{t:g}</text>'
|
||||
)
|
||||
out_lines.append(
|
||||
f'<text class="t2" x="{x0 + PANEL_W - 40:.1f}" y="{y_end + 28}" '
|
||||
f'text-anchor="end" font-size="11">{escape(panel.unit)}</text>'
|
||||
)
|
||||
for i, (subject, label, family) in enumerate(rows):
|
||||
if subject not in panel.values:
|
||||
continue
|
||||
v = panel.values[subject]
|
||||
y = TOP + i * ROW_H + (ROW_H - BAR_H) / 2
|
||||
w = max(v * scale, 1.0)
|
||||
out_lines.append(
|
||||
f'<path class="f-{family}" d="{bar_path(x0, y, w, BAR_H)}">'
|
||||
f"<title>{escape(label)}: {panel.fmt(v)} {escape(panel.unit)}</title></path>"
|
||||
)
|
||||
out_lines.append(
|
||||
f'<text class="t2" x="{x0 + w + 5:.1f}" y="{y + BAR_H - 2:.1f}" '
|
||||
f'font-size="11">{panel.fmt(v)}</text>'
|
||||
)
|
||||
out_lines.append(
|
||||
f'<line class="base" x1="{x0}" y1="{TOP}" x2="{x0}" y2="{y_end}"/>'
|
||||
)
|
||||
out_lines.append("</svg>")
|
||||
out.write_text("\n".join(out_lines) + "\n")
|
||||
|
||||
|
||||
def fmt_ns(v: float) -> str:
|
||||
return f"{v:.0f}" if v >= 10 else f"{v:.1f}"
|
||||
|
||||
|
||||
def fmt_us(v: float) -> str:
|
||||
return f"{v:.0f}" if v >= 10 else f"{v:.1f}"
|
||||
|
||||
|
||||
def fmt_digits(v: float) -> str:
|
||||
return f"{v:.1f}"
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
if len(argv) != 4:
|
||||
print(__doc__, file=sys.stderr)
|
||||
return 2
|
||||
core = medians(argv[1])
|
||||
kernels = medians(argv[2])
|
||||
out_dir = Path(argv[3])
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
core_panels = [
|
||||
Panel(
|
||||
op_title,
|
||||
"ns per operation",
|
||||
{
|
||||
s: ns_per_item(core[f"{op}/full/{s}"])
|
||||
for s, _, _ in CORE_ROWS
|
||||
if f"{op}/full/{s}" in core
|
||||
},
|
||||
fmt_ns,
|
||||
group="ns",
|
||||
)
|
||||
for op, op_title in [("add", "Add"), ("mul", "Multiply"), ("div", "Divide")]
|
||||
]
|
||||
render(
|
||||
"Cost of one operation (full-width operands, lower is faster)",
|
||||
CORE_ROWS,
|
||||
core_panels,
|
||||
out_dir / "number-core-ops.svg",
|
||||
)
|
||||
|
||||
def kernel(name: str) -> dict[str, dict]:
|
||||
prefix = f"kernel/{name}/"
|
||||
return {k[len(prefix) :]: v for k, v in kernels.items() if k.startswith(prefix)}
|
||||
|
||||
for name, title, unit, divisor, fmt in [
|
||||
("amm_swap_in", "AMM swap-in", "ns per call", 1.0, fmt_ns),
|
||||
(
|
||||
"loan_amortize360",
|
||||
"360-payment amortization",
|
||||
"µs per schedule",
|
||||
1e3,
|
||||
fmt_us,
|
||||
),
|
||||
]:
|
||||
k = kernel(name)
|
||||
render(
|
||||
f"{title}: correct digits (higher is better) and cost (lower is faster)",
|
||||
KERNEL_ROWS,
|
||||
[
|
||||
Panel(
|
||||
"Correct digits, worst case",
|
||||
"significant digits",
|
||||
{s: b["digits_min"] for s, b in k.items()},
|
||||
fmt_digits,
|
||||
group="digits",
|
||||
),
|
||||
Panel(
|
||||
"Correct digits, median",
|
||||
"significant digits",
|
||||
{s: b["digits_median"] for s, b in k.items()},
|
||||
fmt_digits,
|
||||
group="digits",
|
||||
),
|
||||
Panel(
|
||||
"Cost",
|
||||
unit,
|
||||
{s: ns_per_item(b) / divisor for s, b in k.items()},
|
||||
fmt,
|
||||
),
|
||||
],
|
||||
out_dir / f"number-{name.replace('_', '-')}.svg",
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
555
src/benchmarks/libxrpl/number_diff/main.cpp
Normal file
555
src/benchmarks/libxrpl/number_diff/main.cpp
Normal file
@@ -0,0 +1,555 @@
|
||||
// number_diff: measures how far Number's results are from correctly rounded
|
||||
// results. See docs/NumberDecimalBenchmark.md, "Divergence harness".
|
||||
//
|
||||
// For every operation, the operands are converted exactly into a 96-digit
|
||||
// mpdecimal value and the operation is evaluated there (correctly rounded at
|
||||
// 96 digits). That near-exact result is then rounded onto the subject type's
|
||||
// own grid of representable values in the active rounding mode, and compared
|
||||
// with what the subject actually returned.
|
||||
//
|
||||
// Rounding a 96-digit result again creates a false tie only if digits 20-96
|
||||
// of a non-terminating result happen to round to exactly 5000...0, which
|
||||
// will not occur at these sample sizes.
|
||||
//
|
||||
// Usage: xrpl.bench.number_diff [--samples N] [--csv]
|
||||
|
||||
#include <xrpl/basics/Number.h>
|
||||
|
||||
#include <benchmarks/libxrpl/number/BoostDecimal.h>
|
||||
#include <benchmarks/libxrpl/number/MpDecimal.h>
|
||||
#include <benchmarks/libxrpl/number/Number128.h>
|
||||
#include <benchmarks/libxrpl/number/NumberLike.h>
|
||||
#include <benchmarks/libxrpl/number/Operands.h>
|
||||
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
#include <benchmarks/libxrpl/number/IntelBid.h>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdlib>
|
||||
#include <exception>
|
||||
#include <format>
|
||||
#include <functional>
|
||||
#include <iostream>
|
||||
#include <limits>
|
||||
#include <optional>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::number_bench {
|
||||
namespace {
|
||||
|
||||
using Exact = MpDecimal<96>;
|
||||
|
||||
constexpr std::uint64_t kSeedLhs = 0xd1ff'0001;
|
||||
constexpr std::uint64_t kSeedRhs = 0xd1ff'0002;
|
||||
|
||||
/**
|
||||
* The set of values a type can represent, as far as rounding is concerned.
|
||||
* `int64Cap` models Number's large scales, whose mantissas above 2^63-1 must
|
||||
* end in 0 (see Number.h, "External Interface").
|
||||
*/
|
||||
struct Grid
|
||||
{
|
||||
unsigned digits;
|
||||
bool int64Cap = false;
|
||||
};
|
||||
|
||||
/**
|
||||
* Integers: the grid of conversions to int64.
|
||||
*/
|
||||
struct IntegerGrid
|
||||
{
|
||||
};
|
||||
|
||||
// ---- Exact arithmetic helpers ----
|
||||
|
||||
Exact
|
||||
fromSubject(DecomposableNumber auto const& x)
|
||||
{
|
||||
return Exact::fromDecomposed(decompose(x));
|
||||
}
|
||||
|
||||
/**
|
||||
* Rounds a non-negative value to `digits` significant digits, toward (down) or away from (up)
|
||||
* zero.
|
||||
*/
|
||||
Exact
|
||||
roundMagnitude(Exact x, unsigned digits, bool up)
|
||||
{
|
||||
mpd_context_t ctx;
|
||||
mpd_maxcontext(&ctx);
|
||||
ctx.prec = digits;
|
||||
ctx.round = up ? MPD_ROUND_UP : MPD_ROUND_DOWN;
|
||||
ctx.traps = 0;
|
||||
std::uint32_t status = 0;
|
||||
mpd_qfinalize(x.get(), &ctx, &status);
|
||||
return x;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rounds a non-negative value to an integer, toward (down) or away from (up) zero.
|
||||
*/
|
||||
Exact
|
||||
roundMagnitudeToInteger(Exact const& x, bool up)
|
||||
{
|
||||
mpd_context_t ctx;
|
||||
mpd_maxcontext(&ctx);
|
||||
ctx.round = up ? MPD_ROUND_UP : MPD_ROUND_DOWN;
|
||||
ctx.traps = 0;
|
||||
std::uint32_t status = 0;
|
||||
Exact result;
|
||||
mpd_qround_to_int(result.get(), x.get(), &ctx, &status);
|
||||
return result;
|
||||
}
|
||||
|
||||
bool
|
||||
isZero(Exact const& x)
|
||||
{
|
||||
return mpd_iszero(x.get()) != 0;
|
||||
}
|
||||
|
||||
bool
|
||||
isNegative(Exact const& x)
|
||||
{
|
||||
return mpd_isnegative(x.get()) != 0 && !isZero(x);
|
||||
}
|
||||
|
||||
/**
|
||||
* True if q (a non-negative integer) is even.
|
||||
*/
|
||||
bool
|
||||
isEvenInteger(Exact const& q)
|
||||
{
|
||||
auto const s = to_string(q);
|
||||
auto const pos = s.find_first_of("Ee");
|
||||
// Positive exponent in scientific notation means trailing zeros.
|
||||
if (pos != std::string::npos && s[pos + 1] != '-')
|
||||
return true;
|
||||
auto const digits = s.substr(0, pos);
|
||||
return ((digits.back() - '0') % 2) == 0;
|
||||
}
|
||||
|
||||
// ---- Candidate representable values around an exact result ----
|
||||
|
||||
/**
|
||||
* The representable magnitudes immediately at or below and at or above |x|.
|
||||
*/
|
||||
struct Bracket
|
||||
{
|
||||
Exact lo;
|
||||
Exact hi;
|
||||
/**
|
||||
* Grid spacing at lo: lo / spacing is lo's integer coefficient.
|
||||
*/
|
||||
Exact loSpacing;
|
||||
/**
|
||||
* lo is 2^63-1 and hi is its successor 2^63+3: the two candidates straddle Number's cap.
|
||||
*/
|
||||
bool straddlesCap = false;
|
||||
};
|
||||
|
||||
Bracket
|
||||
bracket(Exact const& magnitude, Grid grid)
|
||||
{
|
||||
auto const adjusted = static_cast<int>(mpd_adjexp(magnitude.get()));
|
||||
auto lo = roundMagnitude(magnitude, grid.digits, false);
|
||||
auto hi = roundMagnitude(magnitude, grid.digits, true);
|
||||
Exact loSpacing{1, adjusted - static_cast<int>(grid.digits) + 1};
|
||||
bool straddlesCap = false;
|
||||
if (grid.int64Cap)
|
||||
{
|
||||
// Above cap, only every tenth 19-digit value is representable.
|
||||
Exact const cap{static_cast<std::int64_t>(Number::kMaxRep), adjusted - 18};
|
||||
if (hi > cap)
|
||||
hi = roundMagnitude(magnitude, 18, true);
|
||||
if (lo > cap)
|
||||
{
|
||||
auto const lo18 = roundMagnitude(magnitude, 18, false);
|
||||
if (lo18 < cap)
|
||||
{
|
||||
lo = cap;
|
||||
straddlesCap = true;
|
||||
}
|
||||
else
|
||||
{
|
||||
lo = lo18;
|
||||
loSpacing = Exact{1, adjusted - 17};
|
||||
}
|
||||
}
|
||||
}
|
||||
return {std::move(lo), std::move(hi), std::move(loSpacing), straddlesCap};
|
||||
}
|
||||
|
||||
/**
|
||||
* The correctly rounded magnitude, or nullopt if two candidates are equally valid (a cusp tie).
|
||||
*/
|
||||
std::optional<Exact>
|
||||
choose(Exact const& magnitude, Bracket const& b, bool negative, Number::RoundingMode mode)
|
||||
{
|
||||
using enum Number::RoundingMode;
|
||||
if (b.lo == b.hi)
|
||||
return b.lo;
|
||||
switch (mode)
|
||||
{
|
||||
case TowardsZero:
|
||||
return b.lo;
|
||||
case Downward:
|
||||
return negative ? b.hi : b.lo;
|
||||
case Upward:
|
||||
return negative ? b.lo : b.hi;
|
||||
case ToNearest:
|
||||
break;
|
||||
}
|
||||
auto const below = magnitude - b.lo;
|
||||
auto const above = b.hi - magnitude;
|
||||
if (below < above)
|
||||
return b.lo;
|
||||
if (above < below)
|
||||
return b.hi;
|
||||
// Tie at the cap: 2^63-1 and 2^63+3 are both odd, so "even" is undefined.
|
||||
if (b.straddlesCap)
|
||||
return std::nullopt;
|
||||
return isEvenInteger(b.lo / b.loSpacing) ? b.lo : b.hi;
|
||||
}
|
||||
|
||||
// ---- Tallies ----
|
||||
|
||||
struct Tally
|
||||
{
|
||||
std::size_t samples = 0;
|
||||
/**
|
||||
* Exactly the correctly rounded result.
|
||||
*/
|
||||
std::size_t correct = 0;
|
||||
/**
|
||||
* The other neighbor of the exact result: rounded in the wrong direction.
|
||||
*/
|
||||
std::size_t wrongNeighbor = 0;
|
||||
/**
|
||||
* Neither neighbor: off by more than one step.
|
||||
*/
|
||||
std::size_t worse = 0;
|
||||
/**
|
||||
* Returned zero for a non-zero result.
|
||||
*/
|
||||
std::size_t flushed = 0;
|
||||
/**
|
||||
* Threw instead of returning.
|
||||
*/
|
||||
std::size_t threw = 0;
|
||||
/**
|
||||
* Largest |result - exact| / local grid spacing.
|
||||
*/
|
||||
double maxError = 0;
|
||||
};
|
||||
|
||||
void
|
||||
classify(Tally& t, Exact const& result, Exact const& exact, Grid grid, Number::RoundingMode mode)
|
||||
{
|
||||
if (isZero(exact))
|
||||
{
|
||||
++(isZero(result) ? t.correct : t.worse);
|
||||
return;
|
||||
}
|
||||
if (isZero(result))
|
||||
{
|
||||
++t.flushed;
|
||||
return;
|
||||
}
|
||||
bool const negative = isNegative(exact);
|
||||
auto const magnitude = abs(exact);
|
||||
auto const b = bracket(magnitude, grid);
|
||||
auto const resultMagnitude = abs(result);
|
||||
auto const chosen = choose(magnitude, b, negative, mode);
|
||||
|
||||
bool const sameSign = isNegative(result) == negative;
|
||||
bool const isLo = sameSign && resultMagnitude == b.lo;
|
||||
bool const isHi = sameSign && resultMagnitude == b.hi;
|
||||
if (sameSign && (chosen ? resultMagnitude == *chosen : (isLo || isHi)))
|
||||
++t.correct;
|
||||
else if (isLo || isHi)
|
||||
++t.wrongNeighbor;
|
||||
else
|
||||
++t.worse;
|
||||
|
||||
if (b.lo != b.hi)
|
||||
{
|
||||
auto const error = abs(result - exact) / (b.hi - b.lo);
|
||||
t.maxError = std::max(t.maxError, std::stod(to_string(error)));
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
classifyInteger(Tally& t, std::int64_t result, Exact const& exact, Number::RoundingMode mode)
|
||||
{
|
||||
bool const negative = isNegative(exact);
|
||||
auto const magnitude = abs(exact);
|
||||
auto const lo = roundMagnitudeToInteger(magnitude, false);
|
||||
auto const hi = roundMagnitudeToInteger(magnitude, true);
|
||||
Bracket const b{lo, hi, Exact{1}};
|
||||
auto const chosen = *choose(magnitude, b, negative, mode);
|
||||
Exact const r{result};
|
||||
auto const resultMagnitude = abs(r);
|
||||
bool const sameSign = result == 0 || (result < 0) == negative;
|
||||
if (sameSign && resultMagnitude == chosen)
|
||||
++t.correct;
|
||||
else if (sameSign && (resultMagnitude == lo || resultMagnitude == hi))
|
||||
++t.wrongNeighbor;
|
||||
else
|
||||
++t.worse;
|
||||
t.maxError = std::max(t.maxError, std::stod(to_string(abs(r - exact))));
|
||||
}
|
||||
|
||||
// ---- Running cells ----
|
||||
|
||||
struct Row
|
||||
{
|
||||
std::string subject;
|
||||
Number::RoundingMode mode;
|
||||
std::string op;
|
||||
std::string data;
|
||||
Tally tally;
|
||||
};
|
||||
|
||||
template <DecomposableNumber T, class BinaryOp>
|
||||
Tally
|
||||
runBinary(
|
||||
std::vector<RawOperand> const& lhs,
|
||||
std::vector<RawOperand> const& rhs,
|
||||
BinaryOp op,
|
||||
Grid grid,
|
||||
Number::RoundingMode mode)
|
||||
{
|
||||
Tally t;
|
||||
for (std::size_t i = 0; i < lhs.size(); ++i)
|
||||
{
|
||||
T const a{lhs[i].mantissa, lhs[i].exponent};
|
||||
T const b{rhs[i].mantissa, rhs[i].exponent};
|
||||
auto const exact = op(fromSubject(a), fromSubject(b));
|
||||
++t.samples;
|
||||
try
|
||||
{
|
||||
classify(t, fromSubject(T{op(a, b)}), exact, grid, mode);
|
||||
}
|
||||
catch (std::exception const&)
|
||||
{
|
||||
++t.threw;
|
||||
}
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
template <DecomposableNumber T, class UnaryOp>
|
||||
Tally
|
||||
runUnary(std::vector<RawOperand> const& operands, UnaryOp op, Grid grid, Number::RoundingMode mode)
|
||||
{
|
||||
Tally t;
|
||||
for (auto const& raw : operands)
|
||||
{
|
||||
T const a{raw.mantissa, raw.exponent};
|
||||
auto const exact = op(fromSubject(a));
|
||||
++t.samples;
|
||||
try
|
||||
{
|
||||
classify(t, fromSubject(T{op(a)}), exact, grid, mode);
|
||||
}
|
||||
catch (std::exception const&)
|
||||
{
|
||||
++t.threw;
|
||||
}
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
template <DecomposableNumber T>
|
||||
Tally
|
||||
runToInt64(std::vector<RawOperand> const& operands, Number::RoundingMode mode)
|
||||
{
|
||||
Tally t;
|
||||
for (auto const& raw : operands)
|
||||
{
|
||||
T const a{raw.mantissa, raw.exponent};
|
||||
auto const exact = fromSubject(a);
|
||||
++t.samples;
|
||||
try
|
||||
{
|
||||
classifyInteger(t, static_cast<std::int64_t>(a), exact, mode);
|
||||
}
|
||||
catch (std::exception const&)
|
||||
{
|
||||
++t.threw;
|
||||
}
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs every operation and dataset for one subject type and rounding mode.
|
||||
*/
|
||||
template <DecomposableNumber T>
|
||||
void
|
||||
runSubject(
|
||||
std::string_view subject,
|
||||
Grid grid,
|
||||
Number::RoundingMode mode,
|
||||
std::size_t samples,
|
||||
std::vector<Row>& rows)
|
||||
{
|
||||
auto const saved = T::setround(mode);
|
||||
auto const record = [&](std::string op, Dataset d, Tally t) {
|
||||
rows.push_back({std::string{subject}, mode, std::move(op), std::string{toString(d)}, t});
|
||||
};
|
||||
auto const recordPairs = [&](std::string op, std::string data, Tally t) {
|
||||
rows.push_back({std::string{subject}, mode, std::move(op), std::move(data), t});
|
||||
};
|
||||
|
||||
auto const plus = [](auto const& x, auto const& y) { return x + y; };
|
||||
auto const minus = [](auto const& x, auto const& y) { return x - y; };
|
||||
auto const times = [](auto const& x, auto const& y) { return x * y; };
|
||||
auto const divide = [](auto const& x, auto const& y) { return x / y; };
|
||||
|
||||
for (auto const d : {Dataset::Full, Dataset::Iou, Dataset::Mpt})
|
||||
{
|
||||
auto const lhs = makeRaw(d, kSeedLhs, samples);
|
||||
auto const rhs = makeRaw(d, kSeedRhs, samples);
|
||||
record("add", d, runBinary<T>(lhs, rhs, plus, grid, mode));
|
||||
record("sub", d, runBinary<T>(lhs, rhs, minus, grid, mode));
|
||||
record("mul", d, runBinary<T>(lhs, rhs, times, grid, mode));
|
||||
record("div", d, runBinary<T>(lhs, rhs, divide, grid, mode));
|
||||
record("root2", d, runUnary<T>(lhs, [](auto const& x) { return root2(x); }, grid, mode));
|
||||
}
|
||||
|
||||
for (int const gap : {1, 5, 18})
|
||||
{
|
||||
auto const [lhs, rhs] = makeExponentGapPairs(gap, kSeedLhs, samples);
|
||||
recordPairs("add", std::format("gap{}", gap), runBinary<T>(lhs, rhs, plus, grid, mode));
|
||||
recordPairs("sub", std::format("gap{}", gap), runBinary<T>(lhs, rhs, minus, grid, mode));
|
||||
}
|
||||
{
|
||||
auto const [lhs, rhs] = makeCancellationPairs(kSeedLhs, samples);
|
||||
recordPairs("sub", "cancel", runBinary<T>(lhs, rhs, minus, grid, mode));
|
||||
}
|
||||
{
|
||||
auto const [lhs, rhs] = makeTiePairs(kSeedLhs, samples);
|
||||
recordPairs("add", "tie", runBinary<T>(lhs, rhs, plus, grid, mode));
|
||||
}
|
||||
{
|
||||
auto const [lhs, rhs] = makeCuspPairs(kSeedLhs, samples);
|
||||
recordPairs("add", "cusp", runBinary<T>(lhs, rhs, plus, grid, mode));
|
||||
}
|
||||
|
||||
auto const rates = makeRaw(Dataset::Rate, kSeedLhs, samples);
|
||||
for (unsigned const n : {12u, 360u})
|
||||
{
|
||||
record(
|
||||
std::format("power{}", n),
|
||||
Dataset::Rate,
|
||||
runUnary<T>(rates, [n](auto const& x) { return power(x, n); }, grid, mode));
|
||||
}
|
||||
|
||||
for (auto const d : {Dataset::Fraction, Dataset::Half})
|
||||
record("to_int64", d, runToInt64<T>(makeRaw(d, kSeedLhs, samples), mode));
|
||||
|
||||
T::setround(saved);
|
||||
}
|
||||
|
||||
void
|
||||
print(std::vector<Row> const& rows, bool csv)
|
||||
{
|
||||
if (csv)
|
||||
std::cout << "subject,mode,op,data,samples,correct,wrong_neighbor,worse,flushed,threw,max_"
|
||||
"error\n";
|
||||
else
|
||||
std::cout << "| subject | mode | op | data | samples | correct | wrong neighbor | worse | "
|
||||
"flushed | threw | max error |\n"
|
||||
"|---|---|---|---|---:|---:|---:|---:|---:|---:|---:|\n";
|
||||
for (auto const& r : rows)
|
||||
{
|
||||
auto const& t = r.tally;
|
||||
auto const mode = to_string(r.mode);
|
||||
auto const fmt = csv ? "{},{},{},{},{},{},{},{},{},{},{:.4g}\n"
|
||||
: "| {} | {} | {} | {} | {} | {} | {} | {} | {} | {} | {:.4g} |\n";
|
||||
std::cout << std::vformat(
|
||||
fmt,
|
||||
std::make_format_args(
|
||||
r.subject,
|
||||
mode,
|
||||
r.op,
|
||||
r.data,
|
||||
t.samples,
|
||||
t.correct,
|
||||
t.wrongNeighbor,
|
||||
t.worse,
|
||||
t.flushed,
|
||||
t.threw,
|
||||
t.maxError));
|
||||
}
|
||||
}
|
||||
|
||||
int
|
||||
run(int argc, char** argv)
|
||||
{
|
||||
std::size_t samples = 10'000;
|
||||
bool csv = false;
|
||||
for (int i = 1; i < argc; ++i)
|
||||
{
|
||||
std::string_view const arg = argv[i];
|
||||
if (arg == "--samples" && i + 1 < argc)
|
||||
samples = std::strtoull(argv[++i], nullptr, 10);
|
||||
else if (arg == "--csv")
|
||||
csv = true;
|
||||
else
|
||||
{
|
||||
std::cerr << "usage: " << argv[0] << " [--samples N] [--csv]\n";
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
|
||||
using enum Number::RoundingMode;
|
||||
using enum MantissaRange::MantissaScale;
|
||||
constexpr auto kModes = std::to_array({ToNearest, TowardsZero, Downward, Upward});
|
||||
|
||||
std::vector<Row> rows;
|
||||
for (auto const mode : kModes)
|
||||
{
|
||||
for (auto const scale : {Small, LargeLegacy, Large320, Large330})
|
||||
{
|
||||
NumberMantissaScaleGuard const guard{scale};
|
||||
auto const grid =
|
||||
scale == Small ? Grid{.digits = 16} : Grid{.digits = 19, .int64Cap = true};
|
||||
runSubject<Number>("Number." + to_string(scale), grid, mode, samples, rows);
|
||||
}
|
||||
// Validate the oracle and the grid logic against IEEE implementations,
|
||||
// which must be correctly rounded for + - * / and sqrt.
|
||||
runSubject<BoostDecimal64>("BoostDecimal64", Grid{.digits = 16}, mode, samples, rows);
|
||||
runSubject<BoostDecimalFast64>(
|
||||
"BoostDecimalFast64", Grid{.digits = 16}, mode, samples, rows);
|
||||
runSubject<BoostDecimal128>("BoostDecimal128", Grid{.digits = 34}, mode, samples, rows);
|
||||
runSubject<BoostDecimalFast128>(
|
||||
"BoostDecimalFast128", Grid{.digits = 34}, mode, samples, rows);
|
||||
#ifdef XRPL_BENCH_INTEL_DFP
|
||||
runSubject<IntelBid64>("IntelBid64", Grid{.digits = 16}, mode, samples, rows);
|
||||
runSubject<IntelBid128>("IntelBid128", Grid{.digits = 34}, mode, samples, rows);
|
||||
#endif
|
||||
#ifdef XRPL_BENCH_HAS_NUMBER128
|
||||
runSubject<Number128>("Number128", Grid{.digits = Number128::kDigits}, mode, samples, rows);
|
||||
#endif
|
||||
}
|
||||
print(rows, csv);
|
||||
return 0;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace xrpl::number_bench
|
||||
|
||||
int
|
||||
main(int argc, char** argv)
|
||||
{
|
||||
return xrpl::number_bench::run(argc, argv);
|
||||
}
|
||||
Reference in New Issue
Block a user