mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-28 07:48:01 +00:00
Merge branch 'pratik/otel-phase10-workload-validation' into pratik/otel-sync-diagnostics
# Conflicts: # src/xrpld/telemetry/MetricsRegistry.cpp
This commit is contained in:
174
src/tests/libxrpl/beast/insight/Unit.cpp
Normal file
174
src/tests/libxrpl/beast/insight/Unit.cpp
Normal file
@@ -0,0 +1,174 @@
|
||||
/**
|
||||
* GTest unit tests for beast::insight::Unit and its plumbing.
|
||||
*
|
||||
* A metric's unit decides two things that are invisible at the call site: the
|
||||
* name suffix the exporter appends, and which bucket ladder the histogram
|
||||
* view applies. Getting it wrong is silent -- a byte count declared as
|
||||
* milliseconds still records, still exports, still draws a graph, and the
|
||||
* graph is wrong. So each hop the unit has to survive is asserted here
|
||||
* rather than left to inspection.
|
||||
*
|
||||
* The hop that matters most is the group wrapper. Call sites reach a
|
||||
* collector through Groups, so a unit that reaches OTelCollector correctly
|
||||
* but is dropped by the group prefixing layer would pass a naive test while
|
||||
* failing in production.
|
||||
*/
|
||||
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
|
||||
#include <xrpl/beast/insight/Event.h>
|
||||
#include <xrpl/beast/insight/EventImpl.h>
|
||||
#include <xrpl/beast/insight/Groups.h>
|
||||
#include <xrpl/beast/insight/NullCollector.h>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
namespace beast::insight {
|
||||
|
||||
namespace {
|
||||
|
||||
/**
|
||||
* An EventImpl that records what it was notified with.
|
||||
*
|
||||
* Needed because every shipped implementation either discards the sample
|
||||
* (NullCollector) or sends it somewhere external. Asserting the recorded
|
||||
* value proves the raw-integral path preserves it, rather than only proving
|
||||
* that notify() can be called without crashing.
|
||||
*/
|
||||
class RecordingEventImpl : public EventImpl
|
||||
{
|
||||
public:
|
||||
explicit RecordingEventImpl(Unit unit) : EventImpl(unit)
|
||||
{
|
||||
}
|
||||
|
||||
void
|
||||
notify(value_type const& value) override
|
||||
{
|
||||
samples.push_back(value);
|
||||
}
|
||||
|
||||
/**
|
||||
* Every value passed to notify(), in call order.
|
||||
*/
|
||||
std::vector<value_type> samples;
|
||||
};
|
||||
|
||||
} // namespace
|
||||
|
||||
// The unit code is a contract with the collector's Prometheus exporter: it
|
||||
// derives the exported name suffix from this string. Assert the exact codes,
|
||||
// not merely that they differ.
|
||||
TEST(InsightUnit, otelCodeIsTheUcumCodeForEachUnit)
|
||||
{
|
||||
EXPECT_STREQ(otelUnitCode(Unit::Millis), "ms");
|
||||
EXPECT_STREQ(otelUnitCode(Unit::Bytes), "By");
|
||||
}
|
||||
|
||||
// The description is what an operator reads in the metric catalogue, so a
|
||||
// byte-valued instrument must not describe itself as a duration.
|
||||
TEST(InsightUnit, descriptionMatchesWhatTheUnitActuallyMeasures)
|
||||
{
|
||||
EXPECT_STREQ(otelUnitDescription(Unit::Millis), "Duration in ms");
|
||||
EXPECT_STREQ(otelUnitDescription(Unit::Bytes), "Size in bytes");
|
||||
}
|
||||
|
||||
TEST(InsightUnit, defaultEventUnitIsMillisForBackwardCompatibility)
|
||||
{
|
||||
// Every pre-existing makeEvent(name) call site records a duration, so the
|
||||
// one-argument overload must keep meaning milliseconds.
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("legacy");
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, makeEventCarriesTheRequestedUnitToTheImpl)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, prefixedMakeEventCarriesTheUnit)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("rpc", "size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, groupWrapperForwardsTheUnitAlongWithThePrefix)
|
||||
{
|
||||
// ServerHandler creates its events through a Group, not through the
|
||||
// collector directly. If the group's makeEvent override forwards only the
|
||||
// name, the unit silently reverts to milliseconds and the byte histogram
|
||||
// inherits the latency ladder again.
|
||||
auto const collector = NullCollector::make();
|
||||
auto const groups = makeGroups(collector);
|
||||
auto const event = groups->get("rpc")->makeEvent("size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, groupWrapperStillDefaultsToMillis)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const groups = makeGroups(collector);
|
||||
auto const event = groups->get("rpc")->makeEvent("time");
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, rawIntegralNotifyPreservesTheValueExactly)
|
||||
{
|
||||
// The byte path must not be rounded or scaled on its way through the
|
||||
// duration-typed storage field.
|
||||
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Bytes);
|
||||
Event const event(impl);
|
||||
|
||||
event.notify(std::uint64_t{4096});
|
||||
event.notify(std::uint64_t{0});
|
||||
event.notify(std::uint64_t{1'048'577});
|
||||
|
||||
ASSERT_EQ(impl->samples.size(), 3U);
|
||||
EXPECT_EQ(impl->samples[0].count(), 4096);
|
||||
EXPECT_EQ(impl->samples[1].count(), 0);
|
||||
EXPECT_EQ(impl->samples[2].count(), 1'048'577);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, durationNotifyStillRoundsUpToWholeMilliseconds)
|
||||
{
|
||||
// Pre-existing behaviour, asserted so the new overload cannot quietly
|
||||
// change it: Event applies ceil to whole milliseconds, which is why
|
||||
// sub-millisecond resolution is impossible on this path.
|
||||
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Millis);
|
||||
Event const event(impl);
|
||||
|
||||
event.notify(std::chrono::microseconds{40});
|
||||
event.notify(std::chrono::microseconds{1'000});
|
||||
event.notify(std::chrono::milliseconds{7});
|
||||
|
||||
ASSERT_EQ(impl->samples.size(), 3U);
|
||||
EXPECT_EQ(impl->samples[0].count(), 1) << "40us must round up to 1ms, not down to 0";
|
||||
EXPECT_EQ(impl->samples[1].count(), 1);
|
||||
EXPECT_EQ(impl->samples[2].count(), 7);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, notifyOnANullEventIsSafeForBothOverloads)
|
||||
{
|
||||
// A default-constructed Event has no impl. Both overloads must be no-ops
|
||||
// rather than dereferencing null.
|
||||
Event const none;
|
||||
ASSERT_EQ(none.impl(), nullptr);
|
||||
EXPECT_NO_THROW(none.notify(std::uint64_t{4096}));
|
||||
EXPECT_NO_THROW(none.notify(std::chrono::milliseconds{5}));
|
||||
}
|
||||
|
||||
} // namespace beast::insight
|
||||
235
src/tests/libxrpl/telemetry/HistogramBuckets.cpp
Normal file
235
src/tests/libxrpl/telemetry/HistogramBuckets.cpp
Normal file
@@ -0,0 +1,235 @@
|
||||
/**
|
||||
* GTest unit tests for the histogram bucket ladders.
|
||||
*
|
||||
* These ladders decide whether a Grafana percentile panel reports a
|
||||
* measurement or an artefact, and neither failure mode is visible in the
|
||||
* panel itself: a quantile that falls in the `+Inf` bucket reads back as the
|
||||
* second-highest edge, and one that falls inside bucket 0 is interpolated.
|
||||
* Both look like plausible numbers. So the invariants are asserted here
|
||||
* rather than left to review.
|
||||
*
|
||||
* The ladders are `constexpr`, so most of this could be `static_assert`.
|
||||
* They are runtime tests as well so that a failure names which edge is
|
||||
* wrong instead of only failing the compile.
|
||||
*/
|
||||
|
||||
#include <xrpl/telemetry/HistogramBuckets.h>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <span>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::telemetry::buckets {
|
||||
|
||||
// Every ladder must be strictly ascending and non-negative. The SDK places a
|
||||
// sample with std::lower_bound over the edges, so a duplicated or
|
||||
// out-of-order edge silently sends samples to the wrong bucket.
|
||||
class HistogramBucketsTest : public ::testing::TestWithParam<std::span<double const>>
|
||||
{
|
||||
};
|
||||
|
||||
TEST_P(HistogramBucketsTest, isStrictlyAscending)
|
||||
{
|
||||
auto const ladder = GetParam();
|
||||
ASSERT_FALSE(ladder.empty());
|
||||
for (std::size_t i = 1; i < ladder.size(); ++i)
|
||||
EXPECT_LT(ladder[i - 1], ladder[i]) << "edge index " << i << " does not ascend";
|
||||
}
|
||||
|
||||
TEST_P(HistogramBucketsTest, isNonNegativeAndFinite)
|
||||
{
|
||||
for (double const edge : GetParam())
|
||||
{
|
||||
EXPECT_GE(edge, 0.0);
|
||||
EXPECT_TRUE(std::isfinite(edge)) << "edge " << edge << " is not finite";
|
||||
}
|
||||
}
|
||||
|
||||
TEST_P(HistogramBucketsTest, passesTheCompileTimeValidator)
|
||||
{
|
||||
EXPECT_TRUE(isAscendingNonNegative(GetParam()));
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_SUITE_P(
|
||||
AllLadders,
|
||||
HistogramBucketsTest,
|
||||
::testing::Values(
|
||||
std::span<double const>{kMillisecondBuckets},
|
||||
std::span<double const>{kByteBuckets},
|
||||
std::span<double const>{kMicrosecondBuckets},
|
||||
std::span<double const>{kObjectCountBuckets},
|
||||
std::span<double const>{kChargeBuckets}));
|
||||
|
||||
TEST(HistogramBucketsRange, microsecondFloorLandsBelowTheMeasuredMass)
|
||||
{
|
||||
// Measured: 99.3% of job_queued_us samples sat below the old 100 us floor,
|
||||
// so p75/p95/p99 all interpolated inside bucket 0 and returned
|
||||
// 75.5/95.7/99.7 us -- the boundary scaled by the requested quantile,
|
||||
// not a latency. Warm nodestore reads are ~1.5 us, so the floor has to
|
||||
// reach single microseconds and several edges must precede 100 us.
|
||||
EXPECT_LE(kMicrosecondBuckets.front(), 1.0);
|
||||
|
||||
auto const belowHundred =
|
||||
std::ranges::count_if(kMicrosecondBuckets, [](double edge) { return edge < 100.0; });
|
||||
EXPECT_GE(belowHundred, 5) << "too little resolution below 100 us";
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, microsecondCeilingStillReachesOneMinute)
|
||||
{
|
||||
// Job waits and RPC latencies routinely exceed the SDK default ceiling of
|
||||
// 10,000; multi-second stalls must stay measurable rather than censored.
|
||||
EXPECT_EQ(kMicrosecondBuckets.back(), 60'000'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, objectCountLadderCannotSaturate)
|
||||
{
|
||||
// GetObject counts run 1..kHardMaxReplyNodes, so the top edge IS the hard
|
||||
// cap and censoring is impossible by construction.
|
||||
EXPECT_EQ(kObjectCountBuckets.front(), 1.0);
|
||||
EXPECT_EQ(kObjectCountBuckets.back(), 12'288.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, chargeLadderBracketsTheResourceThresholds)
|
||||
{
|
||||
// The two edges that decide a peer's fate must be present so a dashboard
|
||||
// can show how close charges run to each: warning at 5000, drop at 25000.
|
||||
// A leading 0 separates the free tier from everything else.
|
||||
EXPECT_EQ(kChargeBuckets.front(), 0.0);
|
||||
for (double const threshold : {5'000.0, 25'000.0})
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kChargeBuckets, threshold), kChargeBuckets.end())
|
||||
<< threshold << " is a resource threshold and must be an edge";
|
||||
}
|
||||
}
|
||||
|
||||
// The validator must also REJECT. A predicate that only ever returns true
|
||||
// would let every ladder above pass while proving nothing.
|
||||
TEST(HistogramBucketsValidator, rejectsEmptyDescendingDuplicateAndNegative)
|
||||
{
|
||||
EXPECT_FALSE(isAscendingNonNegative(std::span<double const>{}));
|
||||
|
||||
constexpr std::array descending{5.0, 1.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(descending));
|
||||
|
||||
constexpr std::array duplicated{1.0, 1.0, 2.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(duplicated));
|
||||
|
||||
constexpr std::array negative{-1.0, 1.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(negative));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsValidator, acceptsASingleEdgeAndALeadingZero)
|
||||
{
|
||||
constexpr std::array single{1.0};
|
||||
EXPECT_TRUE(isAscendingNonNegative(single));
|
||||
|
||||
// A leading zero is legal: the GetObject charge ladder starts at 0 to
|
||||
// separate the free tier from everything else.
|
||||
constexpr std::array leadingZero{0.0, 100.0};
|
||||
EXPECT_TRUE(isAscendingNonNegative(leadingZero));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondFloorIsOneAndCeilingCoversTheSlowestJob)
|
||||
{
|
||||
// beast::insight::Event rounds durations up to whole milliseconds, so 1
|
||||
// is the smallest edge that can ever collect a sample.
|
||||
EXPECT_EQ(kMillisecondBuckets.front(), 1.0);
|
||||
|
||||
// The updatepaths job type was measured averaging 59,956 ms. A 30 s
|
||||
// ceiling -- the collector's top edge -- would censor it just as the old
|
||||
// 5 s ceiling does, so this ladder has to reach further.
|
||||
EXPECT_GE(kMillisecondBuckets.back(), 120'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderClearsTheMeasuredCensoringPoint)
|
||||
{
|
||||
// rpc_size had 24.9% of samples above the old 5000 ceiling and
|
||||
// jobq_updatepaths had 100%. A ceiling at or below 5000 reintroduces the
|
||||
// exact defect this ladder exists to fix.
|
||||
EXPECT_GT(kMillisecondBuckets.back(), 5'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderContainsEveryRepresentableCollectorEdge)
|
||||
{
|
||||
// Agreement with the collector's spanmetrics ladder over the shared
|
||||
// range is the invariant; edges above its 30 s top are allowed because
|
||||
// jobs outlive spans. Sub-millisecond collector edges are excluded
|
||||
// because Event cannot represent them. check_bucket_parity.py enforces
|
||||
// this against the YAML; this test pins it for the C++ side alone so a
|
||||
// local edit fails fast.
|
||||
constexpr std::array collectorEdges{
|
||||
1.0,
|
||||
5.0,
|
||||
10.0,
|
||||
25.0,
|
||||
50.0,
|
||||
100.0,
|
||||
250.0,
|
||||
500.0,
|
||||
1'000.0,
|
||||
2'000.0,
|
||||
3'000.0,
|
||||
4'000.0,
|
||||
5'000.0,
|
||||
10'000.0,
|
||||
30'000.0};
|
||||
|
||||
for (double const edge : collectorEdges)
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
|
||||
<< edge << " ms is a collector spanmetrics edge and must be present";
|
||||
}
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderResolvesTheOneToFiveSecondBand)
|
||||
{
|
||||
// Without these the 1 s to 5 s span was one four-second-wide bucket, so
|
||||
// any quantile landing inside it was interpolated across four seconds.
|
||||
for (double const edge : {2'000.0, 3'000.0, 4'000.0})
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
|
||||
<< edge << " ms edge missing";
|
||||
}
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, byteLadderBracketsTheMeasuredResponseDistribution)
|
||||
{
|
||||
// Measured: mean 2131 B, half under 1 kB, three quarters under 5 kB, and
|
||||
// the tail above 5 kB has a mean of at most 7538 B -- which puts p99
|
||||
// near 80 kB. The floor must sit at or below the measured median region
|
||||
// and the ceiling well past the p99 bound.
|
||||
EXPECT_LE(kByteBuckets.front(), 512.0);
|
||||
EXPECT_GE(kByteBuckets.back(), 1'048'576.0);
|
||||
|
||||
// Most of the resolution belongs where the distribution actually turns.
|
||||
auto const withinWorkingRange =
|
||||
std::ranges::count_if(kByteBuckets, [](double e) { return e >= 512.0 && e <= 65'536.0; });
|
||||
EXPECT_GE(withinWorkingRange, 6) << "too little resolution between 512 B and 64 kB";
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, byteAndMillisecondLaddersAreDistinct)
|
||||
{
|
||||
// A single shared ladder is what put a byte count on a latency scale and
|
||||
// censored a quarter of its samples.
|
||||
EXPECT_NE(kByteBuckets.size(), kMillisecondBuckets.size());
|
||||
EXPECT_GT(kByteBuckets.back(), kMillisecondBuckets.back());
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsConvert, toVectorPreservesOrderAndSize)
|
||||
{
|
||||
auto const converted = toVector(kByteBuckets);
|
||||
ASSERT_EQ(converted.size(), kByteBuckets.size());
|
||||
EXPECT_TRUE(std::ranges::equal(converted, kByteBuckets));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsConvert, toVectorHandlesAnEmptyLadder)
|
||||
{
|
||||
EXPECT_TRUE(toVector(std::span<double const>{}).empty());
|
||||
}
|
||||
|
||||
} // namespace xrpl::telemetry::buckets
|
||||
Reference in New Issue
Block a user