Merge branch 'pratik/otel-phase10-workload-validation' into pratik/otel-sync-diagnostics

# Conflicts:
#	src/xrpld/telemetry/MetricsRegistry.cpp
This commit is contained in:
Pratik Mankawde
2026-08-21 13:09:09 +01:00
24 changed files with 1190 additions and 123 deletions

View File

@@ -0,0 +1,174 @@
/**
* GTest unit tests for beast::insight::Unit and its plumbing.
*
* A metric's unit decides two things that are invisible at the call site: the
* name suffix the exporter appends, and which bucket ladder the histogram
* view applies. Getting it wrong is silent -- a byte count declared as
* milliseconds still records, still exports, still draws a graph, and the
* graph is wrong. So each hop the unit has to survive is asserted here
* rather than left to inspection.
*
* The hop that matters most is the group wrapper. Call sites reach a
* collector through Groups, so a unit that reaches OTelCollector correctly
* but is dropped by the group prefixing layer would pass a naive test while
* failing in production.
*/
#include <xrpl/beast/insight/Unit.h>
#include <xrpl/beast/insight/Event.h>
#include <xrpl/beast/insight/EventImpl.h>
#include <xrpl/beast/insight/Groups.h>
#include <xrpl/beast/insight/NullCollector.h>
#include <gtest/gtest.h>
#include <chrono>
#include <cstdint>
#include <memory>
#include <vector>
namespace beast::insight {
namespace {
/**
* An EventImpl that records what it was notified with.
*
* Needed because every shipped implementation either discards the sample
* (NullCollector) or sends it somewhere external. Asserting the recorded
* value proves the raw-integral path preserves it, rather than only proving
* that notify() can be called without crashing.
*/
class RecordingEventImpl : public EventImpl
{
public:
explicit RecordingEventImpl(Unit unit) : EventImpl(unit)
{
}
void
notify(value_type const& value) override
{
samples.push_back(value);
}
/**
* Every value passed to notify(), in call order.
*/
std::vector<value_type> samples;
};
} // namespace
// The unit code is a contract with the collector's Prometheus exporter: it
// derives the exported name suffix from this string. Assert the exact codes,
// not merely that they differ.
TEST(InsightUnit, otelCodeIsTheUcumCodeForEachUnit)
{
EXPECT_STREQ(otelUnitCode(Unit::Millis), "ms");
EXPECT_STREQ(otelUnitCode(Unit::Bytes), "By");
}
// The description is what an operator reads in the metric catalogue, so a
// byte-valued instrument must not describe itself as a duration.
TEST(InsightUnit, descriptionMatchesWhatTheUnitActuallyMeasures)
{
EXPECT_STREQ(otelUnitDescription(Unit::Millis), "Duration in ms");
EXPECT_STREQ(otelUnitDescription(Unit::Bytes), "Size in bytes");
}
TEST(InsightUnit, defaultEventUnitIsMillisForBackwardCompatibility)
{
// Every pre-existing makeEvent(name) call site records a duration, so the
// one-argument overload must keep meaning milliseconds.
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("legacy");
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
}
TEST(InsightUnit, makeEventCarriesTheRequestedUnitToTheImpl)
{
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, prefixedMakeEventCarriesTheUnit)
{
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("rpc", "size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, groupWrapperForwardsTheUnitAlongWithThePrefix)
{
// ServerHandler creates its events through a Group, not through the
// collector directly. If the group's makeEvent override forwards only the
// name, the unit silently reverts to milliseconds and the byte histogram
// inherits the latency ladder again.
auto const collector = NullCollector::make();
auto const groups = makeGroups(collector);
auto const event = groups->get("rpc")->makeEvent("size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, groupWrapperStillDefaultsToMillis)
{
auto const collector = NullCollector::make();
auto const groups = makeGroups(collector);
auto const event = groups->get("rpc")->makeEvent("time");
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
}
TEST(InsightUnit, rawIntegralNotifyPreservesTheValueExactly)
{
// The byte path must not be rounded or scaled on its way through the
// duration-typed storage field.
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Bytes);
Event const event(impl);
event.notify(std::uint64_t{4096});
event.notify(std::uint64_t{0});
event.notify(std::uint64_t{1'048'577});
ASSERT_EQ(impl->samples.size(), 3U);
EXPECT_EQ(impl->samples[0].count(), 4096);
EXPECT_EQ(impl->samples[1].count(), 0);
EXPECT_EQ(impl->samples[2].count(), 1'048'577);
}
TEST(InsightUnit, durationNotifyStillRoundsUpToWholeMilliseconds)
{
// Pre-existing behaviour, asserted so the new overload cannot quietly
// change it: Event applies ceil to whole milliseconds, which is why
// sub-millisecond resolution is impossible on this path.
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Millis);
Event const event(impl);
event.notify(std::chrono::microseconds{40});
event.notify(std::chrono::microseconds{1'000});
event.notify(std::chrono::milliseconds{7});
ASSERT_EQ(impl->samples.size(), 3U);
EXPECT_EQ(impl->samples[0].count(), 1) << "40us must round up to 1ms, not down to 0";
EXPECT_EQ(impl->samples[1].count(), 1);
EXPECT_EQ(impl->samples[2].count(), 7);
}
TEST(InsightUnit, notifyOnANullEventIsSafeForBothOverloads)
{
// A default-constructed Event has no impl. Both overloads must be no-ops
// rather than dereferencing null.
Event const none;
ASSERT_EQ(none.impl(), nullptr);
EXPECT_NO_THROW(none.notify(std::uint64_t{4096}));
EXPECT_NO_THROW(none.notify(std::chrono::milliseconds{5}));
}
} // namespace beast::insight

View File

@@ -0,0 +1,235 @@
/**
* GTest unit tests for the histogram bucket ladders.
*
* These ladders decide whether a Grafana percentile panel reports a
* measurement or an artefact, and neither failure mode is visible in the
* panel itself: a quantile that falls in the `+Inf` bucket reads back as the
* second-highest edge, and one that falls inside bucket 0 is interpolated.
* Both look like plausible numbers. So the invariants are asserted here
* rather than left to review.
*
* The ladders are `constexpr`, so most of this could be `static_assert`.
* They are runtime tests as well so that a failure names which edge is
* wrong instead of only failing the compile.
*/
#include <xrpl/telemetry/HistogramBuckets.h>
#include <gtest/gtest.h>
#include <algorithm>
#include <array>
#include <cmath>
#include <cstddef>
#include <span>
#include <vector>
namespace xrpl::telemetry::buckets {
// Every ladder must be strictly ascending and non-negative. The SDK places a
// sample with std::lower_bound over the edges, so a duplicated or
// out-of-order edge silently sends samples to the wrong bucket.
class HistogramBucketsTest : public ::testing::TestWithParam<std::span<double const>>
{
};
TEST_P(HistogramBucketsTest, isStrictlyAscending)
{
auto const ladder = GetParam();
ASSERT_FALSE(ladder.empty());
for (std::size_t i = 1; i < ladder.size(); ++i)
EXPECT_LT(ladder[i - 1], ladder[i]) << "edge index " << i << " does not ascend";
}
TEST_P(HistogramBucketsTest, isNonNegativeAndFinite)
{
for (double const edge : GetParam())
{
EXPECT_GE(edge, 0.0);
EXPECT_TRUE(std::isfinite(edge)) << "edge " << edge << " is not finite";
}
}
TEST_P(HistogramBucketsTest, passesTheCompileTimeValidator)
{
EXPECT_TRUE(isAscendingNonNegative(GetParam()));
}
INSTANTIATE_TEST_SUITE_P(
AllLadders,
HistogramBucketsTest,
::testing::Values(
std::span<double const>{kMillisecondBuckets},
std::span<double const>{kByteBuckets},
std::span<double const>{kMicrosecondBuckets},
std::span<double const>{kObjectCountBuckets},
std::span<double const>{kChargeBuckets}));
TEST(HistogramBucketsRange, microsecondFloorLandsBelowTheMeasuredMass)
{
// Measured: 99.3% of job_queued_us samples sat below the old 100 us floor,
// so p75/p95/p99 all interpolated inside bucket 0 and returned
// 75.5/95.7/99.7 us -- the boundary scaled by the requested quantile,
// not a latency. Warm nodestore reads are ~1.5 us, so the floor has to
// reach single microseconds and several edges must precede 100 us.
EXPECT_LE(kMicrosecondBuckets.front(), 1.0);
auto const belowHundred =
std::ranges::count_if(kMicrosecondBuckets, [](double edge) { return edge < 100.0; });
EXPECT_GE(belowHundred, 5) << "too little resolution below 100 us";
}
TEST(HistogramBucketsRange, microsecondCeilingStillReachesOneMinute)
{
// Job waits and RPC latencies routinely exceed the SDK default ceiling of
// 10,000; multi-second stalls must stay measurable rather than censored.
EXPECT_EQ(kMicrosecondBuckets.back(), 60'000'000.0);
}
TEST(HistogramBucketsRange, objectCountLadderCannotSaturate)
{
// GetObject counts run 1..kHardMaxReplyNodes, so the top edge IS the hard
// cap and censoring is impossible by construction.
EXPECT_EQ(kObjectCountBuckets.front(), 1.0);
EXPECT_EQ(kObjectCountBuckets.back(), 12'288.0);
}
TEST(HistogramBucketsRange, chargeLadderBracketsTheResourceThresholds)
{
// The two edges that decide a peer's fate must be present so a dashboard
// can show how close charges run to each: warning at 5000, drop at 25000.
// A leading 0 separates the free tier from everything else.
EXPECT_EQ(kChargeBuckets.front(), 0.0);
for (double const threshold : {5'000.0, 25'000.0})
{
EXPECT_NE(std::ranges::find(kChargeBuckets, threshold), kChargeBuckets.end())
<< threshold << " is a resource threshold and must be an edge";
}
}
// The validator must also REJECT. A predicate that only ever returns true
// would let every ladder above pass while proving nothing.
TEST(HistogramBucketsValidator, rejectsEmptyDescendingDuplicateAndNegative)
{
EXPECT_FALSE(isAscendingNonNegative(std::span<double const>{}));
constexpr std::array descending{5.0, 1.0};
EXPECT_FALSE(isAscendingNonNegative(descending));
constexpr std::array duplicated{1.0, 1.0, 2.0};
EXPECT_FALSE(isAscendingNonNegative(duplicated));
constexpr std::array negative{-1.0, 1.0};
EXPECT_FALSE(isAscendingNonNegative(negative));
}
TEST(HistogramBucketsValidator, acceptsASingleEdgeAndALeadingZero)
{
constexpr std::array single{1.0};
EXPECT_TRUE(isAscendingNonNegative(single));
// A leading zero is legal: the GetObject charge ladder starts at 0 to
// separate the free tier from everything else.
constexpr std::array leadingZero{0.0, 100.0};
EXPECT_TRUE(isAscendingNonNegative(leadingZero));
}
TEST(HistogramBucketsRange, millisecondFloorIsOneAndCeilingCoversTheSlowestJob)
{
// beast::insight::Event rounds durations up to whole milliseconds, so 1
// is the smallest edge that can ever collect a sample.
EXPECT_EQ(kMillisecondBuckets.front(), 1.0);
// The updatepaths job type was measured averaging 59,956 ms. A 30 s
// ceiling -- the collector's top edge -- would censor it just as the old
// 5 s ceiling does, so this ladder has to reach further.
EXPECT_GE(kMillisecondBuckets.back(), 120'000.0);
}
TEST(HistogramBucketsRange, millisecondLadderClearsTheMeasuredCensoringPoint)
{
// rpc_size had 24.9% of samples above the old 5000 ceiling and
// jobq_updatepaths had 100%. A ceiling at or below 5000 reintroduces the
// exact defect this ladder exists to fix.
EXPECT_GT(kMillisecondBuckets.back(), 5'000.0);
}
TEST(HistogramBucketsRange, millisecondLadderContainsEveryRepresentableCollectorEdge)
{
// Agreement with the collector's spanmetrics ladder over the shared
// range is the invariant; edges above its 30 s top are allowed because
// jobs outlive spans. Sub-millisecond collector edges are excluded
// because Event cannot represent them. check_bucket_parity.py enforces
// this against the YAML; this test pins it for the C++ side alone so a
// local edit fails fast.
constexpr std::array collectorEdges{
1.0,
5.0,
10.0,
25.0,
50.0,
100.0,
250.0,
500.0,
1'000.0,
2'000.0,
3'000.0,
4'000.0,
5'000.0,
10'000.0,
30'000.0};
for (double const edge : collectorEdges)
{
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
<< edge << " ms is a collector spanmetrics edge and must be present";
}
}
TEST(HistogramBucketsRange, millisecondLadderResolvesTheOneToFiveSecondBand)
{
// Without these the 1 s to 5 s span was one four-second-wide bucket, so
// any quantile landing inside it was interpolated across four seconds.
for (double const edge : {2'000.0, 3'000.0, 4'000.0})
{
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
<< edge << " ms edge missing";
}
}
TEST(HistogramBucketsRange, byteLadderBracketsTheMeasuredResponseDistribution)
{
// Measured: mean 2131 B, half under 1 kB, three quarters under 5 kB, and
// the tail above 5 kB has a mean of at most 7538 B -- which puts p99
// near 80 kB. The floor must sit at or below the measured median region
// and the ceiling well past the p99 bound.
EXPECT_LE(kByteBuckets.front(), 512.0);
EXPECT_GE(kByteBuckets.back(), 1'048'576.0);
// Most of the resolution belongs where the distribution actually turns.
auto const withinWorkingRange =
std::ranges::count_if(kByteBuckets, [](double e) { return e >= 512.0 && e <= 65'536.0; });
EXPECT_GE(withinWorkingRange, 6) << "too little resolution between 512 B and 64 kB";
}
TEST(HistogramBucketsRange, byteAndMillisecondLaddersAreDistinct)
{
// A single shared ladder is what put a byte count on a latency scale and
// censored a quarter of its samples.
EXPECT_NE(kByteBuckets.size(), kMillisecondBuckets.size());
EXPECT_GT(kByteBuckets.back(), kMillisecondBuckets.back());
}
TEST(HistogramBucketsConvert, toVectorPreservesOrderAndSize)
{
auto const converted = toVector(kByteBuckets);
ASSERT_EQ(converted.size(), kByteBuckets.size());
EXPECT_TRUE(std::ranges::equal(converted, kByteBuckets));
}
TEST(HistogramBucketsConvert, toVectorHandlesAnEmptyLadder)
{
EXPECT_TRUE(toVector(std::span<double const>{}).empty());
}
} // namespace xrpl::telemetry::buckets