Merge branch 'pratik/otel-phase10-workload-validation' into pratik/otel-sync-diagnostics

# Conflicts:
#	src/xrpld/telemetry/MetricsRegistry.cpp
This commit is contained in:
Pratik Mankawde
2026-08-21 13:09:09 +01:00
24 changed files with 1190 additions and 123 deletions

View File

@@ -9,6 +9,7 @@
#include <xrpl/beast/insight/Hook.h>
#include <xrpl/beast/insight/HookImpl.h>
#include <xrpl/beast/insight/Meter.h>
#include <xrpl/beast/insight/Unit.h>
#include <memory>
#include <string>
@@ -56,12 +57,25 @@ public:
return collector_->makeCounter(makeName(name));
}
using Collector::makeEvent;
Event
makeEvent(std::string const& name) override
{
return collector_->makeEvent(makeName(name));
}
// Forwards the unit as well as the prefixed name. Without this override
// the base-class default would delegate to the single-argument overload
// above and silently drop the unit, which is how a byte-valued Event ends
// up declared as milliseconds -- call sites reach a collector through a
// Group, so this is the hop that actually matters.
Event
makeEvent(std::string const& name, Unit unit) override
{
return collector_->makeEvent(makeName(name), unit);
}
Gauge
makeGauge(std::string const& name) override
{

View File

@@ -11,6 +11,7 @@
#include <xrpl/beast/insight/HookImpl.h>
#include <xrpl/beast/insight/Meter.h>
#include <xrpl/beast/insight/MeterImpl.h>
#include <xrpl/beast/insight/Unit.h>
#include <memory>
#include <string>
@@ -49,7 +50,15 @@ public:
class NullEventImpl : public EventImpl
{
public:
explicit NullEventImpl() = default;
/**
* @param unit What the samples would measure. Recorded even though
* nothing is collected, so a caller can still read back the
* unit it asked for -- which is what makes the null collector
* usable for testing the unit plumbing.
*/
explicit NullEventImpl(Unit unit = Unit::Millis) : EventImpl(unit)
{
}
void
notify(value_type const&) override
@@ -119,12 +128,20 @@ public:
return Counter(std::make_shared<detail::NullCounterImpl>());
}
using Collector::makeEvent;
Event
makeEvent(std::string const&) override
{
return Event(std::make_shared<detail::NullEventImpl>());
}
Event
makeEvent(std::string const&, Unit unit) override
{
return Event(std::make_shared<detail::NullEventImpl>(unit));
}
Gauge
makeGauge(std::string const&) override
{

View File

@@ -41,6 +41,7 @@
#include <xrpl/beast/insight/Hook.h>
#include <xrpl/beast/insight/HookImpl.h>
#include <xrpl/beast/insight/MeterImpl.h>
#include <xrpl/beast/insight/Unit.h>
#include <xrpl/beast/utility/Journal.h>
#include <opentelemetry/metrics/async_instruments.h>
@@ -168,10 +169,17 @@ private:
/**
* @brief OTel-backed implementation of beast::insight::EventImpl.
*
* Wraps an OTel Histogram<double> instrument. Each notify() call
* records the duration in milliseconds. Uses explicit bucket boundaries
* matching the SpanMetrics connector configuration:
* [1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000] ms
* Wraps an OTel Histogram<double> instrument. Each notify() call records one
* sample, interpreted per the Event's unit().
*
* The instrument's declared unit is what selects its bucket ladder: the
* histogram views registered in Telemetry.cpp match on unit, so a `ms`
* instrument gets the millisecond ladder and a `By` instrument the byte
* ladder. The edges themselves live in xrpl/telemetry/HistogramBuckets.h --
* do not restate them here. An earlier version of this comment listed
* `[1, 5, ..., 1000, 5000] ms` as "matching the SpanMetrics connector"; that
* was true when written and silently became false when the connector's
* ladder was extended, which is why the edges now have one owner.
*
* Thread safety: OTel Histogram::Record() is thread-safe by specification.
*/
@@ -183,10 +191,14 @@ public:
* formatName() by the collector: lowercase, with `.` and
* ` ` mapped to `_` (e.g. "rpc_size").
* @param meter OTel Meter used to create the histogram instrument.
* @param unit What the samples measure. Selects the instrument's
* declared unit, its description, and through the unit the
* bucket ladder a histogram view applies.
*/
OTelEventImpl(
std::string const& name,
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter);
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter,
Unit unit);
~OTelEventImpl() override = default;
@@ -474,6 +486,9 @@ public:
Event
makeEvent(std::string const& name) override;
Event
makeEvent(std::string const& name, Unit unit) override;
Gauge
makeGauge(std::string const& name) override;
@@ -652,8 +667,10 @@ OTelCounterImpl::increment(value_type amount)
OTelEventImpl::OTelEventImpl(
std::string const& name,
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter)
: histogram_(meter->CreateDoubleHistogram(name, "Duration in ms", "ms"))
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter,
Unit unit)
: EventImpl(unit)
, histogram_(meter->CreateDoubleHistogram(name, otelUnitDescription(unit), otelUnitCode(unit)))
{
}
@@ -842,7 +859,13 @@ OTelCollectorImp::makeCounter(std::string const& name)
Event
OTelCollectorImp::makeEvent(std::string const& name)
{
return Event(std::make_shared<OTelEventImpl>(formatName(name), otelMeter_));
return makeEvent(name, Unit::Millis);
}
Event
OTelCollectorImp::makeEvent(std::string const& name, Unit unit)
{
return Event(std::make_shared<OTelEventImpl>(formatName(name), otelMeter_, unit));
}
Gauge

View File

@@ -19,10 +19,12 @@
#include <xrpl/telemetry/Telemetry.h>
#include <xrpl/basics/Log.h>
#include <xrpl/beast/insight/Unit.h>
#include <xrpl/beast/utility/Journal.h>
#include <xrpl/telemetry/CoroAwareContextStorage.h>
#include <xrpl/telemetry/DeterministicIdGenerator.h>
#include <xrpl/telemetry/DiscardFlag.h>
#include <xrpl/telemetry/HistogramBuckets.h>
#include <xrpl/telemetry/SpanNames.h>
#include <opentelemetry/context/context.h>
@@ -553,31 +555,46 @@ public:
std::make_unique<metrics_sdk::ViewRegistry>(), resourceAttrs);
meterProvider_->AddMetricReader(std::move(reader));
// Histogram view: SpanMetrics-compatible bucket boundaries (ms) so
// histogram instruments align with the collector's SpanMetrics. The
// view is created with an EMPTY name so it applies the buckets WITHOUT
// renaming instruments — a non-empty view name would collapse every
// matching histogram (ios_latency, rpc_size, rpc_time, pathfind_*)
// into a single series under that one name.
auto histogramSelector = metrics_sdk::InstrumentSelectorFactory::Create(
metrics_sdk::InstrumentType::kHistogram, "*", "ms");
// Meter selector MUST match the meter name used by getMeter() and the
// beast OTelCollector (kMeterName = "xrpld"); otherwise this histogram
// view never applies and duration histograms fall back to the SDK
// default boundaries instead of these SpanMetrics-aligned buckets.
auto meterSelector =
metrics_sdk::MeterSelectorFactory::Create(std::string(kMeterName), "", "");
auto histogramConfig = std::make_shared<metrics_sdk::HistogramAggregationConfig>();
histogramConfig->boundaries_ =
std::vector<double>{1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0, 5000.0};
auto histogramView = metrics_sdk::ViewFactory::Create(
"", // empty name: keep each instrument's own name, only set buckets
"SpanMetrics-compatible histogram buckets",
metrics_sdk::AggregationType::kHistogram,
histogramConfig);
// One histogram view per unit. The unit is the selector, so an
// instrument gets the ladder that fits what it measures -- a byte
// count no longer inherits a latency ladder. Edges come from
// HistogramBuckets.h, which owns every ladder.
//
// Each view keeps the "*" name pattern and an EMPTY view name: a
// non-empty view name would rename every matching histogram to it and
// collapse them (ios_latency, rpc_size, rpc_time, pathfind_*, and all
// the jobq_* pairs) into a single series.
//
// The meter selector MUST match the meter name used by getMeter() and
// the beast OTelCollector (kMeterName = "xrpld"); otherwise a view
// never applies and instruments fall back to the SDK default ladder,
// whose ceiling is 10,000.
auto const addUnitView = [this](
std::string const& unitCode,
std::vector<double> boundaries,
std::string const& description) {
auto selector = metrics_sdk::InstrumentSelectorFactory::Create(
metrics_sdk::InstrumentType::kHistogram, "*", unitCode);
auto meterSelector =
metrics_sdk::MeterSelectorFactory::Create(std::string(kMeterName), "", "");
auto config = std::make_shared<metrics_sdk::HistogramAggregationConfig>();
config->boundaries_ = std::move(boundaries);
auto view = metrics_sdk::ViewFactory::Create(
"", // empty name: keep each instrument's own name, only set buckets
description,
metrics_sdk::AggregationType::kHistogram,
std::move(config));
meterProvider_->AddView(std::move(selector), std::move(meterSelector), std::move(view));
};
meterProvider_->AddView(
std::move(histogramSelector), std::move(meterSelector), std::move(histogramView));
addUnitView(
beast::insight::otelUnitCode(beast::insight::Unit::Millis),
buckets::toVector(buckets::kMillisecondBuckets),
"Duration buckets, 1 ms to 120 s");
addUnitView(
beast::insight::otelUnitCode(beast::insight::Unit::Bytes),
buckets::toVector(buckets::kByteBuckets),
"Size buckets, 512 B to 1 MiB");
// Publish as the global meter provider so developers (and the beast
// OTelCollector shim) reach the same pipeline.

View File

@@ -0,0 +1,174 @@
/**
* GTest unit tests for beast::insight::Unit and its plumbing.
*
* A metric's unit decides two things that are invisible at the call site: the
* name suffix the exporter appends, and which bucket ladder the histogram
* view applies. Getting it wrong is silent -- a byte count declared as
* milliseconds still records, still exports, still draws a graph, and the
* graph is wrong. So each hop the unit has to survive is asserted here
* rather than left to inspection.
*
* The hop that matters most is the group wrapper. Call sites reach a
* collector through Groups, so a unit that reaches OTelCollector correctly
* but is dropped by the group prefixing layer would pass a naive test while
* failing in production.
*/
#include <xrpl/beast/insight/Unit.h>
#include <xrpl/beast/insight/Event.h>
#include <xrpl/beast/insight/EventImpl.h>
#include <xrpl/beast/insight/Groups.h>
#include <xrpl/beast/insight/NullCollector.h>
#include <gtest/gtest.h>
#include <chrono>
#include <cstdint>
#include <memory>
#include <vector>
namespace beast::insight {
namespace {
/**
* An EventImpl that records what it was notified with.
*
* Needed because every shipped implementation either discards the sample
* (NullCollector) or sends it somewhere external. Asserting the recorded
* value proves the raw-integral path preserves it, rather than only proving
* that notify() can be called without crashing.
*/
class RecordingEventImpl : public EventImpl
{
public:
explicit RecordingEventImpl(Unit unit) : EventImpl(unit)
{
}
void
notify(value_type const& value) override
{
samples.push_back(value);
}
/**
* Every value passed to notify(), in call order.
*/
std::vector<value_type> samples;
};
} // namespace
// The unit code is a contract with the collector's Prometheus exporter: it
// derives the exported name suffix from this string. Assert the exact codes,
// not merely that they differ.
TEST(InsightUnit, otelCodeIsTheUcumCodeForEachUnit)
{
EXPECT_STREQ(otelUnitCode(Unit::Millis), "ms");
EXPECT_STREQ(otelUnitCode(Unit::Bytes), "By");
}
// The description is what an operator reads in the metric catalogue, so a
// byte-valued instrument must not describe itself as a duration.
TEST(InsightUnit, descriptionMatchesWhatTheUnitActuallyMeasures)
{
EXPECT_STREQ(otelUnitDescription(Unit::Millis), "Duration in ms");
EXPECT_STREQ(otelUnitDescription(Unit::Bytes), "Size in bytes");
}
TEST(InsightUnit, defaultEventUnitIsMillisForBackwardCompatibility)
{
// Every pre-existing makeEvent(name) call site records a duration, so the
// one-argument overload must keep meaning milliseconds.
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("legacy");
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
}
TEST(InsightUnit, makeEventCarriesTheRequestedUnitToTheImpl)
{
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, prefixedMakeEventCarriesTheUnit)
{
auto const collector = NullCollector::make();
auto const event = collector->makeEvent("rpc", "size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, groupWrapperForwardsTheUnitAlongWithThePrefix)
{
// ServerHandler creates its events through a Group, not through the
// collector directly. If the group's makeEvent override forwards only the
// name, the unit silently reverts to milliseconds and the byte histogram
// inherits the latency ladder again.
auto const collector = NullCollector::make();
auto const groups = makeGroups(collector);
auto const event = groups->get("rpc")->makeEvent("size", Unit::Bytes);
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
}
TEST(InsightUnit, groupWrapperStillDefaultsToMillis)
{
auto const collector = NullCollector::make();
auto const groups = makeGroups(collector);
auto const event = groups->get("rpc")->makeEvent("time");
ASSERT_NE(event.impl(), nullptr);
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
}
TEST(InsightUnit, rawIntegralNotifyPreservesTheValueExactly)
{
// The byte path must not be rounded or scaled on its way through the
// duration-typed storage field.
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Bytes);
Event const event(impl);
event.notify(std::uint64_t{4096});
event.notify(std::uint64_t{0});
event.notify(std::uint64_t{1'048'577});
ASSERT_EQ(impl->samples.size(), 3U);
EXPECT_EQ(impl->samples[0].count(), 4096);
EXPECT_EQ(impl->samples[1].count(), 0);
EXPECT_EQ(impl->samples[2].count(), 1'048'577);
}
TEST(InsightUnit, durationNotifyStillRoundsUpToWholeMilliseconds)
{
// Pre-existing behaviour, asserted so the new overload cannot quietly
// change it: Event applies ceil to whole milliseconds, which is why
// sub-millisecond resolution is impossible on this path.
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Millis);
Event const event(impl);
event.notify(std::chrono::microseconds{40});
event.notify(std::chrono::microseconds{1'000});
event.notify(std::chrono::milliseconds{7});
ASSERT_EQ(impl->samples.size(), 3U);
EXPECT_EQ(impl->samples[0].count(), 1) << "40us must round up to 1ms, not down to 0";
EXPECT_EQ(impl->samples[1].count(), 1);
EXPECT_EQ(impl->samples[2].count(), 7);
}
TEST(InsightUnit, notifyOnANullEventIsSafeForBothOverloads)
{
// A default-constructed Event has no impl. Both overloads must be no-ops
// rather than dereferencing null.
Event const none;
ASSERT_EQ(none.impl(), nullptr);
EXPECT_NO_THROW(none.notify(std::uint64_t{4096}));
EXPECT_NO_THROW(none.notify(std::chrono::milliseconds{5}));
}
} // namespace beast::insight

View File

@@ -0,0 +1,235 @@
/**
* GTest unit tests for the histogram bucket ladders.
*
* These ladders decide whether a Grafana percentile panel reports a
* measurement or an artefact, and neither failure mode is visible in the
* panel itself: a quantile that falls in the `+Inf` bucket reads back as the
* second-highest edge, and one that falls inside bucket 0 is interpolated.
* Both look like plausible numbers. So the invariants are asserted here
* rather than left to review.
*
* The ladders are `constexpr`, so most of this could be `static_assert`.
* They are runtime tests as well so that a failure names which edge is
* wrong instead of only failing the compile.
*/
#include <xrpl/telemetry/HistogramBuckets.h>
#include <gtest/gtest.h>
#include <algorithm>
#include <array>
#include <cmath>
#include <cstddef>
#include <span>
#include <vector>
namespace xrpl::telemetry::buckets {
// Every ladder must be strictly ascending and non-negative. The SDK places a
// sample with std::lower_bound over the edges, so a duplicated or
// out-of-order edge silently sends samples to the wrong bucket.
class HistogramBucketsTest : public ::testing::TestWithParam<std::span<double const>>
{
};
TEST_P(HistogramBucketsTest, isStrictlyAscending)
{
auto const ladder = GetParam();
ASSERT_FALSE(ladder.empty());
for (std::size_t i = 1; i < ladder.size(); ++i)
EXPECT_LT(ladder[i - 1], ladder[i]) << "edge index " << i << " does not ascend";
}
TEST_P(HistogramBucketsTest, isNonNegativeAndFinite)
{
for (double const edge : GetParam())
{
EXPECT_GE(edge, 0.0);
EXPECT_TRUE(std::isfinite(edge)) << "edge " << edge << " is not finite";
}
}
TEST_P(HistogramBucketsTest, passesTheCompileTimeValidator)
{
EXPECT_TRUE(isAscendingNonNegative(GetParam()));
}
INSTANTIATE_TEST_SUITE_P(
AllLadders,
HistogramBucketsTest,
::testing::Values(
std::span<double const>{kMillisecondBuckets},
std::span<double const>{kByteBuckets},
std::span<double const>{kMicrosecondBuckets},
std::span<double const>{kObjectCountBuckets},
std::span<double const>{kChargeBuckets}));
TEST(HistogramBucketsRange, microsecondFloorLandsBelowTheMeasuredMass)
{
// Measured: 99.3% of job_queued_us samples sat below the old 100 us floor,
// so p75/p95/p99 all interpolated inside bucket 0 and returned
// 75.5/95.7/99.7 us -- the boundary scaled by the requested quantile,
// not a latency. Warm nodestore reads are ~1.5 us, so the floor has to
// reach single microseconds and several edges must precede 100 us.
EXPECT_LE(kMicrosecondBuckets.front(), 1.0);
auto const belowHundred =
std::ranges::count_if(kMicrosecondBuckets, [](double edge) { return edge < 100.0; });
EXPECT_GE(belowHundred, 5) << "too little resolution below 100 us";
}
TEST(HistogramBucketsRange, microsecondCeilingStillReachesOneMinute)
{
// Job waits and RPC latencies routinely exceed the SDK default ceiling of
// 10,000; multi-second stalls must stay measurable rather than censored.
EXPECT_EQ(kMicrosecondBuckets.back(), 60'000'000.0);
}
TEST(HistogramBucketsRange, objectCountLadderCannotSaturate)
{
// GetObject counts run 1..kHardMaxReplyNodes, so the top edge IS the hard
// cap and censoring is impossible by construction.
EXPECT_EQ(kObjectCountBuckets.front(), 1.0);
EXPECT_EQ(kObjectCountBuckets.back(), 12'288.0);
}
TEST(HistogramBucketsRange, chargeLadderBracketsTheResourceThresholds)
{
// The two edges that decide a peer's fate must be present so a dashboard
// can show how close charges run to each: warning at 5000, drop at 25000.
// A leading 0 separates the free tier from everything else.
EXPECT_EQ(kChargeBuckets.front(), 0.0);
for (double const threshold : {5'000.0, 25'000.0})
{
EXPECT_NE(std::ranges::find(kChargeBuckets, threshold), kChargeBuckets.end())
<< threshold << " is a resource threshold and must be an edge";
}
}
// The validator must also REJECT. A predicate that only ever returns true
// would let every ladder above pass while proving nothing.
TEST(HistogramBucketsValidator, rejectsEmptyDescendingDuplicateAndNegative)
{
EXPECT_FALSE(isAscendingNonNegative(std::span<double const>{}));
constexpr std::array descending{5.0, 1.0};
EXPECT_FALSE(isAscendingNonNegative(descending));
constexpr std::array duplicated{1.0, 1.0, 2.0};
EXPECT_FALSE(isAscendingNonNegative(duplicated));
constexpr std::array negative{-1.0, 1.0};
EXPECT_FALSE(isAscendingNonNegative(negative));
}
TEST(HistogramBucketsValidator, acceptsASingleEdgeAndALeadingZero)
{
constexpr std::array single{1.0};
EXPECT_TRUE(isAscendingNonNegative(single));
// A leading zero is legal: the GetObject charge ladder starts at 0 to
// separate the free tier from everything else.
constexpr std::array leadingZero{0.0, 100.0};
EXPECT_TRUE(isAscendingNonNegative(leadingZero));
}
TEST(HistogramBucketsRange, millisecondFloorIsOneAndCeilingCoversTheSlowestJob)
{
// beast::insight::Event rounds durations up to whole milliseconds, so 1
// is the smallest edge that can ever collect a sample.
EXPECT_EQ(kMillisecondBuckets.front(), 1.0);
// The updatepaths job type was measured averaging 59,956 ms. A 30 s
// ceiling -- the collector's top edge -- would censor it just as the old
// 5 s ceiling does, so this ladder has to reach further.
EXPECT_GE(kMillisecondBuckets.back(), 120'000.0);
}
TEST(HistogramBucketsRange, millisecondLadderClearsTheMeasuredCensoringPoint)
{
// rpc_size had 24.9% of samples above the old 5000 ceiling and
// jobq_updatepaths had 100%. A ceiling at or below 5000 reintroduces the
// exact defect this ladder exists to fix.
EXPECT_GT(kMillisecondBuckets.back(), 5'000.0);
}
TEST(HistogramBucketsRange, millisecondLadderContainsEveryRepresentableCollectorEdge)
{
// Agreement with the collector's spanmetrics ladder over the shared
// range is the invariant; edges above its 30 s top are allowed because
// jobs outlive spans. Sub-millisecond collector edges are excluded
// because Event cannot represent them. check_bucket_parity.py enforces
// this against the YAML; this test pins it for the C++ side alone so a
// local edit fails fast.
constexpr std::array collectorEdges{
1.0,
5.0,
10.0,
25.0,
50.0,
100.0,
250.0,
500.0,
1'000.0,
2'000.0,
3'000.0,
4'000.0,
5'000.0,
10'000.0,
30'000.0};
for (double const edge : collectorEdges)
{
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
<< edge << " ms is a collector spanmetrics edge and must be present";
}
}
TEST(HistogramBucketsRange, millisecondLadderResolvesTheOneToFiveSecondBand)
{
// Without these the 1 s to 5 s span was one four-second-wide bucket, so
// any quantile landing inside it was interpolated across four seconds.
for (double const edge : {2'000.0, 3'000.0, 4'000.0})
{
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
<< edge << " ms edge missing";
}
}
TEST(HistogramBucketsRange, byteLadderBracketsTheMeasuredResponseDistribution)
{
// Measured: mean 2131 B, half under 1 kB, three quarters under 5 kB, and
// the tail above 5 kB has a mean of at most 7538 B -- which puts p99
// near 80 kB. The floor must sit at or below the measured median region
// and the ceiling well past the p99 bound.
EXPECT_LE(kByteBuckets.front(), 512.0);
EXPECT_GE(kByteBuckets.back(), 1'048'576.0);
// Most of the resolution belongs where the distribution actually turns.
auto const withinWorkingRange =
std::ranges::count_if(kByteBuckets, [](double e) { return e >= 512.0 && e <= 65'536.0; });
EXPECT_GE(withinWorkingRange, 6) << "too little resolution between 512 B and 64 kB";
}
TEST(HistogramBucketsRange, byteAndMillisecondLaddersAreDistinct)
{
// A single shared ladder is what put a byte count on a latency scale and
// censored a quarter of its samples.
EXPECT_NE(kByteBuckets.size(), kMillisecondBuckets.size());
EXPECT_GT(kByteBuckets.back(), kMillisecondBuckets.back());
}
TEST(HistogramBucketsConvert, toVectorPreservesOrderAndSize)
{
auto const converted = toVector(kByteBuckets);
ASSERT_EQ(converted.size(), kByteBuckets.size());
EXPECT_TRUE(std::ranges::equal(converted, kByteBuckets));
}
TEST(HistogramBucketsConvert, toVectorHandlesAnEmptyLadder)
{
EXPECT_TRUE(toVector(std::span<double const>{}).empty());
}
} // namespace xrpl::telemetry::buckets

View File

@@ -16,6 +16,7 @@
#include <xrpl/basics/base64.h>
#include <xrpl/basics/contract.h>
#include <xrpl/basics/make_SSLContext.h>
#include <xrpl/beast/insight/Unit.h>
#include <xrpl/beast/net/IPAddress.h>
#include <xrpl/beast/net/IPAddressConversion.h>
#include <xrpl/beast/rfc2616.h>
@@ -183,7 +184,11 @@ ServerHandler::ServerHandler(
{
auto const& group(cm.group("rpc"));
rpcRequests_ = group->makeCounter("requests");
rpcSize_ = group->makeEvent("size");
// "size" measures the serialized response in bytes, not a duration. It
// has to say so: the unit picks both the exported name suffix and the
// histogram bucket ladder, and borrowing the millisecond ladder censored
// a quarter of these samples.
rpcSize_ = group->makeEvent("size", beast::insight::Unit::Bytes);
rpcTime_ = group->makeEvent("time");
}
@@ -1125,7 +1130,7 @@ ServerHandler::processRequest(
std::chrono::duration_cast<std::chrono::milliseconds>(
std::chrono::high_resolution_clock::now() - start));
++rpcRequests_;
rpcSize_.notify(beast::insight::Event::value_type{response.size()});
rpcSize_.notify(static_cast<std::uint64_t>(response.size()));
response += '\n';

View File

@@ -68,6 +68,7 @@
#include <xrpl/server/LoadFeeTrack.h>
#include <xrpl/server/NetworkOPs.h>
#include <xrpl/telemetry/GetObjectMetricNames.h>
#include <xrpl/telemetry/HistogramBuckets.h>
#include <xrpl/telemetry/SpanNames.h>
#include <opentelemetry/context/context.h>
@@ -89,7 +90,6 @@
#include <opentelemetry/semconv/incubating/service_attributes.h>
#include <algorithm>
#include <array>
#include <atomic>
#include <chrono>
#include <cstddef>
@@ -132,36 +132,16 @@ constexpr char kRpcMethodDurationUs[] = "rpc_method_us";
// after start().
constexpr char kConsensusRoundDurationMs[] = "consensus_round_duration_ms";
/**
* Bucket boundaries for microsecond-valued duration instruments.
*
* 100 µs, 500 µs, 1 ms, 5 ms, 10 ms, 25 ms, 50 ms, 100 ms, 250 ms, 500 ms,
* 1 s, 2.5 s, 5 s, 10 s, 30 s, 60 s. Covers sub-millisecond jobs through
* multi-second stalls without saturating.
*/
constexpr std::array kMicrosecondBoundaries{
100.0,
500.0,
1'000.0,
5'000.0,
10'000.0,
25'000.0,
50'000.0,
100'000.0,
250'000.0,
500'000.0,
1'000'000.0,
2'500'000.0,
5'000'000.0,
10'000'000.0,
30'000'000.0,
60'000'000.0};
/**
* Register an explicit-bucket histogram view.
*
* The SDK's default boundaries top out at 10,000, so any instrument whose
* values exceed that saturates and every quantile reads as the ceiling.
* values exceed that saturates and every quantile reads as the ceiling. The
* floor matters just as much and is easier to miss: a ladder whose first edge
* sits above the mass of the distribution makes every low quantile an
* interpolation inside bucket 0 -- a number derived from the bucket edge
* rather than from any sample. Both ends are chosen from measured
* distributions in HistogramBuckets.h.
*
* @param views The registry to add the view to.
* @param name Instrument name to match (e.g. "job_running_us").
@@ -189,9 +169,7 @@ addHistogramView(
* Register the microsecond-ladder view for a duration instrument.
*
* Job wait/run times and RPC latencies routinely exceed the SDK default
* ceiling, so they all share `kMicrosecondBoundaries`: 100µs, 500µs, 1ms, 5ms,
* 10ms, 25ms, 50ms, 100ms, 250ms, 500ms, 1s, 2.5s, 5s, 10s, 30s, 60s —
* sub-millisecond jobs through multi-second stalls, without saturating.
* ceiling, so they all share `buckets::kMicrosecondBuckets`.
*
* @param views The registry to add the view to.
* @param name Instrument name to match (e.g. "job_running_us").
@@ -199,7 +177,10 @@ addHistogramView(
void
addMicrosecondHistogramView(metric_sdk::ViewRegistry& views, std::string const& name)
{
addHistogramView(views, name, {kMicrosecondBoundaries.begin(), kMicrosecondBoundaries.end()});
addHistogramView(
views,
name,
xrpl::telemetry::buckets::toVector(xrpl::telemetry::buckets::kMicrosecondBuckets));
}
/**
@@ -438,18 +419,13 @@ MetricsRegistry::initExporterAndProvider(
// asks for at most 8, so the low buckets are fine-grained and the upper
// ones follow the charge size bands (64, 1024) up to the hard cap.
addHistogramView(
*views,
kGetObjectRequestObjects,
{1.0, 2.0, 4.0, 8.0, 16.0, 64.0, 256.0, 1'024.0, 4'096.0, 12'288.0});
*views, kGetObjectRequestObjects, buckets::toVector(buckets::kObjectCountBuckets));
// Charge values span 0 (free tier) to ~99k for a full-size all-miss
// request. Boundaries bracket the resource thresholds that decide a
// peer's fate -- kWarningThreshold (5000) and kDropThreshold (25000) --
// so a dashboard can show how close charges run to each.
addHistogramView(
*views,
kGetObjectCharge,
{0.0, 100.0, 500.0, 1'000.0, 5'000.0, 10'000.0, 25'000.0, 50'000.0, 100'000.0});
addHistogramView(*views, kGetObjectCharge, buckets::toVector(buckets::kChargeBuckets));
// Create MeterProvider with resource, then attach the metric reader.
provider_ = metric_sdk::MeterProviderFactory::Create(std::move(views), resourceAttrs);