mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-28 15:58:07 +00:00
Merge branch 'pratik/otel-phase10-workload-validation' into pratik/otel-sync-diagnostics
# Conflicts: # src/xrpld/telemetry/MetricsRegistry.cpp
This commit is contained in:
@@ -9,6 +9,7 @@
|
||||
#include <xrpl/beast/insight/Hook.h>
|
||||
#include <xrpl/beast/insight/HookImpl.h>
|
||||
#include <xrpl/beast/insight/Meter.h>
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
|
||||
#include <memory>
|
||||
#include <string>
|
||||
@@ -56,12 +57,25 @@ public:
|
||||
return collector_->makeCounter(makeName(name));
|
||||
}
|
||||
|
||||
using Collector::makeEvent;
|
||||
|
||||
Event
|
||||
makeEvent(std::string const& name) override
|
||||
{
|
||||
return collector_->makeEvent(makeName(name));
|
||||
}
|
||||
|
||||
// Forwards the unit as well as the prefixed name. Without this override
|
||||
// the base-class default would delegate to the single-argument overload
|
||||
// above and silently drop the unit, which is how a byte-valued Event ends
|
||||
// up declared as milliseconds -- call sites reach a collector through a
|
||||
// Group, so this is the hop that actually matters.
|
||||
Event
|
||||
makeEvent(std::string const& name, Unit unit) override
|
||||
{
|
||||
return collector_->makeEvent(makeName(name), unit);
|
||||
}
|
||||
|
||||
Gauge
|
||||
makeGauge(std::string const& name) override
|
||||
{
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <xrpl/beast/insight/HookImpl.h>
|
||||
#include <xrpl/beast/insight/Meter.h>
|
||||
#include <xrpl/beast/insight/MeterImpl.h>
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
|
||||
#include <memory>
|
||||
#include <string>
|
||||
@@ -49,7 +50,15 @@ public:
|
||||
class NullEventImpl : public EventImpl
|
||||
{
|
||||
public:
|
||||
explicit NullEventImpl() = default;
|
||||
/**
|
||||
* @param unit What the samples would measure. Recorded even though
|
||||
* nothing is collected, so a caller can still read back the
|
||||
* unit it asked for -- which is what makes the null collector
|
||||
* usable for testing the unit plumbing.
|
||||
*/
|
||||
explicit NullEventImpl(Unit unit = Unit::Millis) : EventImpl(unit)
|
||||
{
|
||||
}
|
||||
|
||||
void
|
||||
notify(value_type const&) override
|
||||
@@ -119,12 +128,20 @@ public:
|
||||
return Counter(std::make_shared<detail::NullCounterImpl>());
|
||||
}
|
||||
|
||||
using Collector::makeEvent;
|
||||
|
||||
Event
|
||||
makeEvent(std::string const&) override
|
||||
{
|
||||
return Event(std::make_shared<detail::NullEventImpl>());
|
||||
}
|
||||
|
||||
Event
|
||||
makeEvent(std::string const&, Unit unit) override
|
||||
{
|
||||
return Event(std::make_shared<detail::NullEventImpl>(unit));
|
||||
}
|
||||
|
||||
Gauge
|
||||
makeGauge(std::string const&) override
|
||||
{
|
||||
|
||||
@@ -41,6 +41,7 @@
|
||||
#include <xrpl/beast/insight/Hook.h>
|
||||
#include <xrpl/beast/insight/HookImpl.h>
|
||||
#include <xrpl/beast/insight/MeterImpl.h>
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
#include <xrpl/beast/utility/Journal.h>
|
||||
|
||||
#include <opentelemetry/metrics/async_instruments.h>
|
||||
@@ -168,10 +169,17 @@ private:
|
||||
/**
|
||||
* @brief OTel-backed implementation of beast::insight::EventImpl.
|
||||
*
|
||||
* Wraps an OTel Histogram<double> instrument. Each notify() call
|
||||
* records the duration in milliseconds. Uses explicit bucket boundaries
|
||||
* matching the SpanMetrics connector configuration:
|
||||
* [1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000] ms
|
||||
* Wraps an OTel Histogram<double> instrument. Each notify() call records one
|
||||
* sample, interpreted per the Event's unit().
|
||||
*
|
||||
* The instrument's declared unit is what selects its bucket ladder: the
|
||||
* histogram views registered in Telemetry.cpp match on unit, so a `ms`
|
||||
* instrument gets the millisecond ladder and a `By` instrument the byte
|
||||
* ladder. The edges themselves live in xrpl/telemetry/HistogramBuckets.h --
|
||||
* do not restate them here. An earlier version of this comment listed
|
||||
* `[1, 5, ..., 1000, 5000] ms` as "matching the SpanMetrics connector"; that
|
||||
* was true when written and silently became false when the connector's
|
||||
* ladder was extended, which is why the edges now have one owner.
|
||||
*
|
||||
* Thread safety: OTel Histogram::Record() is thread-safe by specification.
|
||||
*/
|
||||
@@ -183,10 +191,14 @@ public:
|
||||
* formatName() by the collector: lowercase, with `.` and
|
||||
* ` ` mapped to `_` (e.g. "rpc_size").
|
||||
* @param meter OTel Meter used to create the histogram instrument.
|
||||
* @param unit What the samples measure. Selects the instrument's
|
||||
* declared unit, its description, and through the unit the
|
||||
* bucket ladder a histogram view applies.
|
||||
*/
|
||||
OTelEventImpl(
|
||||
std::string const& name,
|
||||
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter);
|
||||
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter,
|
||||
Unit unit);
|
||||
|
||||
~OTelEventImpl() override = default;
|
||||
|
||||
@@ -474,6 +486,9 @@ public:
|
||||
Event
|
||||
makeEvent(std::string const& name) override;
|
||||
|
||||
Event
|
||||
makeEvent(std::string const& name, Unit unit) override;
|
||||
|
||||
Gauge
|
||||
makeGauge(std::string const& name) override;
|
||||
|
||||
@@ -652,8 +667,10 @@ OTelCounterImpl::increment(value_type amount)
|
||||
|
||||
OTelEventImpl::OTelEventImpl(
|
||||
std::string const& name,
|
||||
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter)
|
||||
: histogram_(meter->CreateDoubleHistogram(name, "Duration in ms", "ms"))
|
||||
opentelemetry::nostd::shared_ptr<metrics_api::Meter> const& meter,
|
||||
Unit unit)
|
||||
: EventImpl(unit)
|
||||
, histogram_(meter->CreateDoubleHistogram(name, otelUnitDescription(unit), otelUnitCode(unit)))
|
||||
{
|
||||
}
|
||||
|
||||
@@ -842,7 +859,13 @@ OTelCollectorImp::makeCounter(std::string const& name)
|
||||
Event
|
||||
OTelCollectorImp::makeEvent(std::string const& name)
|
||||
{
|
||||
return Event(std::make_shared<OTelEventImpl>(formatName(name), otelMeter_));
|
||||
return makeEvent(name, Unit::Millis);
|
||||
}
|
||||
|
||||
Event
|
||||
OTelCollectorImp::makeEvent(std::string const& name, Unit unit)
|
||||
{
|
||||
return Event(std::make_shared<OTelEventImpl>(formatName(name), otelMeter_, unit));
|
||||
}
|
||||
|
||||
Gauge
|
||||
|
||||
@@ -19,10 +19,12 @@
|
||||
#include <xrpl/telemetry/Telemetry.h>
|
||||
|
||||
#include <xrpl/basics/Log.h>
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
#include <xrpl/beast/utility/Journal.h>
|
||||
#include <xrpl/telemetry/CoroAwareContextStorage.h>
|
||||
#include <xrpl/telemetry/DeterministicIdGenerator.h>
|
||||
#include <xrpl/telemetry/DiscardFlag.h>
|
||||
#include <xrpl/telemetry/HistogramBuckets.h>
|
||||
#include <xrpl/telemetry/SpanNames.h>
|
||||
|
||||
#include <opentelemetry/context/context.h>
|
||||
@@ -553,31 +555,46 @@ public:
|
||||
std::make_unique<metrics_sdk::ViewRegistry>(), resourceAttrs);
|
||||
meterProvider_->AddMetricReader(std::move(reader));
|
||||
|
||||
// Histogram view: SpanMetrics-compatible bucket boundaries (ms) so
|
||||
// histogram instruments align with the collector's SpanMetrics. The
|
||||
// view is created with an EMPTY name so it applies the buckets WITHOUT
|
||||
// renaming instruments — a non-empty view name would collapse every
|
||||
// matching histogram (ios_latency, rpc_size, rpc_time, pathfind_*)
|
||||
// into a single series under that one name.
|
||||
auto histogramSelector = metrics_sdk::InstrumentSelectorFactory::Create(
|
||||
metrics_sdk::InstrumentType::kHistogram, "*", "ms");
|
||||
// Meter selector MUST match the meter name used by getMeter() and the
|
||||
// beast OTelCollector (kMeterName = "xrpld"); otherwise this histogram
|
||||
// view never applies and duration histograms fall back to the SDK
|
||||
// default boundaries instead of these SpanMetrics-aligned buckets.
|
||||
auto meterSelector =
|
||||
metrics_sdk::MeterSelectorFactory::Create(std::string(kMeterName), "", "");
|
||||
auto histogramConfig = std::make_shared<metrics_sdk::HistogramAggregationConfig>();
|
||||
histogramConfig->boundaries_ =
|
||||
std::vector<double>{1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0, 5000.0};
|
||||
auto histogramView = metrics_sdk::ViewFactory::Create(
|
||||
"", // empty name: keep each instrument's own name, only set buckets
|
||||
"SpanMetrics-compatible histogram buckets",
|
||||
metrics_sdk::AggregationType::kHistogram,
|
||||
histogramConfig);
|
||||
// One histogram view per unit. The unit is the selector, so an
|
||||
// instrument gets the ladder that fits what it measures -- a byte
|
||||
// count no longer inherits a latency ladder. Edges come from
|
||||
// HistogramBuckets.h, which owns every ladder.
|
||||
//
|
||||
// Each view keeps the "*" name pattern and an EMPTY view name: a
|
||||
// non-empty view name would rename every matching histogram to it and
|
||||
// collapse them (ios_latency, rpc_size, rpc_time, pathfind_*, and all
|
||||
// the jobq_* pairs) into a single series.
|
||||
//
|
||||
// The meter selector MUST match the meter name used by getMeter() and
|
||||
// the beast OTelCollector (kMeterName = "xrpld"); otherwise a view
|
||||
// never applies and instruments fall back to the SDK default ladder,
|
||||
// whose ceiling is 10,000.
|
||||
auto const addUnitView = [this](
|
||||
std::string const& unitCode,
|
||||
std::vector<double> boundaries,
|
||||
std::string const& description) {
|
||||
auto selector = metrics_sdk::InstrumentSelectorFactory::Create(
|
||||
metrics_sdk::InstrumentType::kHistogram, "*", unitCode);
|
||||
auto meterSelector =
|
||||
metrics_sdk::MeterSelectorFactory::Create(std::string(kMeterName), "", "");
|
||||
auto config = std::make_shared<metrics_sdk::HistogramAggregationConfig>();
|
||||
config->boundaries_ = std::move(boundaries);
|
||||
auto view = metrics_sdk::ViewFactory::Create(
|
||||
"", // empty name: keep each instrument's own name, only set buckets
|
||||
description,
|
||||
metrics_sdk::AggregationType::kHistogram,
|
||||
std::move(config));
|
||||
meterProvider_->AddView(std::move(selector), std::move(meterSelector), std::move(view));
|
||||
};
|
||||
|
||||
meterProvider_->AddView(
|
||||
std::move(histogramSelector), std::move(meterSelector), std::move(histogramView));
|
||||
addUnitView(
|
||||
beast::insight::otelUnitCode(beast::insight::Unit::Millis),
|
||||
buckets::toVector(buckets::kMillisecondBuckets),
|
||||
"Duration buckets, 1 ms to 120 s");
|
||||
addUnitView(
|
||||
beast::insight::otelUnitCode(beast::insight::Unit::Bytes),
|
||||
buckets::toVector(buckets::kByteBuckets),
|
||||
"Size buckets, 512 B to 1 MiB");
|
||||
|
||||
// Publish as the global meter provider so developers (and the beast
|
||||
// OTelCollector shim) reach the same pipeline.
|
||||
|
||||
174
src/tests/libxrpl/beast/insight/Unit.cpp
Normal file
174
src/tests/libxrpl/beast/insight/Unit.cpp
Normal file
@@ -0,0 +1,174 @@
|
||||
/**
|
||||
* GTest unit tests for beast::insight::Unit and its plumbing.
|
||||
*
|
||||
* A metric's unit decides two things that are invisible at the call site: the
|
||||
* name suffix the exporter appends, and which bucket ladder the histogram
|
||||
* view applies. Getting it wrong is silent -- a byte count declared as
|
||||
* milliseconds still records, still exports, still draws a graph, and the
|
||||
* graph is wrong. So each hop the unit has to survive is asserted here
|
||||
* rather than left to inspection.
|
||||
*
|
||||
* The hop that matters most is the group wrapper. Call sites reach a
|
||||
* collector through Groups, so a unit that reaches OTelCollector correctly
|
||||
* but is dropped by the group prefixing layer would pass a naive test while
|
||||
* failing in production.
|
||||
*/
|
||||
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
|
||||
#include <xrpl/beast/insight/Event.h>
|
||||
#include <xrpl/beast/insight/EventImpl.h>
|
||||
#include <xrpl/beast/insight/Groups.h>
|
||||
#include <xrpl/beast/insight/NullCollector.h>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
namespace beast::insight {
|
||||
|
||||
namespace {
|
||||
|
||||
/**
|
||||
* An EventImpl that records what it was notified with.
|
||||
*
|
||||
* Needed because every shipped implementation either discards the sample
|
||||
* (NullCollector) or sends it somewhere external. Asserting the recorded
|
||||
* value proves the raw-integral path preserves it, rather than only proving
|
||||
* that notify() can be called without crashing.
|
||||
*/
|
||||
class RecordingEventImpl : public EventImpl
|
||||
{
|
||||
public:
|
||||
explicit RecordingEventImpl(Unit unit) : EventImpl(unit)
|
||||
{
|
||||
}
|
||||
|
||||
void
|
||||
notify(value_type const& value) override
|
||||
{
|
||||
samples.push_back(value);
|
||||
}
|
||||
|
||||
/**
|
||||
* Every value passed to notify(), in call order.
|
||||
*/
|
||||
std::vector<value_type> samples;
|
||||
};
|
||||
|
||||
} // namespace
|
||||
|
||||
// The unit code is a contract with the collector's Prometheus exporter: it
|
||||
// derives the exported name suffix from this string. Assert the exact codes,
|
||||
// not merely that they differ.
|
||||
TEST(InsightUnit, otelCodeIsTheUcumCodeForEachUnit)
|
||||
{
|
||||
EXPECT_STREQ(otelUnitCode(Unit::Millis), "ms");
|
||||
EXPECT_STREQ(otelUnitCode(Unit::Bytes), "By");
|
||||
}
|
||||
|
||||
// The description is what an operator reads in the metric catalogue, so a
|
||||
// byte-valued instrument must not describe itself as a duration.
|
||||
TEST(InsightUnit, descriptionMatchesWhatTheUnitActuallyMeasures)
|
||||
{
|
||||
EXPECT_STREQ(otelUnitDescription(Unit::Millis), "Duration in ms");
|
||||
EXPECT_STREQ(otelUnitDescription(Unit::Bytes), "Size in bytes");
|
||||
}
|
||||
|
||||
TEST(InsightUnit, defaultEventUnitIsMillisForBackwardCompatibility)
|
||||
{
|
||||
// Every pre-existing makeEvent(name) call site records a duration, so the
|
||||
// one-argument overload must keep meaning milliseconds.
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("legacy");
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, makeEventCarriesTheRequestedUnitToTheImpl)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, prefixedMakeEventCarriesTheUnit)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const event = collector->makeEvent("rpc", "size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, groupWrapperForwardsTheUnitAlongWithThePrefix)
|
||||
{
|
||||
// ServerHandler creates its events through a Group, not through the
|
||||
// collector directly. If the group's makeEvent override forwards only the
|
||||
// name, the unit silently reverts to milliseconds and the byte histogram
|
||||
// inherits the latency ladder again.
|
||||
auto const collector = NullCollector::make();
|
||||
auto const groups = makeGroups(collector);
|
||||
auto const event = groups->get("rpc")->makeEvent("size", Unit::Bytes);
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Bytes);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, groupWrapperStillDefaultsToMillis)
|
||||
{
|
||||
auto const collector = NullCollector::make();
|
||||
auto const groups = makeGroups(collector);
|
||||
auto const event = groups->get("rpc")->makeEvent("time");
|
||||
ASSERT_NE(event.impl(), nullptr);
|
||||
EXPECT_EQ(event.impl()->unit(), Unit::Millis);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, rawIntegralNotifyPreservesTheValueExactly)
|
||||
{
|
||||
// The byte path must not be rounded or scaled on its way through the
|
||||
// duration-typed storage field.
|
||||
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Bytes);
|
||||
Event const event(impl);
|
||||
|
||||
event.notify(std::uint64_t{4096});
|
||||
event.notify(std::uint64_t{0});
|
||||
event.notify(std::uint64_t{1'048'577});
|
||||
|
||||
ASSERT_EQ(impl->samples.size(), 3U);
|
||||
EXPECT_EQ(impl->samples[0].count(), 4096);
|
||||
EXPECT_EQ(impl->samples[1].count(), 0);
|
||||
EXPECT_EQ(impl->samples[2].count(), 1'048'577);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, durationNotifyStillRoundsUpToWholeMilliseconds)
|
||||
{
|
||||
// Pre-existing behaviour, asserted so the new overload cannot quietly
|
||||
// change it: Event applies ceil to whole milliseconds, which is why
|
||||
// sub-millisecond resolution is impossible on this path.
|
||||
auto const impl = std::make_shared<RecordingEventImpl>(Unit::Millis);
|
||||
Event const event(impl);
|
||||
|
||||
event.notify(std::chrono::microseconds{40});
|
||||
event.notify(std::chrono::microseconds{1'000});
|
||||
event.notify(std::chrono::milliseconds{7});
|
||||
|
||||
ASSERT_EQ(impl->samples.size(), 3U);
|
||||
EXPECT_EQ(impl->samples[0].count(), 1) << "40us must round up to 1ms, not down to 0";
|
||||
EXPECT_EQ(impl->samples[1].count(), 1);
|
||||
EXPECT_EQ(impl->samples[2].count(), 7);
|
||||
}
|
||||
|
||||
TEST(InsightUnit, notifyOnANullEventIsSafeForBothOverloads)
|
||||
{
|
||||
// A default-constructed Event has no impl. Both overloads must be no-ops
|
||||
// rather than dereferencing null.
|
||||
Event const none;
|
||||
ASSERT_EQ(none.impl(), nullptr);
|
||||
EXPECT_NO_THROW(none.notify(std::uint64_t{4096}));
|
||||
EXPECT_NO_THROW(none.notify(std::chrono::milliseconds{5}));
|
||||
}
|
||||
|
||||
} // namespace beast::insight
|
||||
235
src/tests/libxrpl/telemetry/HistogramBuckets.cpp
Normal file
235
src/tests/libxrpl/telemetry/HistogramBuckets.cpp
Normal file
@@ -0,0 +1,235 @@
|
||||
/**
|
||||
* GTest unit tests for the histogram bucket ladders.
|
||||
*
|
||||
* These ladders decide whether a Grafana percentile panel reports a
|
||||
* measurement or an artefact, and neither failure mode is visible in the
|
||||
* panel itself: a quantile that falls in the `+Inf` bucket reads back as the
|
||||
* second-highest edge, and one that falls inside bucket 0 is interpolated.
|
||||
* Both look like plausible numbers. So the invariants are asserted here
|
||||
* rather than left to review.
|
||||
*
|
||||
* The ladders are `constexpr`, so most of this could be `static_assert`.
|
||||
* They are runtime tests as well so that a failure names which edge is
|
||||
* wrong instead of only failing the compile.
|
||||
*/
|
||||
|
||||
#include <xrpl/telemetry/HistogramBuckets.h>
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <span>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::telemetry::buckets {
|
||||
|
||||
// Every ladder must be strictly ascending and non-negative. The SDK places a
|
||||
// sample with std::lower_bound over the edges, so a duplicated or
|
||||
// out-of-order edge silently sends samples to the wrong bucket.
|
||||
class HistogramBucketsTest : public ::testing::TestWithParam<std::span<double const>>
|
||||
{
|
||||
};
|
||||
|
||||
TEST_P(HistogramBucketsTest, isStrictlyAscending)
|
||||
{
|
||||
auto const ladder = GetParam();
|
||||
ASSERT_FALSE(ladder.empty());
|
||||
for (std::size_t i = 1; i < ladder.size(); ++i)
|
||||
EXPECT_LT(ladder[i - 1], ladder[i]) << "edge index " << i << " does not ascend";
|
||||
}
|
||||
|
||||
TEST_P(HistogramBucketsTest, isNonNegativeAndFinite)
|
||||
{
|
||||
for (double const edge : GetParam())
|
||||
{
|
||||
EXPECT_GE(edge, 0.0);
|
||||
EXPECT_TRUE(std::isfinite(edge)) << "edge " << edge << " is not finite";
|
||||
}
|
||||
}
|
||||
|
||||
TEST_P(HistogramBucketsTest, passesTheCompileTimeValidator)
|
||||
{
|
||||
EXPECT_TRUE(isAscendingNonNegative(GetParam()));
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_SUITE_P(
|
||||
AllLadders,
|
||||
HistogramBucketsTest,
|
||||
::testing::Values(
|
||||
std::span<double const>{kMillisecondBuckets},
|
||||
std::span<double const>{kByteBuckets},
|
||||
std::span<double const>{kMicrosecondBuckets},
|
||||
std::span<double const>{kObjectCountBuckets},
|
||||
std::span<double const>{kChargeBuckets}));
|
||||
|
||||
TEST(HistogramBucketsRange, microsecondFloorLandsBelowTheMeasuredMass)
|
||||
{
|
||||
// Measured: 99.3% of job_queued_us samples sat below the old 100 us floor,
|
||||
// so p75/p95/p99 all interpolated inside bucket 0 and returned
|
||||
// 75.5/95.7/99.7 us -- the boundary scaled by the requested quantile,
|
||||
// not a latency. Warm nodestore reads are ~1.5 us, so the floor has to
|
||||
// reach single microseconds and several edges must precede 100 us.
|
||||
EXPECT_LE(kMicrosecondBuckets.front(), 1.0);
|
||||
|
||||
auto const belowHundred =
|
||||
std::ranges::count_if(kMicrosecondBuckets, [](double edge) { return edge < 100.0; });
|
||||
EXPECT_GE(belowHundred, 5) << "too little resolution below 100 us";
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, microsecondCeilingStillReachesOneMinute)
|
||||
{
|
||||
// Job waits and RPC latencies routinely exceed the SDK default ceiling of
|
||||
// 10,000; multi-second stalls must stay measurable rather than censored.
|
||||
EXPECT_EQ(kMicrosecondBuckets.back(), 60'000'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, objectCountLadderCannotSaturate)
|
||||
{
|
||||
// GetObject counts run 1..kHardMaxReplyNodes, so the top edge IS the hard
|
||||
// cap and censoring is impossible by construction.
|
||||
EXPECT_EQ(kObjectCountBuckets.front(), 1.0);
|
||||
EXPECT_EQ(kObjectCountBuckets.back(), 12'288.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, chargeLadderBracketsTheResourceThresholds)
|
||||
{
|
||||
// The two edges that decide a peer's fate must be present so a dashboard
|
||||
// can show how close charges run to each: warning at 5000, drop at 25000.
|
||||
// A leading 0 separates the free tier from everything else.
|
||||
EXPECT_EQ(kChargeBuckets.front(), 0.0);
|
||||
for (double const threshold : {5'000.0, 25'000.0})
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kChargeBuckets, threshold), kChargeBuckets.end())
|
||||
<< threshold << " is a resource threshold and must be an edge";
|
||||
}
|
||||
}
|
||||
|
||||
// The validator must also REJECT. A predicate that only ever returns true
|
||||
// would let every ladder above pass while proving nothing.
|
||||
TEST(HistogramBucketsValidator, rejectsEmptyDescendingDuplicateAndNegative)
|
||||
{
|
||||
EXPECT_FALSE(isAscendingNonNegative(std::span<double const>{}));
|
||||
|
||||
constexpr std::array descending{5.0, 1.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(descending));
|
||||
|
||||
constexpr std::array duplicated{1.0, 1.0, 2.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(duplicated));
|
||||
|
||||
constexpr std::array negative{-1.0, 1.0};
|
||||
EXPECT_FALSE(isAscendingNonNegative(negative));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsValidator, acceptsASingleEdgeAndALeadingZero)
|
||||
{
|
||||
constexpr std::array single{1.0};
|
||||
EXPECT_TRUE(isAscendingNonNegative(single));
|
||||
|
||||
// A leading zero is legal: the GetObject charge ladder starts at 0 to
|
||||
// separate the free tier from everything else.
|
||||
constexpr std::array leadingZero{0.0, 100.0};
|
||||
EXPECT_TRUE(isAscendingNonNegative(leadingZero));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondFloorIsOneAndCeilingCoversTheSlowestJob)
|
||||
{
|
||||
// beast::insight::Event rounds durations up to whole milliseconds, so 1
|
||||
// is the smallest edge that can ever collect a sample.
|
||||
EXPECT_EQ(kMillisecondBuckets.front(), 1.0);
|
||||
|
||||
// The updatepaths job type was measured averaging 59,956 ms. A 30 s
|
||||
// ceiling -- the collector's top edge -- would censor it just as the old
|
||||
// 5 s ceiling does, so this ladder has to reach further.
|
||||
EXPECT_GE(kMillisecondBuckets.back(), 120'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderClearsTheMeasuredCensoringPoint)
|
||||
{
|
||||
// rpc_size had 24.9% of samples above the old 5000 ceiling and
|
||||
// jobq_updatepaths had 100%. A ceiling at or below 5000 reintroduces the
|
||||
// exact defect this ladder exists to fix.
|
||||
EXPECT_GT(kMillisecondBuckets.back(), 5'000.0);
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderContainsEveryRepresentableCollectorEdge)
|
||||
{
|
||||
// Agreement with the collector's spanmetrics ladder over the shared
|
||||
// range is the invariant; edges above its 30 s top are allowed because
|
||||
// jobs outlive spans. Sub-millisecond collector edges are excluded
|
||||
// because Event cannot represent them. check_bucket_parity.py enforces
|
||||
// this against the YAML; this test pins it for the C++ side alone so a
|
||||
// local edit fails fast.
|
||||
constexpr std::array collectorEdges{
|
||||
1.0,
|
||||
5.0,
|
||||
10.0,
|
||||
25.0,
|
||||
50.0,
|
||||
100.0,
|
||||
250.0,
|
||||
500.0,
|
||||
1'000.0,
|
||||
2'000.0,
|
||||
3'000.0,
|
||||
4'000.0,
|
||||
5'000.0,
|
||||
10'000.0,
|
||||
30'000.0};
|
||||
|
||||
for (double const edge : collectorEdges)
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
|
||||
<< edge << " ms is a collector spanmetrics edge and must be present";
|
||||
}
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, millisecondLadderResolvesTheOneToFiveSecondBand)
|
||||
{
|
||||
// Without these the 1 s to 5 s span was one four-second-wide bucket, so
|
||||
// any quantile landing inside it was interpolated across four seconds.
|
||||
for (double const edge : {2'000.0, 3'000.0, 4'000.0})
|
||||
{
|
||||
EXPECT_NE(std::ranges::find(kMillisecondBuckets, edge), kMillisecondBuckets.end())
|
||||
<< edge << " ms edge missing";
|
||||
}
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, byteLadderBracketsTheMeasuredResponseDistribution)
|
||||
{
|
||||
// Measured: mean 2131 B, half under 1 kB, three quarters under 5 kB, and
|
||||
// the tail above 5 kB has a mean of at most 7538 B -- which puts p99
|
||||
// near 80 kB. The floor must sit at or below the measured median region
|
||||
// and the ceiling well past the p99 bound.
|
||||
EXPECT_LE(kByteBuckets.front(), 512.0);
|
||||
EXPECT_GE(kByteBuckets.back(), 1'048'576.0);
|
||||
|
||||
// Most of the resolution belongs where the distribution actually turns.
|
||||
auto const withinWorkingRange =
|
||||
std::ranges::count_if(kByteBuckets, [](double e) { return e >= 512.0 && e <= 65'536.0; });
|
||||
EXPECT_GE(withinWorkingRange, 6) << "too little resolution between 512 B and 64 kB";
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsRange, byteAndMillisecondLaddersAreDistinct)
|
||||
{
|
||||
// A single shared ladder is what put a byte count on a latency scale and
|
||||
// censored a quarter of its samples.
|
||||
EXPECT_NE(kByteBuckets.size(), kMillisecondBuckets.size());
|
||||
EXPECT_GT(kByteBuckets.back(), kMillisecondBuckets.back());
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsConvert, toVectorPreservesOrderAndSize)
|
||||
{
|
||||
auto const converted = toVector(kByteBuckets);
|
||||
ASSERT_EQ(converted.size(), kByteBuckets.size());
|
||||
EXPECT_TRUE(std::ranges::equal(converted, kByteBuckets));
|
||||
}
|
||||
|
||||
TEST(HistogramBucketsConvert, toVectorHandlesAnEmptyLadder)
|
||||
{
|
||||
EXPECT_TRUE(toVector(std::span<double const>{}).empty());
|
||||
}
|
||||
|
||||
} // namespace xrpl::telemetry::buckets
|
||||
@@ -16,6 +16,7 @@
|
||||
#include <xrpl/basics/base64.h>
|
||||
#include <xrpl/basics/contract.h>
|
||||
#include <xrpl/basics/make_SSLContext.h>
|
||||
#include <xrpl/beast/insight/Unit.h>
|
||||
#include <xrpl/beast/net/IPAddress.h>
|
||||
#include <xrpl/beast/net/IPAddressConversion.h>
|
||||
#include <xrpl/beast/rfc2616.h>
|
||||
@@ -183,7 +184,11 @@ ServerHandler::ServerHandler(
|
||||
{
|
||||
auto const& group(cm.group("rpc"));
|
||||
rpcRequests_ = group->makeCounter("requests");
|
||||
rpcSize_ = group->makeEvent("size");
|
||||
// "size" measures the serialized response in bytes, not a duration. It
|
||||
// has to say so: the unit picks both the exported name suffix and the
|
||||
// histogram bucket ladder, and borrowing the millisecond ladder censored
|
||||
// a quarter of these samples.
|
||||
rpcSize_ = group->makeEvent("size", beast::insight::Unit::Bytes);
|
||||
rpcTime_ = group->makeEvent("time");
|
||||
}
|
||||
|
||||
@@ -1125,7 +1130,7 @@ ServerHandler::processRequest(
|
||||
std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
std::chrono::high_resolution_clock::now() - start));
|
||||
++rpcRequests_;
|
||||
rpcSize_.notify(beast::insight::Event::value_type{response.size()});
|
||||
rpcSize_.notify(static_cast<std::uint64_t>(response.size()));
|
||||
|
||||
response += '\n';
|
||||
|
||||
|
||||
@@ -68,6 +68,7 @@
|
||||
#include <xrpl/server/LoadFeeTrack.h>
|
||||
#include <xrpl/server/NetworkOPs.h>
|
||||
#include <xrpl/telemetry/GetObjectMetricNames.h>
|
||||
#include <xrpl/telemetry/HistogramBuckets.h>
|
||||
#include <xrpl/telemetry/SpanNames.h>
|
||||
|
||||
#include <opentelemetry/context/context.h>
|
||||
@@ -89,7 +90,6 @@
|
||||
#include <opentelemetry/semconv/incubating/service_attributes.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <cstddef>
|
||||
@@ -132,36 +132,16 @@ constexpr char kRpcMethodDurationUs[] = "rpc_method_us";
|
||||
// after start().
|
||||
constexpr char kConsensusRoundDurationMs[] = "consensus_round_duration_ms";
|
||||
|
||||
/**
|
||||
* Bucket boundaries for microsecond-valued duration instruments.
|
||||
*
|
||||
* 100 µs, 500 µs, 1 ms, 5 ms, 10 ms, 25 ms, 50 ms, 100 ms, 250 ms, 500 ms,
|
||||
* 1 s, 2.5 s, 5 s, 10 s, 30 s, 60 s. Covers sub-millisecond jobs through
|
||||
* multi-second stalls without saturating.
|
||||
*/
|
||||
constexpr std::array kMicrosecondBoundaries{
|
||||
100.0,
|
||||
500.0,
|
||||
1'000.0,
|
||||
5'000.0,
|
||||
10'000.0,
|
||||
25'000.0,
|
||||
50'000.0,
|
||||
100'000.0,
|
||||
250'000.0,
|
||||
500'000.0,
|
||||
1'000'000.0,
|
||||
2'500'000.0,
|
||||
5'000'000.0,
|
||||
10'000'000.0,
|
||||
30'000'000.0,
|
||||
60'000'000.0};
|
||||
|
||||
/**
|
||||
* Register an explicit-bucket histogram view.
|
||||
*
|
||||
* The SDK's default boundaries top out at 10,000, so any instrument whose
|
||||
* values exceed that saturates and every quantile reads as the ceiling.
|
||||
* values exceed that saturates and every quantile reads as the ceiling. The
|
||||
* floor matters just as much and is easier to miss: a ladder whose first edge
|
||||
* sits above the mass of the distribution makes every low quantile an
|
||||
* interpolation inside bucket 0 -- a number derived from the bucket edge
|
||||
* rather than from any sample. Both ends are chosen from measured
|
||||
* distributions in HistogramBuckets.h.
|
||||
*
|
||||
* @param views The registry to add the view to.
|
||||
* @param name Instrument name to match (e.g. "job_running_us").
|
||||
@@ -189,9 +169,7 @@ addHistogramView(
|
||||
* Register the microsecond-ladder view for a duration instrument.
|
||||
*
|
||||
* Job wait/run times and RPC latencies routinely exceed the SDK default
|
||||
* ceiling, so they all share `kMicrosecondBoundaries`: 100µs, 500µs, 1ms, 5ms,
|
||||
* 10ms, 25ms, 50ms, 100ms, 250ms, 500ms, 1s, 2.5s, 5s, 10s, 30s, 60s —
|
||||
* sub-millisecond jobs through multi-second stalls, without saturating.
|
||||
* ceiling, so they all share `buckets::kMicrosecondBuckets`.
|
||||
*
|
||||
* @param views The registry to add the view to.
|
||||
* @param name Instrument name to match (e.g. "job_running_us").
|
||||
@@ -199,7 +177,10 @@ addHistogramView(
|
||||
void
|
||||
addMicrosecondHistogramView(metric_sdk::ViewRegistry& views, std::string const& name)
|
||||
{
|
||||
addHistogramView(views, name, {kMicrosecondBoundaries.begin(), kMicrosecondBoundaries.end()});
|
||||
addHistogramView(
|
||||
views,
|
||||
name,
|
||||
xrpl::telemetry::buckets::toVector(xrpl::telemetry::buckets::kMicrosecondBuckets));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -438,18 +419,13 @@ MetricsRegistry::initExporterAndProvider(
|
||||
// asks for at most 8, so the low buckets are fine-grained and the upper
|
||||
// ones follow the charge size bands (64, 1024) up to the hard cap.
|
||||
addHistogramView(
|
||||
*views,
|
||||
kGetObjectRequestObjects,
|
||||
{1.0, 2.0, 4.0, 8.0, 16.0, 64.0, 256.0, 1'024.0, 4'096.0, 12'288.0});
|
||||
*views, kGetObjectRequestObjects, buckets::toVector(buckets::kObjectCountBuckets));
|
||||
|
||||
// Charge values span 0 (free tier) to ~99k for a full-size all-miss
|
||||
// request. Boundaries bracket the resource thresholds that decide a
|
||||
// peer's fate -- kWarningThreshold (5000) and kDropThreshold (25000) --
|
||||
// so a dashboard can show how close charges run to each.
|
||||
addHistogramView(
|
||||
*views,
|
||||
kGetObjectCharge,
|
||||
{0.0, 100.0, 500.0, 1'000.0, 5'000.0, 10'000.0, 25'000.0, 50'000.0, 100'000.0});
|
||||
addHistogramView(*views, kGetObjectCharge, buckets::toVector(buckets::kChargeBuckets));
|
||||
|
||||
// Create MeterProvider with resource, then attach the metric reader.
|
||||
provider_ = metric_sdk::MeterProviderFactory::Create(std::move(views), resourceAttrs);
|
||||
|
||||
Reference in New Issue
Block a user