mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-27 23:38:08 +00:00
feat(telemetry): count job stalls and expose cache lock-hold peaks
Two new signals for the rotation-stall proof chain:
- `jobq_stall_total{job_type="<name>"}` — one increment per finished job whose
running duration reached `kJobStallThresholdUs` (1 s, LoadMonitor's own
warn bar). A process-wide freeze shows up as several job types crossing
the bar in the same second. One `int64` compare on the finish hook path.
- `cache_metrics{metric="treenode_lock_hold_peak_us"}` and
`{metric="fullbelow_lock_hold_peak_us"}` — observed once per collect tick
from `TaggedCache::takeLockHoldPeak()`, so the tick reads the longest hold
since the previous tick and resets. Two atomic exchanges per tick.
The two Observe calls live in a new `observeCacheLockHoldPeaks` helper so
`registerCacheHitRateGauge`'s callback stays inside the 80-line limit.
This commit is contained in:
@@ -504,6 +504,8 @@ MetricsRegistry::initSyncInstruments()
|
||||
jobQueuedCounter_ = meter_->CreateUInt64Counter("job_queued_total", "Total jobs enqueued");
|
||||
jobStartedCounter_ = meter_->CreateUInt64Counter("job_started_total", "Total jobs started");
|
||||
jobFinishedCounter_ = meter_->CreateUInt64Counter("job_finished_total", "Total jobs completed");
|
||||
jobStallCounter_ = meter_->CreateUInt64Counter(
|
||||
metric::jobqStallTotal, "Jobs whose run time reached the 1 s stall threshold");
|
||||
jobQueuedDurationHistogram_ = meter_->CreateDoubleHistogram(
|
||||
kJobQueuedDurationUs, "Time jobs spent waiting in the queue (microseconds)");
|
||||
jobRunningDurationHistogram_ =
|
||||
@@ -686,6 +688,10 @@ MetricsRegistry::recordJobFinished(
|
||||
{{label::jobType, std::string(jobType)}, {label::handler, handler}},
|
||||
opentelemetry::context::Context{});
|
||||
}
|
||||
// One compare per job finish. A process-wide freeze shows up here as
|
||||
// several job types crossing the bar in the same second.
|
||||
if (runningDurUs >= kJobStallThresholdUs && jobStallCounter_)
|
||||
jobStallCounter_->Add(1, {{label::jobType, std::string(jobType)}});
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -829,6 +835,10 @@ MetricsRegistry::registerCacheHitRateGauge()
|
||||
opentelemetry::nostd::get<opentelemetry::nostd::shared_ptr<
|
||||
opentelemetry::metrics::ObserverResultT<double>>>(result)
|
||||
->Observe(static_cast<double>(alSize), {{label::metric, "AL_size"}});
|
||||
|
||||
// Longest TaggedCache mutex hold since the last tick.
|
||||
// Split out to keep this callback under the 80-line limit.
|
||||
self->observeCacheLockHoldPeaks(result, app);
|
||||
}
|
||||
catch (...) // NOLINT(bugprone-empty-catch)
|
||||
{
|
||||
@@ -838,6 +848,28 @@ MetricsRegistry::registerCacheHitRateGauge()
|
||||
this);
|
||||
}
|
||||
|
||||
void
|
||||
MetricsRegistry::observeCacheLockHoldPeaks(
|
||||
opentelemetry::metrics::ObserverResult& result,
|
||||
ServiceRegistry& app) const
|
||||
{
|
||||
auto const tnPeak = app.getNodeFamily().getTreeNodeCache()->takeLockHoldPeak();
|
||||
opentelemetry::nostd::get<
|
||||
opentelemetry::nostd::shared_ptr<opentelemetry::metrics::ObserverResultT<double>>>(result)
|
||||
->Observe(
|
||||
static_cast<double>(
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(tnPeak).count()),
|
||||
{{label::metric, lval::cache_metrics::treenodeLockHoldPeakUs}});
|
||||
|
||||
auto const fbPeak = app.getNodeFamily().getFullBelowCache()->takeLockHoldPeak();
|
||||
opentelemetry::nostd::get<
|
||||
opentelemetry::nostd::shared_ptr<opentelemetry::metrics::ObserverResultT<double>>>(result)
|
||||
->Observe(
|
||||
static_cast<double>(
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(fbPeak).count()),
|
||||
{{label::metric, lval::cache_metrics::fullbelowLockHoldPeakUs}});
|
||||
}
|
||||
|
||||
void
|
||||
MetricsRegistry::registerTxqGauge()
|
||||
{
|
||||
|
||||
@@ -260,6 +260,14 @@ namespace telemetry {
|
||||
* - Adding a new OBSERVABLE gauge still requires eager central
|
||||
* registration -- pull-model instruments cannot be lazily created.
|
||||
*/
|
||||
/**
|
||||
* Run time at which a finished job counts as a stall, in microseconds.
|
||||
* Equal to LoadMonitor's 1 s warn threshold (LoadMonitor.cpp
|
||||
* addLoadSample) so this counter and the "Job: ... run:" log line
|
||||
* describe the same event.
|
||||
*/
|
||||
inline constexpr std::int64_t kJobStallThresholdUs = 1'000'000;
|
||||
|
||||
class MetricsRegistry
|
||||
{
|
||||
public:
|
||||
@@ -1016,6 +1024,11 @@ private:
|
||||
* Counter: job_finished_total{job_type="<name>",handler="<name>"}
|
||||
*/
|
||||
opentelemetry::nostd::unique_ptr<opentelemetry::metrics::Counter<uint64_t>> jobFinishedCounter_;
|
||||
/**
|
||||
* Counter: jobq_stall_total{job_type="<name>"} — one per finished job
|
||||
* whose run time reached kJobStallThresholdUs.
|
||||
*/
|
||||
opentelemetry::nostd::unique_ptr<opentelemetry::metrics::Counter<uint64_t>> jobStallCounter_;
|
||||
/**
|
||||
* Histogram: job_queued_us{job_type="<name>",handler="<name>"}
|
||||
*/
|
||||
@@ -1279,6 +1292,14 @@ private:
|
||||
registerJqTransOverflowCounter(); // gap-fill: overlay overflow total
|
||||
void
|
||||
registerCacheHitRateGauge();
|
||||
/**
|
||||
* Observe the two TaggedCache lock-hold peaks onto the cache_metrics
|
||||
* gauge. Split out to keep registerCacheHitRateGauge's callback under
|
||||
* the 80-line limit.
|
||||
*/
|
||||
void
|
||||
observeCacheLockHoldPeaks(opentelemetry::metrics::ObserverResult& result, ServiceRegistry& app)
|
||||
const;
|
||||
void
|
||||
registerTxqGauge();
|
||||
void
|
||||
|
||||
Reference in New Issue
Block a user