feat(telemetry): count job stalls and expose cache lock-hold peaks

Two new signals for the rotation-stall proof chain:

- `jobq_stall_total{job_type="<name>"}` — one increment per finished job whose
  running duration reached `kJobStallThresholdUs` (1 s, LoadMonitor's own
  warn bar). A process-wide freeze shows up as several job types crossing
  the bar in the same second. One `int64` compare on the finish hook path.

- `cache_metrics{metric="treenode_lock_hold_peak_us"}` and
  `{metric="fullbelow_lock_hold_peak_us"}` — observed once per collect tick
  from `TaggedCache::takeLockHoldPeak()`, so the tick reads the longest hold
  since the previous tick and resets. Two atomic exchanges per tick.

The two Observe calls live in a new `observeCacheLockHoldPeaks` helper so
`registerCacheHitRateGauge`'s callback stays inside the 80-line limit.
This commit is contained in:
Pratik Mankawde
2026-09-14 20:39:55 +01:00
parent 70766746f3
commit ee730ee4e1
2 changed files with 53 additions and 0 deletions

View File

@@ -504,6 +504,8 @@ MetricsRegistry::initSyncInstruments()
jobQueuedCounter_ = meter_->CreateUInt64Counter("job_queued_total", "Total jobs enqueued");
jobStartedCounter_ = meter_->CreateUInt64Counter("job_started_total", "Total jobs started");
jobFinishedCounter_ = meter_->CreateUInt64Counter("job_finished_total", "Total jobs completed");
jobStallCounter_ = meter_->CreateUInt64Counter(
metric::jobqStallTotal, "Jobs whose run time reached the 1 s stall threshold");
jobQueuedDurationHistogram_ = meter_->CreateDoubleHistogram(
kJobQueuedDurationUs, "Time jobs spent waiting in the queue (microseconds)");
jobRunningDurationHistogram_ =
@@ -686,6 +688,10 @@ MetricsRegistry::recordJobFinished(
{{label::jobType, std::string(jobType)}, {label::handler, handler}},
opentelemetry::context::Context{});
}
// One compare per job finish. A process-wide freeze shows up here as
// several job types crossing the bar in the same second.
if (runningDurUs >= kJobStallThresholdUs && jobStallCounter_)
jobStallCounter_->Add(1, {{label::jobType, std::string(jobType)}});
#endif
}
@@ -829,6 +835,10 @@ MetricsRegistry::registerCacheHitRateGauge()
opentelemetry::nostd::get<opentelemetry::nostd::shared_ptr<
opentelemetry::metrics::ObserverResultT<double>>>(result)
->Observe(static_cast<double>(alSize), {{label::metric, "AL_size"}});
// Longest TaggedCache mutex hold since the last tick.
// Split out to keep this callback under the 80-line limit.
self->observeCacheLockHoldPeaks(result, app);
}
catch (...) // NOLINT(bugprone-empty-catch)
{
@@ -838,6 +848,28 @@ MetricsRegistry::registerCacheHitRateGauge()
this);
}
void
MetricsRegistry::observeCacheLockHoldPeaks(
opentelemetry::metrics::ObserverResult& result,
ServiceRegistry& app) const
{
auto const tnPeak = app.getNodeFamily().getTreeNodeCache()->takeLockHoldPeak();
opentelemetry::nostd::get<
opentelemetry::nostd::shared_ptr<opentelemetry::metrics::ObserverResultT<double>>>(result)
->Observe(
static_cast<double>(
std::chrono::duration_cast<std::chrono::microseconds>(tnPeak).count()),
{{label::metric, lval::cache_metrics::treenodeLockHoldPeakUs}});
auto const fbPeak = app.getNodeFamily().getFullBelowCache()->takeLockHoldPeak();
opentelemetry::nostd::get<
opentelemetry::nostd::shared_ptr<opentelemetry::metrics::ObserverResultT<double>>>(result)
->Observe(
static_cast<double>(
std::chrono::duration_cast<std::chrono::microseconds>(fbPeak).count()),
{{label::metric, lval::cache_metrics::fullbelowLockHoldPeakUs}});
}
void
MetricsRegistry::registerTxqGauge()
{

View File

@@ -260,6 +260,14 @@ namespace telemetry {
* - Adding a new OBSERVABLE gauge still requires eager central
* registration -- pull-model instruments cannot be lazily created.
*/
/**
* Run time at which a finished job counts as a stall, in microseconds.
* Equal to LoadMonitor's 1 s warn threshold (LoadMonitor.cpp
* addLoadSample) so this counter and the "Job: ... run:" log line
* describe the same event.
*/
inline constexpr std::int64_t kJobStallThresholdUs = 1'000'000;
class MetricsRegistry
{
public:
@@ -1016,6 +1024,11 @@ private:
* Counter: job_finished_total{job_type="<name>",handler="<name>"}
*/
opentelemetry::nostd::unique_ptr<opentelemetry::metrics::Counter<uint64_t>> jobFinishedCounter_;
/**
* Counter: jobq_stall_total{job_type="<name>"} — one per finished job
* whose run time reached kJobStallThresholdUs.
*/
opentelemetry::nostd::unique_ptr<opentelemetry::metrics::Counter<uint64_t>> jobStallCounter_;
/**
* Histogram: job_queued_us{job_type="<name>",handler="<name>"}
*/
@@ -1279,6 +1292,14 @@ private:
registerJqTransOverflowCounter(); // gap-fill: overlay overflow total
void
registerCacheHitRateGauge();
/**
* Observe the two TaggedCache lock-hold peaks onto the cache_metrics
* gauge. Split out to keep registerCacheHitRateGauge's callback under
* the 80-line limit.
*/
void
observeCacheLockHoldPeaks(opentelemetry::metrics::ObserverResult& result, ServiceRegistry& app)
const;
void
registerTxqGauge();
void