|
|
|
|
@@ -84,9 +84,9 @@ namespace {
|
|
|
|
|
// referenced twice — once to register the explicit-bucket view and once
|
|
|
|
|
// to create the instrument — so they are named constants to keep the two
|
|
|
|
|
// sites in sync (a mismatch would silently drop the bucket override).
|
|
|
|
|
constexpr char kJobQueuedDurationUs[] = "xrpld_job_queued_duration_us";
|
|
|
|
|
constexpr char kJobRunningDurationUs[] = "xrpld_job_running_duration_us";
|
|
|
|
|
constexpr char kRpcMethodDurationUs[] = "xrpld_rpc_method_duration_us";
|
|
|
|
|
constexpr char kJobQueuedDurationUs[] = "job_queued_us";
|
|
|
|
|
constexpr char kJobRunningDurationUs[] = "job_running_us";
|
|
|
|
|
constexpr char kRpcMethodDurationUs[] = "rpc_method_us";
|
|
|
|
|
|
|
|
|
|
/** Register an explicit-bucket histogram view for a microsecond-valued
|
|
|
|
|
* instrument.
|
|
|
|
|
@@ -98,7 +98,7 @@ constexpr char kRpcMethodDurationUs[] = "xrpld_rpc_method_duration_us";
|
|
|
|
|
* 60 s to capture the real distribution.
|
|
|
|
|
*
|
|
|
|
|
* @param views The registry to add the view to.
|
|
|
|
|
* @param name Instrument name to match (e.g. "xrpld_job_running_duration_us").
|
|
|
|
|
* @param name Instrument name to match (e.g. "job_running_us").
|
|
|
|
|
*/
|
|
|
|
|
void
|
|
|
|
|
addMicrosecondHistogramView(metric_sdk::ViewRegistry& views, std::string const& name)
|
|
|
|
|
@@ -227,45 +227,41 @@ void
|
|
|
|
|
MetricsRegistry::initSyncInstruments()
|
|
|
|
|
{
|
|
|
|
|
// RPC per-method counters and histogram.
|
|
|
|
|
rpcStartedCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_rpc_method_started_total", "Total RPC method calls started");
|
|
|
|
|
rpcStartedCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("rpc_method_started_total", "Total RPC method calls started");
|
|
|
|
|
rpcFinishedCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_rpc_method_finished_total", "Total RPC method calls completed successfully");
|
|
|
|
|
"rpc_method_finished_total", "Total RPC method calls completed successfully");
|
|
|
|
|
rpcErroredCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_rpc_method_errored_total", "Total RPC method calls that errored");
|
|
|
|
|
"rpc_method_errored_total", "Total RPC method calls that errored");
|
|
|
|
|
rpcDurationHistogram_ = meter_->CreateDoubleHistogram(
|
|
|
|
|
kRpcMethodDurationUs, "RPC method execution time in microseconds");
|
|
|
|
|
|
|
|
|
|
// Job queue per-type counters and histograms.
|
|
|
|
|
jobQueuedCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("xrpld_job_queued_total", "Total jobs enqueued");
|
|
|
|
|
jobStartedCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("xrpld_job_started_total", "Total jobs started");
|
|
|
|
|
jobFinishedCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("xrpld_job_finished_total", "Total jobs completed");
|
|
|
|
|
jobQueuedCounter_ = meter_->CreateUInt64Counter("job_queued_total", "Total jobs enqueued");
|
|
|
|
|
jobStartedCounter_ = meter_->CreateUInt64Counter("job_started_total", "Total jobs started");
|
|
|
|
|
jobFinishedCounter_ = meter_->CreateUInt64Counter("job_finished_total", "Total jobs completed");
|
|
|
|
|
jobQueuedDurationHistogram_ = meter_->CreateDoubleHistogram(
|
|
|
|
|
kJobQueuedDurationUs, "Time jobs spent waiting in the queue (microseconds)");
|
|
|
|
|
jobRunningDurationHistogram_ =
|
|
|
|
|
meter_->CreateDoubleHistogram(kJobRunningDurationUs, "Job execution time in microseconds");
|
|
|
|
|
|
|
|
|
|
// --- External dashboard parity counters (Task 7.14) ---
|
|
|
|
|
ledgersClosedCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_ledgers_closed_total", "Total ledgers closed by consensus");
|
|
|
|
|
ledgersClosedCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("ledgers_closed_total", "Total ledgers closed by consensus");
|
|
|
|
|
validationsSentCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_validations_sent_total", "Total validations sent by this node");
|
|
|
|
|
"validations_sent_total", "Total validations sent by this node");
|
|
|
|
|
validationsCheckedCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_validations_checked_total", "Total network validations received and checked");
|
|
|
|
|
"validations_checked_total", "Total network validations received and checked");
|
|
|
|
|
stateChangesCounter_ =
|
|
|
|
|
meter_->CreateUInt64Counter("xrpld_state_changes_total", "Total operating mode changes");
|
|
|
|
|
meter_->CreateUInt64Counter("state_changes_total", "Total operating mode changes");
|
|
|
|
|
jqTransOverflowCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_jq_trans_overflow_total", "Total job queue transaction overflows");
|
|
|
|
|
"jq_trans_overflow_total", "Total job queue transaction overflows");
|
|
|
|
|
ledgerHistoryMismatchCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_ledger_history_mismatch_total",
|
|
|
|
|
"Total built-vs-validated ledger mismatches by reason");
|
|
|
|
|
"ledger_history_mismatch_total", "Total built-vs-validated ledger mismatches by reason");
|
|
|
|
|
txqExpiredCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_txq_expired_total", "Total transactions expired out of the transaction queue");
|
|
|
|
|
"txq_expired_total", "Total transactions expired out of the transaction queue");
|
|
|
|
|
txqDroppedCounter_ = meter_->CreateUInt64Counter(
|
|
|
|
|
"xrpld_txq_dropped_total", "Total transactions refused admission to the queue by reason");
|
|
|
|
|
"txq_dropped_total", "Total transactions refused admission to the queue by reason");
|
|
|
|
|
// Note: xrpld_validation_agreements_total / xrpld_validation_missed_total
|
|
|
|
|
// are monotonic ObservableCounters created in registerValidationTotalsCounters()
|
|
|
|
|
// (below), observed from ValidationTracker's gross lifetime tallies.
|
|
|
|
|
@@ -464,7 +460,7 @@ MetricsRegistry::registerCacheHitRateGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.2: Cache hit rate and size gauges ---
|
|
|
|
|
cacheHitRateGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("xrpld_cache_metrics", "Cache hit rates and sizes");
|
|
|
|
|
meter_->CreateDoubleObservableGauge("cache_metrics", "Cache hit rates and sizes");
|
|
|
|
|
cacheHitRateGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -534,8 +530,7 @@ void
|
|
|
|
|
MetricsRegistry::registerTxqGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.3: TxQ metrics gauges ---
|
|
|
|
|
txqGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("xrpld_txq_metrics", "Transaction queue metrics");
|
|
|
|
|
txqGauge_ = meter_->CreateDoubleObservableGauge("txq_metrics", "Transaction queue metrics");
|
|
|
|
|
txqGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -583,7 +578,7 @@ MetricsRegistry::registerObjectCountGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.6: Counted object instance gauges ---
|
|
|
|
|
objectCountGauge_ = meter_->CreateInt64ObservableGauge(
|
|
|
|
|
"xrpld_object_count", "Live instance counts for key internal object types");
|
|
|
|
|
"object_count", "Live instance counts for key internal object types");
|
|
|
|
|
objectCountGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -614,8 +609,8 @@ void
|
|
|
|
|
MetricsRegistry::registerLoadFactorGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.7: Load factor breakdown gauges ---
|
|
|
|
|
loadFactorGauge_ = meter_->CreateDoubleObservableGauge(
|
|
|
|
|
"xrpld_load_factor_metrics", "Fee load factor breakdown");
|
|
|
|
|
loadFactorGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("load_factor_metrics", "Fee load factor breakdown");
|
|
|
|
|
loadFactorGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -692,7 +687,7 @@ MetricsRegistry::registerNodeStoreGauge()
|
|
|
|
|
// libxrpl nodestore code — the MetricsRegistry reads the existing atomic
|
|
|
|
|
// counters from Database via its public accessors.
|
|
|
|
|
nodeStoreGauge_ = meter_->CreateInt64ObservableGauge(
|
|
|
|
|
"xrpld_nodestore_state", "NodeStore I/O counters, queue depth, and write load");
|
|
|
|
|
"nodestore_state", "NodeStore I/O counters, queue depth, and write load");
|
|
|
|
|
nodeStoreGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -771,7 +766,7 @@ MetricsRegistry::registerServerInfoGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.7a: Server info gauges ---
|
|
|
|
|
serverInfoGauge_ =
|
|
|
|
|
meter_->CreateInt64ObservableGauge("xrpld_server_info", "Server-level health metrics");
|
|
|
|
|
meter_->CreateInt64ObservableGauge("server_info", "Server-level health metrics");
|
|
|
|
|
serverInfoGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -839,8 +834,7 @@ void
|
|
|
|
|
MetricsRegistry::registerBuildInfoGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.7b: Build info gauge ---
|
|
|
|
|
buildInfoGauge_ =
|
|
|
|
|
meter_->CreateInt64ObservableGauge("xrpld_build_info", "Build version information");
|
|
|
|
|
buildInfoGauge_ = meter_->CreateInt64ObservableGauge("build_info", "Build version information");
|
|
|
|
|
buildInfoGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* /* state */) {
|
|
|
|
|
try
|
|
|
|
|
@@ -861,7 +855,7 @@ MetricsRegistry::registerCompleteLedgersGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.7c: Complete ledgers range gauge ---
|
|
|
|
|
completeLedgersGauge_ = meter_->CreateInt64ObservableGauge(
|
|
|
|
|
"xrpld_complete_ledgers", "Complete ledger range start/end pairs");
|
|
|
|
|
"complete_ledgers", "Complete ledger range start/end pairs");
|
|
|
|
|
completeLedgersGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -919,8 +913,8 @@ void
|
|
|
|
|
MetricsRegistry::registerDbMetricsGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 9.7d: Database size and fetch rate gauges ---
|
|
|
|
|
dbMetricsGauge_ = meter_->CreateInt64ObservableGauge(
|
|
|
|
|
"xrpld_db_metrics", "Database storage sizes and fetch rates");
|
|
|
|
|
dbMetricsGauge_ =
|
|
|
|
|
meter_->CreateInt64ObservableGauge("db_metrics", "Database storage sizes and fetch rates");
|
|
|
|
|
dbMetricsGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -958,8 +952,8 @@ void
|
|
|
|
|
MetricsRegistry::registerValidatorHealthGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 7.9: Validator health gauges ---
|
|
|
|
|
validatorHealthGauge_ = meter_->CreateDoubleObservableGauge(
|
|
|
|
|
"xrpld_validator_health", "Validator health indicators");
|
|
|
|
|
validatorHealthGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("validator_health", "Validator health indicators");
|
|
|
|
|
validatorHealthGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1008,7 +1002,7 @@ MetricsRegistry::registerPeerQualityGauge()
|
|
|
|
|
// Uses Peer::json() to read latency and version since those accessors
|
|
|
|
|
// are not on the abstract Peer interface (they live on PeerImp).
|
|
|
|
|
peerQualityGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("xrpld_peer_quality", "Peer network quality metrics");
|
|
|
|
|
meter_->CreateDoubleObservableGauge("peer_quality", "Peer network quality metrics");
|
|
|
|
|
peerQualityGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1113,7 +1107,7 @@ MetricsRegistry::registerReduceRelayGauge()
|
|
|
|
|
// feature is saving bandwidth; a high not_enabled count means stale peers
|
|
|
|
|
// force full relay.
|
|
|
|
|
reduceRelayGauge_ = meter_->CreateInt64ObservableGauge(
|
|
|
|
|
"xrpld_reduce_relay_metrics", "Transaction reduce-relay efficiency metrics");
|
|
|
|
|
"reduce_relay_metrics", "Transaction reduce-relay efficiency metrics");
|
|
|
|
|
reduceRelayGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1159,8 +1153,8 @@ void
|
|
|
|
|
MetricsRegistry::registerLedgerEconomyGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 7.11: Ledger economy gauges ---
|
|
|
|
|
ledgerEconomyGauge_ = meter_->CreateDoubleObservableGauge(
|
|
|
|
|
"xrpld_ledger_economy", "Ledger fee and economy metrics");
|
|
|
|
|
ledgerEconomyGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("ledger_economy", "Ledger fee and economy metrics");
|
|
|
|
|
ledgerEconomyGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1224,7 +1218,7 @@ MetricsRegistry::registerStateTrackingGauge()
|
|
|
|
|
{
|
|
|
|
|
// --- Task 7.12: State tracking gauges ---
|
|
|
|
|
stateTrackingGauge_ =
|
|
|
|
|
meter_->CreateDoubleObservableGauge("xrpld_state_tracking", "Node state and mode tracking");
|
|
|
|
|
meter_->CreateDoubleObservableGauge("state_tracking", "Node state and mode tracking");
|
|
|
|
|
stateTrackingGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1279,7 +1273,7 @@ MetricsRegistry::registerStorageDetailGauge()
|
|
|
|
|
// --- Task 7.13: Storage detail gauges ---
|
|
|
|
|
// Reports NuDB on-disk size via the NodeStore JSON counters interface.
|
|
|
|
|
storageDetailGauge_ =
|
|
|
|
|
meter_->CreateInt64ObservableGauge("xrpld_storage_detail", "Storage detail metrics");
|
|
|
|
|
meter_->CreateInt64ObservableGauge("storage_detail", "Storage detail metrics");
|
|
|
|
|
storageDetailGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1317,8 +1311,7 @@ MetricsRegistry::registerValidationAgreementGauge()
|
|
|
|
|
// window data is read (the callback fires every ~10 s from the
|
|
|
|
|
// PeriodicExportingMetricReader thread).
|
|
|
|
|
validationAgreementGauge_ = meter_->CreateDoubleObservableGauge(
|
|
|
|
|
"xrpld_validation_agreement",
|
|
|
|
|
"Validation agreement percentages and counts (1h/24h windows)");
|
|
|
|
|
"validation_agreement", "Validation agreement percentages and counts (1h/24h windows)");
|
|
|
|
|
validationAgreementGauge_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
@@ -1377,7 +1370,7 @@ MetricsRegistry::registerValidationTotalsCounters()
|
|
|
|
|
// tallies are read; the callback fires every ~10 s from the
|
|
|
|
|
// PeriodicExportingMetricReader thread.
|
|
|
|
|
validationAgreementsObservable_ = meter_->CreateInt64ObservableCounter(
|
|
|
|
|
"xrpld_validation_agreements_total",
|
|
|
|
|
"validation_agreements_total",
|
|
|
|
|
"Lifetime validations that initially agreed with network consensus");
|
|
|
|
|
validationAgreementsObservable_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
@@ -1399,8 +1392,7 @@ MetricsRegistry::registerValidationTotalsCounters()
|
|
|
|
|
this);
|
|
|
|
|
|
|
|
|
|
validationMissedObservable_ = meter_->CreateInt64ObservableCounter(
|
|
|
|
|
"xrpld_validation_missed_total",
|
|
|
|
|
"Lifetime validations that initially missed network consensus");
|
|
|
|
|
"validation_missed_total", "Lifetime validations that initially missed network consensus");
|
|
|
|
|
validationMissedObservable_->AddCallback(
|
|
|
|
|
[](opentelemetry::metrics::ObserverResult result, void* state) {
|
|
|
|
|
auto* self = static_cast<MetricsRegistry*>(state);
|
|
|
|
|
|