Merge branch 'pratik/otel-phase7-native-metrics' into pratik/otel-phase8-log-correlation

This commit is contained in:
Pratik Mankawde
2026-08-07 11:54:57 +01:00
7 changed files with 725 additions and 725 deletions

View File

@@ -87,6 +87,43 @@
"panels": [],
"id": 3
},
{
"title": "Validation Send Rate [$xrpl_network_type]",
"description": "**What:** How often this node issues a validation confirming it has fully validated a ledger.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Should closely track the ledger close rate on a healthy validator.\n**Healthy range:** About one per ledger (~0.25/s on mainnet).\n**Watch for:** Validations lagging closes suggests the node is falling behind on validation.\n**Source:** src/xrpld/app/consensus/RCLConsensus.cpp:RCLConsensus::Adaptor::createValidationSpan",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 10
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", consensus_mode=~\"$consensus_mode\", span_name=\"consensus.validation.send\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Validations / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: validations/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 4
},
{
"title": "Consensus Proposals Sent Rate",
"description": "**What:** How often this node broadcasts its own consensus proposal (its candidate transaction set) to the network.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** A proposing validator emits a steady stream, roughly one per consensus round; observing-only nodes emit none.\n**Healthy range:** About one proposal per round (~0.2-0.3/s on mainnet) for a proposing node.\n**Watch for:** A validator that drops to zero has stopped proposing and may be unhealthy or misconfigured.\n**Source:** src/xrpld/app/consensus/RCLConsensus.cpp:RCLConsensus::Adaptor::propose",
@@ -95,7 +132,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 10
"y": 14
},
"options": {
"tooltip": {
@@ -130,7 +167,7 @@
},
"overrides": []
},
"id": 4
"id": 5
},
{
"title": "Ledger Close Duration",
@@ -140,7 +177,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 10
"y": 14
},
"options": {
"tooltip": {
@@ -175,43 +212,6 @@
},
"overrides": []
},
"id": 5
},
{
"title": "Validation Send Rate [$xrpl_network_type]",
"description": "**What:** How often this node issues a validation confirming it has fully validated a ledger.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Should closely track the ledger close rate on a healthy validator.\n**Healthy range:** About one per ledger (~0.25/s on mainnet).\n**Watch for:** Validations lagging closes suggests the node is falling behind on validation.\n**Source:** src/xrpld/app/consensus/RCLConsensus.cpp:RCLConsensus::Adaptor::createValidationSpan",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 18
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", consensus_mode=~\"$consensus_mode\", span_name=\"consensus.validation.send\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Validations / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: validations/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 6
},
{

View File

@@ -9,6 +9,67 @@
"id": null,
"links": [],
"panels": [
{
"title": "Overlay Traffic Heatmap (All Categories, Bytes In) [$xrpl_network_type]",
"description": "###### What this is:\n*All overlay traffic categories ranked by inbound bytes, giving an at-a-glance view of which message types consume the most receive bandwidth.*\n\n###### How it's computed:\n*Top categories by latest inbound byte value across all traffic categories. Each bar is labelled with its traffic category followed by the node identity; the shared `_bytes_in` suffix is dropped from the category name because the panel already reports inbound bytes.*\n\n###### Reading it:\n*The longest bars are the biggest bandwidth consumers; on a synced node transactions, proposals, and validations usually lead.*\n\n###### Healthy range:\n*workload-dependent.*\n\n###### Watch for:\n*A single ledger-data or fetch category dominating (ongoing sync) or an unexpected category topping the list.*\n\n###### Source:\n[OverlayImpl.cpp](https://github.com/XRPLF/rippled/blob/develop/src/xrpld/overlay/detail/OverlayImpl.cpp)\n\n###### Function:\n`OverlayImpl ctor (TrafficGauges)`",
"type": "bargauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 0
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
},
"displayMode": "gradient",
"orientation": "horizontal",
"reduceOptions": {
"calcs": ["lastNotNull"],
"fields": "",
"values": false
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "label_replace(topk(20, {service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", xrpl_work_item=~\"$xrpl_work_item\", xrpl_branch=~\"$xrpl_branch\", xrpl_node_role=~\"$xrpl_node_role\", __name__=~\".*_bytes_in\", __name__!~\"total_.*\"}), \"series\", \"$1\", \"__name__\", \"(.*)_bytes_in\")"
}
],
"fieldConfig": {
"defaults": {
"displayName": "${__field.labels.series} [${__field.labels.service_instance_id} ${__field.labels.xrpl_branch} ${__field.labels.xrpl_node_role} ${__field.labels.xrpl_work_item}]",
"unit": "decbytes",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 1048576
},
{
"color": "red",
"value": 104857600
}
]
}
},
"overrides": []
},
"id": 1,
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2
},
{
"title": "Ledger Data \u2014 Ledger",
"description": "###### What this is:\n*Inbound bytes for ledger-data message categories, split into aggregate get/share plus the transaction-set, transaction-node, and account-state-node sub-types the node receives from peers.*\n\n###### How it's computed:\n*Per-category inbound byte rate per node.*\n\n###### Reading it:\n*Normally low and flat once synced. Account-state-node traffic dominates during state sync; transaction-set-candidate traffic dominates during consensus catch-up.*\n\n###### Healthy range:\n*workload-dependent; low and steady on a synced node.*\n\n###### Watch for:\n*Sustained high account-state or tx-node inbound bytes on a node that should be caught up (repeated re-sync, missing history), or a single peer driving all traffic.*\n\n###### Source:\n[OverlayImpl.cpp](https://github.com/XRPLF/rippled/blob/develop/src/xrpld/overlay/detail/OverlayImpl.cpp)\n\n###### Function:\n`OverlayImpl ctor (TrafficGauges)`",
@@ -17,7 +78,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 0
"y": 4
},
"options": {
"tooltip": {
@@ -57,7 +118,7 @@
},
"overrides": []
},
"id": 1
"id": 2
},
{
"title": "Ledger Data \u2014 Transaction",
@@ -67,7 +128,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 0
"y": 4
},
"options": {
"tooltip": {
@@ -119,7 +180,7 @@
},
"overrides": []
},
"id": 2
"id": 3
},
{
"title": "Ledger Data \u2014 Account State",
@@ -129,7 +190,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 8
"y": 12
},
"options": {
"tooltip": {
@@ -169,7 +230,7 @@
},
"overrides": []
},
"id": 3
"id": 4
},
{
"title": "Ledger Traffic \u2014 Ledger",
@@ -179,7 +240,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 8
"y": 12
},
"options": {
"tooltip": {
@@ -219,7 +280,7 @@
},
"overrides": []
},
"id": 4
"id": 5
},
{
"title": "Ledger Traffic \u2014 Transaction",
@@ -229,7 +290,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 16
"y": 20
},
"options": {
"tooltip": {
@@ -281,7 +342,7 @@
},
"overrides": []
},
"id": 5
"id": 6
},
{
"title": "Ledger Traffic \u2014 Account State",
@@ -291,7 +352,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 16
"y": 20
},
"options": {
"tooltip": {
@@ -331,7 +392,7 @@
},
"overrides": []
},
"id": 6
"id": 7
},
{
"title": "GetObject \u2014 Ledger",
@@ -341,7 +402,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 24
"y": 28
},
"options": {
"tooltip": {
@@ -381,7 +442,7 @@
},
"overrides": []
},
"id": 7
"id": 8
},
{
"title": "GetObject \u2014 Transaction",
@@ -391,7 +452,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 24
"y": 28
},
"options": {
"tooltip": {
@@ -443,7 +504,7 @@
},
"overrides": []
},
"id": 8
"id": 9
},
{
"title": "GetObject \u2014 Account State",
@@ -453,7 +514,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 32
"y": 36
},
"options": {
"tooltip": {
@@ -493,7 +554,7 @@
},
"overrides": []
},
"id": 9
"id": 10
},
{
"title": "GetObject Messages \u2014 Ledger",
@@ -503,7 +564,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 32
"y": 36
},
"options": {
"tooltip": {
@@ -537,7 +598,7 @@
},
"overrides": []
},
"id": 10
"id": 11
},
{
"title": "GetObject Messages \u2014 Transaction",
@@ -547,7 +608,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 40
"y": 44
},
"options": {
"tooltip": {
@@ -587,7 +648,7 @@
},
"overrides": []
},
"id": 11
"id": 12
},
{
"title": "GetObject Messages \u2014 Account State",
@@ -597,7 +658,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 40
"y": 44
},
"options": {
"tooltip": {
@@ -631,7 +692,7 @@
},
"overrides": []
},
"id": 12
"id": 13
},
{
"title": "GetObject Messages \u2014 Specials",
@@ -641,7 +702,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 48
"y": 52
},
"options": {
"tooltip": {
@@ -687,7 +748,7 @@
},
"overrides": []
},
"id": 13
"id": 14
},
{
"title": "GetObject \u2014 Specials",
@@ -697,7 +758,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 48
"y": 52
},
"options": {
"tooltip": {
@@ -767,68 +828,7 @@
},
"overrides": []
},
"id": 14
},
{
"title": "Overlay Traffic Heatmap (All Categories, Bytes In) [$xrpl_network_type]",
"description": "###### What this is:\n*All overlay traffic categories ranked by inbound bytes, giving an at-a-glance view of which message types consume the most receive bandwidth.*\n\n###### How it's computed:\n*Top categories by latest inbound byte value across all traffic categories. Each bar is labelled with its traffic category followed by the node identity; the shared `_bytes_in` suffix is dropped from the category name because the panel already reports inbound bytes.*\n\n###### Reading it:\n*The longest bars are the biggest bandwidth consumers; on a synced node transactions, proposals, and validations usually lead.*\n\n###### Healthy range:\n*workload-dependent.*\n\n###### Watch for:\n*A single ledger-data or fetch category dominating (ongoing sync) or an unexpected category topping the list.*\n\n###### Source:\n[OverlayImpl.cpp](https://github.com/XRPLF/rippled/blob/develop/src/xrpld/overlay/detail/OverlayImpl.cpp)\n\n###### Function:\n`OverlayImpl ctor (TrafficGauges)`",
"type": "bargauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 56
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
},
"displayMode": "gradient",
"orientation": "horizontal",
"reduceOptions": {
"calcs": ["lastNotNull"],
"fields": "",
"values": false
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "label_replace(topk(20, {service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", xrpl_work_item=~\"$xrpl_work_item\", xrpl_branch=~\"$xrpl_branch\", xrpl_node_role=~\"$xrpl_node_role\", __name__=~\".*_bytes_in\", __name__!~\"total_.*\"}), \"series\", \"$1\", \"__name__\", \"(.*)_bytes_in\")"
}
],
"fieldConfig": {
"defaults": {
"displayName": "${__field.labels.series} [${__field.labels.service_instance_id} ${__field.labels.xrpl_branch} ${__field.labels.xrpl_node_role} ${__field.labels.xrpl_work_item}]",
"unit": "decbytes",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 1048576
},
{
"color": "red",
"value": 104857600
}
]
}
},
"overrides": []
},
"id": 15,
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2
"id": 15
},
{
"title": "Sync Diagnostics",

View File

@@ -58,6 +58,43 @@
"maxPerRow": 2,
"id": 2
},
{
"title": "Ledger Validation Rate [$xrpl_network_type]",
"description": "**What:** How often ledgers reach the trusted-validation quorum and become fully validated.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Should match the build and close rates on a healthy node.\n**Healthy range:** About 0.25/s on mainnet.\n**Watch for:** A shortfall means ledgers are not accumulating enough trusted validations to be accepted.\n**Source:** src/xrpld/app/ledger/detail/LedgerMaster.cpp:LedgerMaster::checkAccept",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 1
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", span_name=\"ledger.validate\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Validations / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: ledgers/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 3
},
{
"title": "Ledger Build Duration",
"description": "**What:** Time to build a full ledger, including applying transactions, writing state, and finalising acceptance.\n**How it's computed:** 95th and 50th percentile of per-build durations over a 5-minute window, per node.\n**Reading it:** Scales with transaction volume and disk I/O.\n**Healthy range:** Tens to a few hundred milliseconds under typical load.\n**Watch for:** Long build times indicate expensive transaction sets or I/O pressure while flushing state.\n**Source:** src/xrpld/app/ledger/detail/BuildLedger.cpp:buildLedgerImpl",
@@ -109,43 +146,6 @@
},
"overrides": []
},
"id": 3
},
{
"title": "Ledger Validation Rate [$xrpl_network_type]",
"description": "**What:** How often ledgers reach the trusted-validation quorum and become fully validated.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Should match the build and close rates on a healthy node.\n**Healthy range:** About 0.25/s on mainnet.\n**Watch for:** A shortfall means ledgers are not accumulating enough trusted validations to be accepted.\n**Source:** src/xrpld/app/ledger/detail/LedgerMaster.cpp:LedgerMaster::checkAccept",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 13
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", span_name=\"ledger.validate\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Validations / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: ledgers/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 4
},
{
@@ -155,8 +155,8 @@
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 17
"x": 12,
"y": 5
},
"options": {
"tooltip": {
@@ -197,7 +197,7 @@
"h": 1,
"w": 24,
"x": 0,
"y": 25
"y": 13
},
"collapsed": false,
"panels": [],
@@ -211,7 +211,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 26
"y": 14
},
"options": {
"tooltip": {
@@ -264,7 +264,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 26
"y": 14
},
"options": {
"tooltip": {
@@ -308,7 +308,7 @@
"h": 1,
"w": 24,
"x": 0,
"y": 34
"y": 22
},
"collapsed": false,
"panels": [],
@@ -322,7 +322,7 @@
"h": 4,
"w": 12,
"x": 0,
"y": 35
"y": 23
},
"options": {
"tooltip": {
@@ -359,7 +359,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 39
"y": 27
},
"options": {
"tooltip": {

View File

@@ -113,6 +113,90 @@
"maxPerRow": 2,
"id": 2
},
{
"title": "Ledger Fetch Rate [$xrpl_network_type]",
"description": "**What:** Rate at which the node requests ledgers from peers.\n**How it's computed:** Per-second rate of the ledger-fetches counter over 5 minutes.\n**Reading it:** Near zero when synced; rises when catching up or backfilling history.\n**Healthy range:** workload-dependent; low on a synced node.\n**Watch for:** Sustained high fetch rate on a node that should be current (repeatedly missing ledgers).\n**Source:** src/xrpld/app/ledger/detail/InboundLedgers.cpp InboundLedgersImp ctor",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 4
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "rate(ledger_fetches_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[$__rate_interval])",
"legendFormat": "Fetches / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: fetches/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 3
},
{
"title": "Ledger History Mismatches [$xrpl_network_type]",
"description": "**What:** Rate of ledger-history hash mismatches: built ledgers whose hash disagrees with the validated hash.\n**How it's computed:** Per-second rate of the history-mismatch counter over 5 minutes.\n**Reading it:** Should be flat at zero.\n**Healthy range:** zero.\n**Watch for:** Any non-zero value (consensus divergence, corrupted history, or database issues).\n**Source:** src/xrpld/telemetry/MetricsRegistry.cpp incrementLedgerHistoryMismatch (caller LedgerHistory.cpp)",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 4
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "rate(ledger_history_mismatch_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[$__rate_interval])",
"legendFormat": "Mismatches / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: mismatches/s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 0.01
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 4
},
{
"title": "Operating Mode (Time Share)",
"description": "**What:** Fraction of time in each operating mode (per-interval): Disconnected, Connected, Syncing, Tracking, Full.\n**How it's computed:** Per-mode duration rate divided by the sum of all five mode rates over the interval; the five shares stack to 100%.\n**Reading it:** Full should fill almost the entire band on a healthy node.\n**Healthy range:** Full near 100%; other modes near 0.\n**Watch for:** A rising share of Syncing, Connected, or Disconnected (instability or repeated resync).\n**Source:** src/xrpld/app/misc/NetworkOPs.cpp NetworkOPsImp::Stats ctor",
@@ -121,7 +205,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 4
"y": 8
},
"options": {
"tooltip": {
@@ -186,7 +270,7 @@
},
"overrides": []
},
"id": 3
"id": 5
},
{
"title": "Operating Mode Transitions",
@@ -196,7 +280,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 4
"y": 8
},
"options": {
"tooltip": {
@@ -258,7 +342,7 @@
},
"overrides": []
},
"id": 4
"id": 6
},
{
"title": "I/O Latency",
@@ -268,7 +352,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 12
"y": 16
},
"options": {
"tooltip": {
@@ -309,7 +393,7 @@
},
"overrides": []
},
"id": 5
"id": 7
},
{
"title": "Job Queue Depth",
@@ -319,7 +403,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 12
"y": 16
},
"options": {
"tooltip": {
@@ -354,90 +438,6 @@
},
"overrides": []
},
"id": 6
},
{
"title": "Ledger Fetch Rate [$xrpl_network_type]",
"description": "**What:** Rate at which the node requests ledgers from peers.\n**How it's computed:** Per-second rate of the ledger-fetches counter over 5 minutes.\n**Reading it:** Near zero when synced; rises when catching up or backfilling history.\n**Healthy range:** workload-dependent; low on a synced node.\n**Watch for:** Sustained high fetch rate on a node that should be current (repeatedly missing ledgers).\n**Source:** src/xrpld/app/ledger/detail/InboundLedgers.cpp InboundLedgersImp ctor",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 20
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "rate(ledger_fetches_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[$__rate_interval])",
"legendFormat": "Fetches / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: fetches/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 7
},
{
"title": "Ledger History Mismatches [$xrpl_network_type]",
"description": "**What:** Rate of ledger-history hash mismatches: built ledgers whose hash disagrees with the validated hash.\n**How it's computed:** Per-second rate of the history-mismatch counter over 5 minutes.\n**Reading it:** Should be flat at zero.\n**Healthy range:** zero.\n**Watch for:** Any non-zero value (consensus divergence, corrupted history, or database issues).\n**Source:** src/xrpld/telemetry/MetricsRegistry.cpp incrementLedgerHistoryMismatch (caller LedgerHistory.cpp)",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 20
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "rate(ledger_history_mismatch_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[$__rate_interval])",
"legendFormat": "Mismatches / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: mismatches/s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 0.01
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 8
},
{
@@ -453,6 +453,112 @@
"panels": [],
"id": 9
},
{
"title": "FullBelowCache Hit Rate [$xrpl_network_type]",
"description": "**What:** Hit-rate percentage of the FullBelowCache.\n**How it's computed:** Current value of the cache hit-rate gauge (0-100%).\n**Reading it:** Higher is better; means the node reuses knowledge of complete subtrees.\n**Healthy range:** above ~50% in steady state.\n**Watch for:** Low hit rate during steady state (cache too small or constant re-acquisition).\n**Source:** include/xrpl/basics/TaggedCache.h TaggedCache::Stats ctor",
"type": "gauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 25
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "node_family_full_below_cache_hit_rate{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}",
"legendFormat": "Hit Rate"
}
],
"fieldConfig": {
"defaults": {
"unit": "percent",
"min": 0,
"max": 100,
"thresholds": {
"steps": [
{
"color": "red",
"value": null
},
{
"color": "yellow",
"value": 25
},
{
"color": "green",
"value": 50
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 10
},
{
"title": "Ledger Publish Gap [$xrpl_network_type]",
"description": "**What:** Difference between published and validated ledger ages: how far the publish pipeline trails validation.\n**How it's computed:** Published-ledger age minus validated-ledger age, in seconds.\n**Reading it:** Near zero means publishing keeps up with validation.\n**Healthy range:** within a few seconds of zero.\n**Watch for:** A growing gap (publish backlog, stale data for subscribers).\n**Source:** src/xrpld/app/ledger/LedgerMaster.h LedgerMaster::Stats ctor",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 25
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "ledgermaster_published_ledger_age{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"} - ledgermaster_validated_ledger_age{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}",
"legendFormat": "Publish Gap"
}
],
"fieldConfig": {
"defaults": {
"unit": "s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 5
},
{
"color": "red",
"value": 10
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 11
},
{
"title": "Key Jobs Execution Time",
"description": "**What:** Execution time of the most important job types (accept/advance ledger, transaction, write objects, heartbeat, sweep, trusted validation/proposal, publish, client RPC, ledger data) at the selected quantile.\n**How it's computed:** Selected quantile of each job's execution-time histogram over 5 minutes.\n**Reading it:** Lower and stable is better; shows where consensus, transaction, and maintenance time goes.\n**Healthy range:** workload-dependent; most jobs in low tens of milliseconds.\n**Watch for:** One job type spiking (a specific bottleneck) or broad increases (overload).\n**Source:** include/xrpl/core/JobTypeData.h JobTypeData ctor",
@@ -461,7 +567,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 25
"y": 29
},
"options": {
"tooltip": {
@@ -565,7 +671,7 @@
},
"overrides": []
},
"id": 10
"id": 12
},
{
"title": "Key Jobs Dequeue Wait Time",
@@ -575,7 +681,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 25
"y": 29
},
"options": {
"tooltip": {
@@ -679,7 +785,7 @@
},
"overrides": []
},
"id": 11
"id": 13
},
{
"title": "FullBelowCache Size",
@@ -689,7 +795,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 33
"y": 37
},
"options": {
"tooltip": {
@@ -723,112 +829,6 @@
},
"overrides": []
},
"id": 12
},
{
"title": "FullBelowCache Hit Rate [$xrpl_network_type]",
"description": "**What:** Hit-rate percentage of the FullBelowCache.\n**How it's computed:** Current value of the cache hit-rate gauge (0-100%).\n**Reading it:** Higher is better; means the node reuses knowledge of complete subtrees.\n**Healthy range:** above ~50% in steady state.\n**Watch for:** Low hit rate during steady state (cache too small or constant re-acquisition).\n**Source:** include/xrpl/basics/TaggedCache.h TaggedCache::Stats ctor",
"type": "gauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 41
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "node_family_full_below_cache_hit_rate{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}",
"legendFormat": "Hit Rate"
}
],
"fieldConfig": {
"defaults": {
"unit": "percent",
"min": 0,
"max": 100,
"thresholds": {
"steps": [
{
"color": "red",
"value": null
},
{
"color": "yellow",
"value": 25
},
{
"color": "green",
"value": 50
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 13
},
{
"title": "Ledger Publish Gap [$xrpl_network_type]",
"description": "**What:** Difference between published and validated ledger ages: how far the publish pipeline trails validation.\n**How it's computed:** Published-ledger age minus validated-ledger age, in seconds.\n**Reading it:** Near zero means publishing keeps up with validation.\n**Healthy range:** within a few seconds of zero.\n**Watch for:** A growing gap (publish backlog, stale data for subscribers).\n**Source:** src/xrpld/app/ledger/LedgerMaster.h LedgerMaster::Stats ctor",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 41
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "ledgermaster_published_ledger_age{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"} - ledgermaster_validated_ledger_age{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}",
"legendFormat": "Publish Gap"
}
],
"fieldConfig": {
"defaults": {
"unit": "s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 5
},
{
"color": "red",
"value": 10
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 14
},
{
@@ -838,8 +838,8 @@
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 45
"x": 12,
"y": 37
},
"options": {
"tooltip": {
@@ -889,7 +889,7 @@
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"x": 0,
"y": 45
},
"options": {
@@ -933,8 +933,8 @@
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 53
"x": 12,
"y": 45
},
"options": {
"tooltip": {

View File

@@ -247,112 +247,6 @@
"panels": [],
"id": 6
},
{
"title": "Pathfinding Fast Duration",
"description": "**What:** P95 and P50 execution time of the fast pathfinding search (simplified, favoring speed over completeness).\n**How it's computed:** 95th and 50th percentiles over a 5-minute window.\n**Reading it:** Lower is better; fast pathfinding should stay quick.\n**Healthy range:** workload-dependent; typically well below full pathfinding.\n**Watch for:** Rising fast-path latency (pathfinding load or complex order books).\n**Source:** src/xrpld/rpc/detail/PathRequestManager.h ctor",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 22
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, service_instance_id) (rate(pathfind_fast_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 Fast Pathfind [{{service_instance_id}}]"
},
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.5, sum by (le, service_instance_id) (rate(pathfind_fast_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P50 Fast Pathfind [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Duration (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 7
},
{
"title": "Pathfinding Full Duration",
"description": "**What:** P95 and P50 execution time of the full, exhaustive pathfinding search.\n**How it's computed:** 95th and 50th percentiles over a 5-minute window.\n**Reading it:** Full pathfinding is heavier and slower than fast mode.\n**Healthy range:** workload-dependent.\n**Watch for:** High or rising full-path latency (expensive path computation under subscription load).\n**Source:** src/xrpld/rpc/detail/PathRequestManager.h ctor",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 22
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, service_instance_id) (rate(pathfind_full_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 Full Pathfind [{{service_instance_id}}]"
},
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.5, sum by (le, service_instance_id) (rate(pathfind_full_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P50 Full Pathfind [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Duration (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 8
},
{
"title": "Resource Warnings Rate [$xrpl_network_type]",
"description": "**What:** Rate of resource-limit warnings raised when a peer or client exceeds its usage warning threshold.\n**How it's computed:** Per-second rate of the warn counter over 5 minutes.\n**Reading it:** Occasional warnings are normal under load; a rising rate flags aggressive clients.\n**Healthy range:** workload-dependent; low.\n**Watch for:** A climbing warning rate (clients approaching limits, a precursor to drops).\n**Source:** include/xrpl/resource/detail/Logic.h Logic::Stats ctor",
@@ -361,7 +255,7 @@
"h": 4,
"w": 12,
"x": 0,
"y": 30
"y": 22
},
"options": {
"tooltip": {
@@ -404,7 +298,7 @@
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 9
"id": 7
},
{
"title": "Resource Drops Rate [$xrpl_network_type]",
@@ -414,7 +308,7 @@
"h": 4,
"w": 12,
"x": 12,
"y": 30
"y": 22
},
"options": {
"tooltip": {
@@ -457,6 +351,112 @@
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 8
},
{
"title": "Pathfinding Fast Duration",
"description": "**What:** P95 and P50 execution time of the fast pathfinding search (simplified, favoring speed over completeness).\n**How it's computed:** 95th and 50th percentiles over a 5-minute window.\n**Reading it:** Lower is better; fast pathfinding should stay quick.\n**Healthy range:** workload-dependent; typically well below full pathfinding.\n**Watch for:** Rising fast-path latency (pathfinding load or complex order books).\n**Source:** src/xrpld/rpc/detail/PathRequestManager.h ctor",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 26
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, service_instance_id) (rate(pathfind_fast_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 Fast Pathfind [{{service_instance_id}}]"
},
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.5, sum by (le, service_instance_id) (rate(pathfind_fast_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P50 Fast Pathfind [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Duration (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 9
},
{
"title": "Pathfinding Full Duration",
"description": "**What:** P95 and P50 execution time of the full, exhaustive pathfinding search.\n**How it's computed:** 95th and 50th percentiles over a 5-minute window.\n**Reading it:** Full pathfinding is heavier and slower than fast mode.\n**Healthy range:** workload-dependent.\n**Watch for:** High or rising full-path latency (expensive path computation under subscription load).\n**Source:** src/xrpld/rpc/detail/PathRequestManager.h ctor",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 26
},
"options": {
"tooltip": {
"maxHeight": 600,
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, service_instance_id) (rate(pathfind_full_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 Full Pathfind [{{service_instance_id}}]"
},
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.5, sum by (le, service_instance_id) (rate(pathfind_full_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\"}[5m])))",
"interval": "15s",
"legendFormat": "P50 Full Pathfind [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Duration (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 10
},
{

View File

@@ -8,96 +8,6 @@
"id": null,
"links": [],
"panels": [
{
"title": "RPC Request Rate by Command",
"description": "**What:** Per-command throughput of RPC calls the node serves (e.g. server_info, submit, account_info).\n**How it's computed:** Per-second rate split by command name over a 5-minute window, per node.\n**Reading it:** Shows which API endpoints drive load and how demand shifts over time.\n**Healthy range:** Workload-dependent; varies with client mix.\n**Watch for:** A sudden spike on one command can indicate a runaway client or scripted abuse.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 0
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (command, service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "{{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "reqps",
"custom": {
"axisLabel": "Requests / Sec",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 1
},
{
"title": "RPC Latency P95 by Command",
"description": "**What:** 95th-percentile response time for each RPC command.\n**How it's computed:** 95th percentile of per-command durations over a 5-minute window, per node.\n**Reading it:** Read-only lookups should be fast; heavy commands like path finding are naturally slower.\n**Healthy range:** Sub-second for most commands; command-dependent.\n**Watch for:** Rising p95 on a normally fast command signals contention or an expensive query pattern.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 0
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, command, service_instance_id) (rate(span_duration_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 {{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Latency (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 2
},
{
"title": "RPC Error Rate [$xrpl_network_type]",
"description": "**What:** Percentage of each RPC command's calls that ended in an error.\n**How it's computed:** Error calls divided by total calls per command over a 5-minute window, per node.\n**Reading it:** Green below 1%, yellow 1-5%, red above 5%.\n**Healthy range:** Below 1% for healthy commands.\n**Watch for:** A sustained error spike on one command is consistent with a client probing or misusing that endpoint.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
@@ -106,7 +16,7 @@
"h": 4,
"w": 12,
"x": 0,
"y": 8
"y": 0
},
"options": {
"tooltip": {
@@ -149,8 +59,172 @@
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 1
},
{
"title": "Top Commands by Volume [$xrpl_network_type]",
"description": "**What:** The ten most-called RPC commands by total invocations recently.\n**How it's computed:** Ranked total counts per command over a 5-minute window, per node.\n**Reading it:** Identifies the hottest API endpoints driving node load.\n**Healthy range:** Workload-dependent.\n**Watch for:** An unexpected command dominating the ranking can indicate scripted abuse or a misbehaving client.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "bargauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 0
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "topk(10, sum by (command, service_instance_id) (increase(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[$__rate_interval])))",
"interval": "15s",
"legendFormat": "{{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "none"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 2
},
{
"title": "WebSocket Message Rate [$xrpl_network_type]",
"description": "**What:** Rate of incoming WebSocket RPC messages the server processes.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Non-zero only when clients connect over WebSocket rather than HTTP.\n**Healthy range:** Zero is normal for HTTP-only deployments; otherwise workload-dependent.\n**Watch for:** An unexpected surge points to a chatty or abusive WebSocket client.\n**Source:** src/xrpld/rpc/detail/ServerHandler.cpp:ServerHandler::processSession",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 4
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=\"rpc.ws_message\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "WS Messages / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: messages/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 3
},
{
"title": "RPC Request Rate by Command",
"description": "**What:** Per-command throughput of RPC calls the node serves (e.g. server_info, submit, account_info).\n**How it's computed:** Per-second rate split by command name over a 5-minute window, per node.\n**Reading it:** Shows which API endpoints drive load and how demand shifts over time.\n**Healthy range:** Workload-dependent; varies with client mix.\n**Watch for:** A sudden spike on one command can indicate a runaway client or scripted abuse.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 8
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (command, service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "{{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "reqps",
"custom": {
"axisLabel": "Requests / Sec",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 4
},
{
"title": "RPC Latency P95 by Command",
"description": "**What:** 95th-percentile response time for each RPC command.\n**How it's computed:** 95th percentile of per-command durations over a 5-minute window, per node.\n**Reading it:** Read-only lookups should be fast; heavy commands like path finding are naturally slower.\n**Healthy range:** Sub-second for most commands; command-dependent.\n**Watch for:** Rising p95 on a normally fast command signals contention or an expensive query pattern.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "timeseries",
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 8
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "histogram_quantile(0.95, sum by (le, command, service_instance_id) (rate(span_duration_milliseconds_bucket{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[5m])))",
"interval": "15s",
"legendFormat": "P95 {{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "ms",
"custom": {
"axisLabel": "Latency (ms)",
"spanNulls": 1800000,
"insertNulls": false,
"showPoints": "auto",
"pointSize": 5,
"lineWidth": 1,
"fillOpacity": 0,
"gradientMode": "none"
}
},
"overrides": []
},
"id": 5
},
{
"title": "RPC Latency Heatmap",
"description": "**What:** Distribution of RPC response times over time across all commands.\n**How it's computed:** Counts of requests by duration band in each 5-minute window, per node.\n**Reading it:** A single tight band is healthy; brighter cells mark the common latency.\n**Healthy range:** Concentrated at low latency; command-dependent.\n**Watch for:** Two separate bands (bimodal latency) reveal a slow path affecting some requests.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
@@ -159,7 +233,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 12
"y": 16
},
"options": {
"tooltip": {
@@ -191,7 +265,7 @@
}
}
},
"id": 4
"id": 6
},
{
"title": "Overall RPC Throughput",
@@ -201,7 +275,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 12
"y": 16
},
"options": {
"tooltip": {
@@ -244,7 +318,7 @@
},
"overrides": []
},
"id": 5
"id": 7
},
{
"title": "RPC Success vs Error",
@@ -254,7 +328,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 20
"y": 24
},
"options": {
"tooltip": {
@@ -297,80 +371,6 @@
},
"overrides": []
},
"id": 6
},
{
"title": "Top Commands by Volume [$xrpl_network_type]",
"description": "**What:** The ten most-called RPC commands by total invocations recently.\n**How it's computed:** Ranked total counts per command over a 5-minute window, per node.\n**Reading it:** Identifies the hottest API endpoints driving node load.\n**Healthy range:** Workload-dependent.\n**Watch for:** An unexpected command dominating the ranking can indicate scripted abuse or a misbehaving client.\n**Source:** src/xrpld/rpc/detail/RPCHandler.cpp:callMethod",
"type": "bargauge",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 28
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "topk(10, sum by (command, service_instance_id) (increase(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=~\"rpc.command.*\"}[$__rate_interval])))",
"interval": "15s",
"legendFormat": "{{command}} [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "none"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 7
},
{
"title": "WebSocket Message Rate [$xrpl_network_type]",
"description": "**What:** Rate of incoming WebSocket RPC messages the server processes.\n**How it's computed:** Per-second rate over a 5-minute window, per node.\n**Reading it:** Non-zero only when clients connect over WebSocket rather than HTTP.\n**Healthy range:** Zero is normal for HTTP-only deployments; otherwise workload-dependent.\n**Watch for:** An unexpected surge points to a chatty or abusive WebSocket client.\n**Source:** src/xrpld/rpc/detail/ServerHandler.cpp:ServerHandler::processSession",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 12,
"y": 28
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc"
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", command=~\"$command\", span_name=\"rpc.ws_message\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "WS Messages / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: messages/s"
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 8
},
{

View File

@@ -8,6 +8,60 @@
"id": null,
"links": [],
"panels": [
{
"title": "Transaction Apply Failed Rate [$xrpl_network_type]",
"description": "**What:** Rate of transactor-stage applies that returned a non-success result while building a ledger.\n**How it's computed:** Per-second rate of tx.transactor spans at stage=apply whose ter_result is not tesSUCCESS, per node.\n**Reading it:** Green below 1/s, yellow 1-5/s, red above 5/s.\n**Healthy range:** A steady low non-zero rate is normal (e.g. tefPAST_SEQ, tecUNFUNDED and other benign conflicts).\n**Watch for:** A sustained jump above the normal baseline can indicate malformed transaction floods or a systemic issue.\n**Source:** src/libxrpl/tx/Transactor.cpp:Transactor::operator()",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 0
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", span_name=\"tx.transactor\", stage=\"apply\", ter_result!~\"tesSUCCESS|\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Failed / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: transactions/s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 1
},
{
"color": "red",
"value": 5
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 1
},
{
"title": "Transaction Processing Rate",
"description": "**What:** Rate of transactions entering the processing pipeline (submitted locally or relayed by peers) versus raw transaction messages arriving from peers before deduplication.\n**How it's computed:** Per-second rate of each over a 5-minute window, per node.\n**Reading it:** The received line sits above the processed line by the volume of duplicates filtered out.\n**Healthy range:** Workload-dependent; scales with network transaction volume.\n**Watch for:** A large and growing gap means heavy duplicate traffic; a processed-rate collapse means submissions are not being handled.\n**Source:** src/xrpld/app/misc/NetworkOPs.cpp:NetworkOPsImp::processTransaction, src/xrpld/overlay/detail/PeerImp.cpp:PeerImp::handleTransaction",
@@ -16,7 +70,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 0
"y": 4
},
"options": {
"tooltip": {
@@ -59,7 +113,7 @@
},
"overrides": []
},
"id": 1
"id": 2
},
{
"title": "Transaction Processing Latency by Type",
@@ -69,7 +123,7 @@
"h": 8,
"w": 24,
"x": 0,
"y": 8
"y": 12
},
"options": {
"tooltip": {
@@ -117,7 +171,7 @@
},
"overrides": []
},
"id": 2
"id": 3
},
{
"title": "Transaction Path Distribution",
@@ -127,7 +181,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 16
"y": 20
},
"options": {
"tooltip": {
@@ -151,7 +205,7 @@
},
"overrides": []
},
"id": 3
"id": 4
},
{
"title": "Transaction Receive vs Suppressed",
@@ -161,7 +215,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 16
"y": 20
},
"options": {
"tooltip": {
@@ -196,7 +250,7 @@
},
"overrides": []
},
"id": 4
"id": 5
},
{
"title": "Transaction Processing Duration Heatmap",
@@ -206,7 +260,7 @@
"h": 8,
"w": 12,
"x": 0,
"y": 24
"y": 28
},
"options": {
"tooltip": {
@@ -238,7 +292,7 @@
}
}
},
"id": 5
"id": 6
},
{
"title": "Transaction Apply Duration per Ledger",
@@ -248,7 +302,7 @@
"h": 8,
"w": 12,
"x": 12,
"y": 24
"y": 28
},
"options": {
"tooltip": {
@@ -291,60 +345,6 @@
},
"overrides": []
},
"id": 6
},
{
"title": "Transaction Apply Failed Rate [$xrpl_network_type]",
"description": "**What:** Rate of transactor-stage applies that returned a non-success result while building a ledger.\n**How it's computed:** Per-second rate of tx.transactor spans at stage=apply whose ter_result is not tesSUCCESS, per node.\n**Reading it:** Green below 1/s, yellow 1-5/s, red above 5/s.\n**Healthy range:** A steady low non-zero rate is normal (e.g. tefPAST_SEQ, tecUNFUNDED and other benign conflicts).\n**Watch for:** A sustained jump above the normal baseline can indicate malformed transaction floods or a systemic issue.\n**Source:** src/libxrpl/tx/Transactor.cpp:Transactor::operator()",
"type": "stat",
"gridPos": {
"h": 4,
"w": 12,
"x": 0,
"y": 32
},
"options": {
"tooltip": {
"mode": "multi",
"sort": "desc",
"maxHeight": 600
}
},
"targets": [
{
"datasource": {
"type": "prometheus"
},
"expr": "sum by (service_instance_id) (rate(span_calls_total{service_instance_id=~\"$node\", deployment_environment=~\"$deployment_environment\", xrpl_network_type=~\"$xrpl_network_type\", service_name=~\"$service_name\", span_name=\"tx.transactor\", stage=\"apply\", ter_result!~\"tesSUCCESS|\"}[$__rate_interval]))",
"interval": "15s",
"legendFormat": "Failed / Sec [{{service_instance_id}}]"
}
],
"fieldConfig": {
"defaults": {
"unit": "suffix: transactions/s",
"thresholds": {
"steps": [
{
"color": "green",
"value": null
},
{
"color": "yellow",
"value": 1
},
{
"color": "red",
"value": 5
}
]
}
},
"overrides": []
},
"repeat": "xrpl_network_type",
"repeatDirection": "h",
"maxPerRow": 2,
"id": 7
},
{