mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-27 23:38:08 +00:00
fix(telemetry): replace the alert notification body with the rule's own prose
Grafana's default notification body appends every expression node's value and every label, so an alert arrived as `Value: A=0, B=0, C=1` over a five-line `key = value` dump. The refIds mean nothing to a reader and the labels repeat the title. Add templates.yaml and point both Slack receivers and the email receiver at it, so the body is the rule's own description plus its remediation line, and any panel_* annotation renders as a link. Rewrite all 13 descriptions to be status-neutral, since the same annotation is rendered when the alert resolves: a firing-only wording made a resolved notification claim the node had stopped closing ledgers while reporting a healthy rate. The reason to care moves to a new `action` annotation, which the template prints only while firing. Each description now formats its value with printf and states its threshold, rather than emitting a bare float. Also fix the Slack title, which referenced a `rulename` label that no rule sets and so rendered blank; `alertname` is the label Grafana always provides.
This commit is contained in:
@@ -61,12 +61,12 @@ contactPoints:
|
||||
# selects Slack webhook mode (no recipient/token required) and keeps
|
||||
# provisioning valid. Replace with a real webhook to enable delivery.
|
||||
url: https://hooks.slack.invalid/disabled
|
||||
# `rulename` is used rather than CommonLabels.service_instance_id
|
||||
# because on NoData/Error evaluations Grafana replaces the query's
|
||||
# label set with only {datasource_uid, ref_id}, so the node label is
|
||||
# absent and the title would render blank — and several rules are
|
||||
# deliberately configured to fire that way.
|
||||
title: "{{ .CommonLabels.rulename }}"
|
||||
# Title and body both come from templates.yaml. `alertname` is the
|
||||
# label Grafana sets on every alert; `service_instance_id` is a query
|
||||
# label and is absent on NoData/Error evaluations, so the templates
|
||||
# guard it rather than referencing it bare.
|
||||
title: '{{ template "xrpld.title" . }}'
|
||||
text: '{{ template "xrpld.body" . }}'
|
||||
disableResolveMessage: false
|
||||
|
||||
# --- Critical tier: Slack + email ---
|
||||
@@ -78,7 +78,10 @@ contactPoints:
|
||||
settings:
|
||||
# Disabled placeholder — see xrpld-slack-default above.
|
||||
url: https://hooks.slack.invalid/disabled
|
||||
title: "[CRITICAL] {{ .CommonLabels.rulename }}"
|
||||
# The title template already carries the severity label, so this tier
|
||||
# needs no separate "[CRITICAL]" prefix.
|
||||
title: '{{ template "xrpld.title" . }}'
|
||||
text: '{{ template "xrpld.body" . }}'
|
||||
disableResolveMessage: false
|
||||
- uid: xrpld-email-critical
|
||||
type: email
|
||||
@@ -90,6 +93,12 @@ contactPoints:
|
||||
addresses: alerts-disabled@xrpld.invalid
|
||||
# One message listing all recipients, rather than one message each.
|
||||
singleEmail: true
|
||||
# Setting `message` replaces Grafana's default email body, which is
|
||||
# what removes the raw value dump and the label list. Grafana's own
|
||||
# header ("N firing alert instances", "Grouped by") and footer are
|
||||
# fixed chrome and stay.
|
||||
subject: '{{ template "xrpld.title" . }}'
|
||||
message: '{{ template "xrpld.body" . }}'
|
||||
disableResolveMessage: false
|
||||
|
||||
# To retire a receiver that a running Grafana has already stored, uncomment
|
||||
|
||||
@@ -67,9 +67,11 @@ groups:
|
||||
annotations:
|
||||
summary: "Ledger history mismatch on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} recorded
|
||||
{{ $values.B.Value }} ledger history mismatch(es) in the last 15m.
|
||||
The node's built ledger diverges from the validated network chain.
|
||||
Ledger history mismatches on {{ $labels.service_instance_id }} in the last
|
||||
15m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 0).
|
||||
action: >-
|
||||
A built ledger diverged from the validated network chain, which can mean
|
||||
corrupt local history. Check byzantine ledger jumps and the node-store.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -131,9 +133,11 @@ groups:
|
||||
annotations:
|
||||
summary: "Ledger closing stalled on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has closed no ledgers for
|
||||
several minutes (5m rate has decayed to zero). Consensus or ledger
|
||||
advancement is stuck, or the process is gone.
|
||||
Ledger close rate on {{ $labels.service_instance_id }} over 5m: {{ printf
|
||||
"%.4f" $values.B.Value }} ledgers/s (threshold: below 0.001).
|
||||
action: >-
|
||||
Consensus or ledger advancement is stuck, or the process is gone. Check
|
||||
consensus round duration, worker pool saturation and peer supply.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -206,9 +210,11 @@ groups:
|
||||
annotations:
|
||||
summary: "Validated ledger stale on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has a validated ledger age of
|
||||
{{ $values.B.Value }}s (>60s). The node is not keeping up with the
|
||||
validated network chain.
|
||||
Validated ledger age on {{ $labels.service_instance_id }}: {{ printf "%.0f"
|
||||
$values.B.Value }}s (threshold: over 60s).
|
||||
action: >-
|
||||
The node is no longer tracking the network. Check peer connectivity, node-
|
||||
store IO latency and consensus rounds.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -285,9 +291,10 @@ groups:
|
||||
annotations:
|
||||
summary: "Validations missed on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Validator {{ $labels.service_instance_id }} is missing
|
||||
{{ $values.B.Value }} (fraction) of its validations over 15m. Its
|
||||
validations are not agreeing with the validated ledger, which risks
|
||||
Missed-validation fraction on {{ $labels.service_instance_id }} over 15m: {{
|
||||
printf "%.3f" $values.B.Value }} (threshold: over 0.1).
|
||||
action: >-
|
||||
Its validations are not agreeing with the validated ledger, which risks
|
||||
removal from UNLs.
|
||||
data:
|
||||
- refId: A
|
||||
@@ -358,9 +365,11 @@ groups:
|
||||
annotations:
|
||||
summary: "No validations checked on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has checked no incoming
|
||||
validations for several minutes (5m rate has decayed to zero). The
|
||||
validation stream from peers may have stopped.
|
||||
Validations-checked rate on {{ $labels.service_instance_id }} over 5m: {{
|
||||
printf "%.4f" $values.B.Value }} per s (threshold: below 0.001).
|
||||
action: >-
|
||||
The validation stream from peers may have stopped. Check peer count and the
|
||||
overlay.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -431,9 +440,11 @@ groups:
|
||||
annotations:
|
||||
summary: "Job queue transaction overflow on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} overflowed its transaction
|
||||
job queue {{ $values.B.Value }} time(s) in the last 15m.
|
||||
Transactions are being dropped under load.
|
||||
Transaction job-queue overflows on {{ $labels.service_instance_id }} in the
|
||||
last 15m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 0).
|
||||
action: >-
|
||||
Transactions are being dropped under load. Check job-queue depth and worker
|
||||
saturation.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -502,9 +513,10 @@ groups:
|
||||
annotations:
|
||||
summary: "Job queue latency high on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has a p99 job-queue wait of
|
||||
{{ $values.B.Value }}µs (>1s) over 5m. The node is saturated and jobs are
|
||||
backing up.
|
||||
p99 job-queue wait on {{ $labels.service_instance_id }} over 5m: {{ printf
|
||||
"%.0f" $values.B.Value }}us (threshold: over 1000000us, i.e. 1s).
|
||||
action: >-
|
||||
The node is saturated and jobs are backing up. Check worker pool saturation.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -572,9 +584,10 @@ groups:
|
||||
annotations:
|
||||
summary: "Node store IO latency high on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has a p95 node-store IO
|
||||
latency of {{ $values.B.Value }}ms (>1s) over 10m. Check disk
|
||||
utilisation and whether the store is on a slow volume.
|
||||
p95 node-store IO latency on {{ $labels.service_instance_id }} over 10m: {{
|
||||
printf "%.0f" $values.B.Value }}ms (threshold: over 1000ms).
|
||||
action: >-
|
||||
Check disk utilisation and whether the store is on a slow volume.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -655,11 +668,13 @@ groups:
|
||||
annotations:
|
||||
summary: "Node state flapping on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} re-entered the FULL state
|
||||
{{ $values.B.Value }} time(s) in the last hour past its first hour
|
||||
of uptime. It is flapping out of sync rather than holding FULL.
|
||||
Likely the online-delete rotation cache-freshen; check the rotation
|
||||
spans and cache lock-hold peak.
|
||||
FULL-state re-entries on {{ $labels.service_instance_id }} in the last hour:
|
||||
{{ printf "%.0f" $values.B.Value }} (threshold: more than 0, past the first
|
||||
hour of uptime).
|
||||
action: >-
|
||||
The node is flapping out of sync rather than holding FULL. A likely cause is
|
||||
the online-delete rotation cache-freshen; check the rotation spans and the
|
||||
cache lock-hold peak.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -724,9 +739,12 @@ groups:
|
||||
annotations:
|
||||
summary: "Node not in FULL state on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has been below FULL
|
||||
(state={{ $values.B.Value }}; 0=disconnected 1=connected 2=syncing
|
||||
3=tracking 4=full) for 15m. It is not fully synced with the network.
|
||||
Server state on {{ $labels.service_instance_id }} for the last 15m: {{
|
||||
printf "%.0f" $values.B.Value }} (threshold: below 4; 0=disconnected
|
||||
1=connected 2=syncing 3=tracking 4=full).
|
||||
action: >-
|
||||
The node is not fully synced with the network. Check peer supply and sync
|
||||
progress.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -807,9 +825,11 @@ groups:
|
||||
annotations:
|
||||
summary: "Manifest job convoy on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} has {{ $values.B.Value }}
|
||||
manifest jobs waiting (>3) for 10m. Peer manifest dumps are
|
||||
saturating the job pool and convoying on the manifest cache lock.
|
||||
Manifest jobs waiting on {{ $labels.service_instance_id }} over 10m: {{
|
||||
printf "%.0f" $values.B.Value }} (threshold: more than 3).
|
||||
action: >-
|
||||
Peer manifest dumps are saturating the job pool and convoying on the
|
||||
manifest cache lock.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
@@ -884,9 +904,10 @@ groups:
|
||||
annotations:
|
||||
summary: "Inbound manifest flood on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} is receiving
|
||||
{{ $values.B.Value }} B/s of manifest traffic over 10m, above the
|
||||
512 KiB/s (524288 B/s) threshold.
|
||||
Inbound manifest traffic on {{ $labels.service_instance_id }} over 10m: {{
|
||||
printf "%.0f" $values.B.Value }} B/s (threshold: over 524288 B/s, i.e. 512
|
||||
KiB/s).
|
||||
action: >-
|
||||
A peer is flooding oversized TMManifests dumps.
|
||||
data:
|
||||
- refId: A
|
||||
@@ -955,9 +976,10 @@ groups:
|
||||
annotations:
|
||||
summary: "Resource-driven peer disconnects on {{ $labels.service_instance_id }}"
|
||||
description: >-
|
||||
Node {{ $labels.service_instance_id }} disconnected
|
||||
{{ $values.B.Value }} peer(s) for resource-budget violations in the
|
||||
last 30m. Sustained disconnects can starve the node of peers.
|
||||
Resource-budget peer disconnects on {{ $labels.service_instance_id }} in the
|
||||
last 30m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 5).
|
||||
action: >-
|
||||
Sustained disconnects can starve the node of peers.
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
# Notification templates for the xrpld OTel alerts.
|
||||
#
|
||||
# Why these exist: Grafana's built-in message body appends a raw dump of every
|
||||
# expression node's value and every label, so a notification reads
|
||||
# `Value: A=0, B=0, C=1` followed by five `key = value` lines. The refIds carry
|
||||
# no meaning to a reader, and the label dump repeats what the title already
|
||||
# says. These templates replace that body with the rule's own prose.
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# What a template may and may not use
|
||||
# ---------------------------------------------------------------------------
|
||||
# The function set here is much smaller than Go's text/template plus sprig.
|
||||
# Available and used below: `match` (regexp), `reReplaceAll`, `title`, `printf`,
|
||||
# `len`, `index`, `eq`. NOT available — each fails the whole template with
|
||||
# `function "X" not defined`, which silently ships the raw template text in the
|
||||
# notification instead: `hasPrefix`, `hasSuffix`, `contains`, `humanize`,
|
||||
# `humanizeDuration`, `humanizePercentage`.
|
||||
#
|
||||
# A template error does NOT mark the rule unhealthy. Rules keep reporting
|
||||
# `health=ok` and the broken text is only visible in the delivered message, so
|
||||
# any change here must be checked against a real notification, not against rule
|
||||
# state.
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# Two label traps
|
||||
# ---------------------------------------------------------------------------
|
||||
# * `service_instance_id` is a QUERY label, so it is absent whenever Grafana
|
||||
# evaluates NoData or Error — those evaluations carry only
|
||||
# {datasource_uid, ref_id}. Two rules set `noDataState: Alerting` and so
|
||||
# fire that way deliberately. `xrpld.node` below falls back rather than
|
||||
# rendering an empty string.
|
||||
# * `alertname` is set by Grafana on every alert. There is no `rulename`
|
||||
# label, so referencing one renders blank.
|
||||
#
|
||||
# Values are deliberately NOT printed. Each rule's `description` already states
|
||||
# its own measurement, formatted and with its threshold, which is the readable
|
||||
# form of what the raw value dump was trying to say.
|
||||
|
||||
apiVersion: 1
|
||||
|
||||
templates:
|
||||
- orgId: 1
|
||||
name: xrpld.notifications
|
||||
template: |
|
||||
{{- /* Node identity, or a marker when the query label is absent. */ -}}
|
||||
{{ define "xrpld.node" -}}
|
||||
{{ if .Labels.service_instance_id }}{{ .Labels.service_instance_id }}{{ else }}unknown node (NoData/Error evaluation){{ end }}
|
||||
{{- end }}
|
||||
|
||||
{{- /* One line, used as the Slack title and the email subject. */ -}}
|
||||
{{ define "xrpld.title" -}}
|
||||
{{ if .Alerts.Firing }}🔥 FIRING{{ else }}✅ RESOLVED{{ end }}
|
||||
{{- with .CommonLabels.alertname }} · {{ . }}{{ end }}
|
||||
{{- with .CommonLabels.severity }} ({{ . }}){{ end }}
|
||||
{{- with .CommonLabels.service_instance_id }} · {{ . }}{{ end }}
|
||||
{{- end }}
|
||||
|
||||
{{- /* Message body: one block per alert, prose only. */ -}}
|
||||
{{ define "xrpld.body" -}}
|
||||
{{ with .Alerts.Firing }}{{ range . }}
|
||||
:rotating_light: *FIRING* · *{{ .Labels.alertname }}* on `{{ template "xrpld.node" . }}`
|
||||
{{ .Annotations.description }}
|
||||
{{- with .Annotations.action }}
|
||||
*Why it matters:* {{ . }}
|
||||
{{- end }}
|
||||
{{- range $key, $url := .Annotations }}
|
||||
{{- if match "^panel_" $key }}
|
||||
:chart_with_upwards_trend: <{{ $url }}|{{ title (reReplaceAll "_" " " (reReplaceAll "^panel_" "" $key)) }}>
|
||||
{{- end }}{{ end }}
|
||||
{{ end }}{{ end }}
|
||||
{{- with .Alerts.Resolved }}{{ range . }}
|
||||
:white_check_mark: *RESOLVED* · *{{ .Labels.alertname }}* on `{{ template "xrpld.node" . }}`
|
||||
{{ .Annotations.description }}
|
||||
{{ end }}{{ end }}
|
||||
{{- end }}
|
||||
Reference in New Issue
Block a user