diff --git a/docker/telemetry/grafana/provisioning/alerting/contactpoints.yaml b/docker/telemetry/grafana/provisioning/alerting/contactpoints.yaml
index 2fd6ce6a52..b905c889f4 100644
--- a/docker/telemetry/grafana/provisioning/alerting/contactpoints.yaml
+++ b/docker/telemetry/grafana/provisioning/alerting/contactpoints.yaml
@@ -96,16 +96,14 @@ contactPoints:
addresses: alerts-disabled@xrpld.invalid
# One message listing all recipients, rather than one message each.
singleEmail: true
- # Setting `message` replaces Grafana's default email body, which is
- # what removes the raw value dump and the label list. Grafana's own
- # header ("N firing alert instances", "Grouped by") and footer are
- # fixed chrome and stay.
- #
- # This uses the EMAIL body, not the Slack one. Email escapes HTML and
- # cannot render Slack markup, so a shared body arrives as literal
- # asterisks and raw URLs. See templates.yaml.
- subject: '{{ template "xrpld.title" . }}'
- message: '{{ template "xrpld.email.body" . }}'
+ # DELIBERATELY no `subject`/`message` override here, unlike the Slack
+ # receivers above. Grafana's default email body is the good one: bold
+ # section headings, a styled label table, annotations rendered as real
+ # clickable links, and the View/Silence buttons. Overriding `message`
+ # replaces all of that with plain text, because email escapes HTML, so
+ # a custom body cannot reproduce any of it. The value line the default
+ # body carries is made readable by NAMING each rule's query steps (see
+ # rules.yaml), not by replacing the body.
disableResolveMessage: false
# To retire a receiver that a running Grafana has already stored, uncomment
diff --git a/docker/telemetry/grafana/provisioning/alerting/rules.yaml b/docker/telemetry/grafana/provisioning/alerting/rules.yaml
index e6adaa7ee9..1d88132564 100644
--- a/docker/telemetry/grafana/provisioning/alerting/rules.yaml
+++ b/docker/telemetry/grafana/provisioning/alerting/rules.yaml
@@ -56,7 +56,7 @@ groups:
# hence increase() over a wide window rather than a decaying rate().
- uid: xrpld-ledger-history-mismatch
title: LedgerHistoryMismatch
- condition: C
+ condition: is_mismatching
for: 2m
isPaused: true
noDataState: NoData
@@ -68,45 +68,45 @@ groups:
summary: "Ledger history mismatch on {{ $labels.service_instance_id }}"
description: >-
Ledger history mismatches on {{ $labels.service_instance_id }} in the last
- 15m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 0).
+ 15m: {{ printf "%.0f" $values.mismatches_now.Value }} (threshold: more than 0).
action: >-
A built ledger diverged from the validated network chain, which can mean
corrupt local history. Check byzantine ledger jumps and the node-store.
data:
- - refId: A
+ - refId: mismatches_15m
relativeTimeRange:
from: 1200
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: mismatches_15m
expr: sum by (service_instance_id) (increase(ledger_history_mismatch_total{service_name="xrpld"}[15m]))
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: mismatches_now
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: mismatches_now
type: reduce
reducer: last
- expression: A
+ expression: mismatches_15m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_mismatching
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_mismatching
type: threshold
- expression: B
+ expression: mismatches_now
conditions:
- evaluator:
type: gt
@@ -122,7 +122,7 @@ groups:
# dead node trips the threshold instead of vanishing from the result.
- uid: xrpld-ledger-close-stalled
title: LedgerCloseStalled
- condition: C
+ condition: is_stalled
for: 3m
isPaused: true
noDataState: Alerting
@@ -134,18 +134,18 @@ groups:
summary: "Ledger closing stalled on {{ $labels.service_instance_id }}"
description: >-
Ledger close rate on {{ $labels.service_instance_id }} over 5m: {{ printf
- "%.4f" $values.B.Value }} ledgers/s (threshold: below 0.001).
+ "%.4f" $values.close_rate_now.Value }} ledgers/s (threshold: below 0.001).
action: >-
Consensus or ledger advancement is stuck, or the process is gone. Check
consensus round duration, worker pool saturation and peer supply.
data:
- - refId: A
+ - refId: close_rate_5m
relativeTimeRange:
from: 600
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: close_rate_5m
expr: |-
sum by (service_instance_id) (rate(ledgers_closed_total{service_name="xrpld"}[5m]))
or (0 * max by (service_instance_id) (max_over_time(ledgers_closed_total{service_name="xrpld"}[1h])))
@@ -153,28 +153,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: close_rate_now
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: close_rate_now
type: reduce
reducer: last
- expression: A
+ expression: close_rate_5m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_stalled
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_stalled
type: threshold
- expression: B
+ expression: close_rate_now
conditions:
- evaluator:
type: lt
@@ -199,7 +199,7 @@ groups:
# validated ledger is caught by LedgerCloseStalled and NodeNotFull.
- uid: xrpld-validated-ledger-stale
title: ValidatedLedgerStale
- condition: C
+ condition: is_stale
for: 5m
isPaused: true
noDataState: NoData
@@ -211,45 +211,45 @@ groups:
summary: "Validated ledger stale on {{ $labels.service_instance_id }}"
description: >-
Validated ledger age on {{ $labels.service_instance_id }}: {{ printf "%.0f"
- $values.B.Value }}s (threshold: over 60s).
+ $values.ledger_age_now.Value }}s (threshold: over 60s).
action: >-
The node is no longer tracking the network. Check peer connectivity, node-
store IO latency and consensus rounds.
data:
- - refId: A
+ - refId: ledger_age_s
relativeTimeRange:
from: 600
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: ledger_age_s
expr: max by (service_instance_id) (ledgermaster_validated_ledger_age{service_name="xrpld"} < 1209600)
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: ledger_age_now
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: ledger_age_now
type: reduce
reducer: last
- expression: A
+ expression: ledger_age_s
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_stale
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_stale
type: threshold
- expression: B
+ expression: ledger_age_now
conditions:
- evaluator:
type: gt
@@ -280,7 +280,7 @@ groups:
# disagreement among nodes that do validate.
- uid: xrpld-validations-missed
title: ValidationsMissed
- condition: C
+ condition: is_missing_validations
for: 15m
isPaused: true
noDataState: NoData
@@ -292,18 +292,18 @@ groups:
summary: "Validations missed on {{ $labels.service_instance_id }}"
description: >-
Missed-validation fraction on {{ $labels.service_instance_id }} over 15m: {{
- printf "%.3f" $values.B.Value }} (threshold: over 0.1).
+ printf "%.3f" $values.missed_ratio_now.Value }} (threshold: over 0.1).
action: >-
Its validations are not agreeing with the validated ledger, which risks
removal from UNLs.
data:
- - refId: A
+ - refId: missed_ratio_15m
relativeTimeRange:
from: 1200
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: missed_ratio_15m
expr: |-
(
sum by (service_instance_id) (rate(validation_missed_total{service_name="xrpld"}[15m]))
@@ -318,28 +318,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: missed_ratio_now
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: missed_ratio_now
type: reduce
reducer: last
- expression: A
+ expression: missed_ratio_15m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_missing_validations
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_missing_validations
type: threshold
- expression: B
+ expression: missed_ratio_now
conditions:
- evaluator:
type: gt
@@ -354,7 +354,7 @@ groups:
# (see the LedgerCloseStalled comment).
- uid: xrpld-validations-not-checked
title: ValidationsNotChecked
- condition: C
+ condition: is_not_checking
for: 5m
isPaused: true
noDataState: Alerting
@@ -366,18 +366,18 @@ groups:
summary: "No validations checked on {{ $labels.service_instance_id }}"
description: >-
Validations-checked rate on {{ $labels.service_instance_id }} over 5m: {{
- printf "%.4f" $values.B.Value }} per s (threshold: below 0.001).
+ printf "%.4f" $values.check_rate_now.Value }} per s (threshold: below 0.001).
action: >-
The validation stream from peers may have stopped. Check peer count and the
overlay.
data:
- - refId: A
+ - refId: check_rate_5m
relativeTimeRange:
from: 600
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: check_rate_5m
expr: |-
sum by (service_instance_id) (rate(validations_checked_total{service_name="xrpld"}[5m]))
or (0 * max by (service_instance_id) (max_over_time(validations_checked_total{service_name="xrpld"}[1h])))
@@ -385,28 +385,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: check_rate_now
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: check_rate_now
type: reduce
reducer: last
- expression: A
+ expression: check_rate_5m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_not_checking
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_not_checking
type: threshold
- expression: B
+ expression: check_rate_now
conditions:
- evaluator:
type: lt
@@ -429,7 +429,7 @@ groups:
# sparse counters).
- uid: xrpld-jobqueue-tx-overflow
title: JobQueueTxOverflow
- condition: C
+ condition: is_overflowing
for: 2m
isPaused: true
noDataState: NoData
@@ -441,45 +441,45 @@ groups:
summary: "Job queue transaction overflow on {{ $labels.service_instance_id }}"
description: >-
Transaction job-queue overflows on {{ $labels.service_instance_id }} in the
- last 15m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 0).
+ last 15m: {{ printf "%.0f" $values.overflows_now.Value }} (threshold: more than 0).
action: >-
Transactions are being dropped under load. Check job-queue depth and worker
saturation.
data:
- - refId: A
+ - refId: overflows_15m
relativeTimeRange:
from: 1200
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: overflows_15m
expr: sum by (service_instance_id) (increase(jq_trans_overflow_total{service_name="xrpld"}[15m]))
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: overflows_now
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: overflows_now
type: reduce
reducer: last
- expression: A
+ expression: overflows_15m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_overflowing
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_overflowing
type: threshold
- expression: B
+ expression: overflows_now
conditions:
- evaluator:
type: gt
@@ -502,7 +502,7 @@ groups:
# healthy p99 and fires only on genuine saturation, which is the intent.
- uid: xrpld-jobqueue-latency-high
title: JobQueueLatencyHigh
- condition: C
+ condition: is_queue_slow
for: 5m
isPaused: true
noDataState: NoData
@@ -514,44 +514,44 @@ groups:
summary: "Job queue latency high on {{ $labels.service_instance_id }}"
description: >-
p99 job-queue wait on {{ $labels.service_instance_id }} over 5m: {{ printf
- "%.0f" $values.B.Value }}us (threshold: over 1000000us, i.e. 1s).
+ "%.0f" $values.queue_p99_us_now.Value }}us (threshold: over 1000000us, i.e. 1s).
action: >-
The node is saturated and jobs are backing up. Check worker pool saturation.
data:
- - refId: A
+ - refId: queue_p99_us_5m
relativeTimeRange:
from: 600
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: queue_p99_us_5m
expr: histogram_quantile(0.99, sum by (le, service_instance_id) (rate(job_queued_us_bucket{service_name="xrpld"}[5m])))
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: queue_p99_us_now
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: queue_p99_us_now
type: reduce
reducer: last
- expression: A
+ expression: queue_p99_us_5m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_queue_slow
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_queue_slow
type: threshold
- expression: B
+ expression: queue_p99_us_now
conditions:
- evaluator:
type: gt
@@ -573,7 +573,7 @@ groups:
# is 12.9ms, far below the threshold.
- uid: xrpld-nodestore-io-latency-high
title: NodeStoreIOLatencyHigh
- condition: C
+ condition: is_io_slow
for: 10m
isPaused: true
noDataState: NoData
@@ -585,44 +585,44 @@ groups:
summary: "Node store IO latency high on {{ $labels.service_instance_id }}"
description: >-
p95 node-store IO latency on {{ $labels.service_instance_id }} over 10m: {{
- printf "%.0f" $values.B.Value }}ms (threshold: over 1000ms).
+ printf "%.0f" $values.io_p95_ms_now.Value }}ms (threshold: over 1000ms).
action: >-
Check disk utilisation and whether the store is on a slow volume.
data:
- - refId: A
+ - refId: io_p95_ms_10m
relativeTimeRange:
from: 900
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: io_p95_ms_10m
expr: histogram_quantile(0.95, sum by (le, service_instance_id) (rate(ios_latency_milliseconds_bucket{service_name="xrpld"}[10m])))
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: io_p95_ms_now
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: io_p95_ms_now
type: reduce
reducer: last
- expression: A
+ expression: io_p95_ms_10m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_io_slow
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_io_slow
type: threshold
- expression: B
+ expression: io_p95_ms_now
conditions:
- evaluator:
type: gt
@@ -657,7 +657,7 @@ groups:
# the gate suppresses that first hour so a restart does not page.
- uid: xrpld-node-state-flapping
title: NodeStateFlapping
- condition: C
+ condition: is_flapping
for: 15m
isPaused: true
noDataState: OK
@@ -669,20 +669,20 @@ groups:
summary: "Node state flapping on {{ $labels.service_instance_id }}"
description: >-
FULL-state re-entries on {{ $labels.service_instance_id }} in the last hour:
- {{ printf "%.0f" $values.B.Value }} (threshold: more than 0, past the first
+ {{ printf "%.0f" $values.full_transitions_now.Value }} (threshold: more than 0, past the first
hour of uptime).
action: >-
The node is flapping out of sync rather than holding FULL. A likely cause is
the online-delete rotation cache-freshen; check the rotation spans and the
cache lock-hold peak.
data:
- - refId: A
+ - refId: full_transitions_1h
relativeTimeRange:
from: 3900
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: full_transitions_1h
expr: |-
sum by (service_instance_id) (increase(state_accounting_full_transitions{service_name="xrpld"}[1h]))
and on (service_instance_id)
@@ -691,28 +691,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: full_transitions_now
relativeTimeRange:
from: 3900
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: full_transitions_now
type: reduce
reducer: last
- expression: A
+ expression: full_transitions_1h
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_flapping
relativeTimeRange:
from: 3900
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_flapping
type: threshold
- expression: B
+ expression: full_transitions_now
conditions:
- evaluator:
type: gt
@@ -728,7 +728,7 @@ groups:
# SYNCING=2, TRACKING=3, FULL=4.
- uid: xrpld-node-not-full
title: NodeNotFull
- condition: C
+ condition: is_not_full
for: 15m
isPaused: true
noDataState: NoData
@@ -740,19 +740,19 @@ groups:
summary: "Node not in FULL state on {{ $labels.service_instance_id }}"
description: >-
Server state on {{ $labels.service_instance_id }} for the last 15m: {{
- printf "%.0f" $values.B.Value }} (threshold: below 4; 0=disconnected
+ printf "%.0f" $values.server_state_now.Value }} (threshold: below 4; 0=disconnected
1=connected 2=syncing 3=tracking 4=full).
action: >-
The node is not fully synced with the network. Check peer supply and sync
progress.
data:
- - refId: A
+ - refId: server_state_code
relativeTimeRange:
from: 1200
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: server_state_code
expr: |-
max by (service_instance_id) (server_info{service_name="xrpld", metric="server_state"})
and on (service_instance_id)
@@ -761,28 +761,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: server_state_now
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: server_state_now
type: reduce
reducer: last
- expression: A
+ expression: server_state_code
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_not_full
relativeTimeRange:
from: 1200
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_not_full
type: threshold
- expression: B
+ expression: server_state_now
conditions:
- evaluator:
type: lt
@@ -814,7 +814,7 @@ groups:
# burst (peaks of 5 and 11, lasting well under 10m) from paging.
- uid: xrpld-manifest-job-convoy
title: ManifestJobQueueConvoy
- condition: C
+ condition: is_convoying
for: 10m
isPaused: true
noDataState: NoData
@@ -826,45 +826,45 @@ groups:
summary: "Manifest job convoy on {{ $labels.service_instance_id }}"
description: >-
Manifest jobs waiting on {{ $labels.service_instance_id }} over 10m: {{
- printf "%.0f" $values.B.Value }} (threshold: more than 3).
+ printf "%.0f" $values.manifest_waiting_now.Value }} (threshold: more than 3).
action: >-
Peer manifest dumps are saturating the job pool and convoying on the
manifest cache lock.
data:
- - refId: A
+ - refId: manifest_waiting_10m
relativeTimeRange:
from: 900
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: manifest_waiting_10m
expr: sum by (service_instance_id) (jobq_manifest_waiting{service_name="xrpld"})
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: manifest_waiting_now
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: manifest_waiting_now
type: reduce
reducer: last
- expression: A
+ expression: manifest_waiting_10m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_convoying
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_convoying
type: threshold
- expression: B
+ expression: manifest_waiting_now
conditions:
- evaluator:
type: gt
@@ -893,7 +893,7 @@ groups:
# alerted. ManifestJobQueueConvoy covers that window via the job pool.
- uid: xrpld-manifest-flood-inbound
title: ManifestFloodInbound
- condition: C
+ condition: is_flooding
for: 10m
isPaused: true
noDataState: NoData
@@ -905,18 +905,18 @@ groups:
summary: "Inbound manifest flood on {{ $labels.service_instance_id }}"
description: >-
Inbound manifest traffic on {{ $labels.service_instance_id }} over 10m: {{
- printf "%.0f" $values.B.Value }} B/s (threshold: over 524288 B/s, i.e. 512
+ printf "%.0f" $values.manifest_bytes_s_now.Value }} B/s (threshold: over 524288 B/s, i.e. 512
KiB/s).
action: >-
A peer is flooding oversized TMManifests dumps.
data:
- - refId: A
+ - refId: manifest_bytes_s_10m
relativeTimeRange:
from: 900
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: manifest_bytes_s_10m
expr: |-
sum by (service_instance_id) (rate(overhead_manifest_bytes_in{service_name="xrpld"}[10m]))
and on (service_instance_id)
@@ -925,28 +925,28 @@ groups:
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: manifest_bytes_s_now
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: manifest_bytes_s_now
type: reduce
reducer: last
- expression: A
+ expression: manifest_bytes_s_10m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_flooding
relativeTimeRange:
from: 900
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_flooding
type: threshold
- expression: B
+ expression: manifest_bytes_s_now
conditions:
- evaluator:
type: gt
@@ -965,7 +965,7 @@ groups:
# trips on a node already in trouble.
- uid: xrpld-peer-resource-disconnects
title: PeerResourceDisconnects
- condition: C
+ condition: is_disconnecting
for: 5m
isPaused: true
noDataState: NoData
@@ -977,44 +977,44 @@ groups:
summary: "Resource-driven peer disconnects on {{ $labels.service_instance_id }}"
description: >-
Resource-budget peer disconnects on {{ $labels.service_instance_id }} in the
- last 30m: {{ printf "%.0f" $values.B.Value }} (threshold: more than 5).
+ last 30m: {{ printf "%.0f" $values.disconnects_now.Value }} (threshold: more than 5).
action: >-
Sustained disconnects can starve the node of peers.
data:
- - refId: A
+ - refId: disconnects_30m
relativeTimeRange:
from: 2100
to: 0
datasourceUid: prometheus
model:
- refId: A
+ refId: disconnects_30m
expr: sum by (service_instance_id) (increase(server_info{service_name="xrpld", metric="peer_disconnects_resources"}[30m]))
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- - refId: B
+ - refId: disconnects_now
relativeTimeRange:
from: 2100
to: 0
datasourceUid: __expr__
model:
- refId: B
+ refId: disconnects_now
type: reduce
reducer: last
- expression: A
+ expression: disconnects_30m
datasource:
type: __expr__
uid: __expr__
- - refId: C
+ - refId: is_disconnecting
relativeTimeRange:
from: 2100
to: 0
datasourceUid: __expr__
model:
- refId: C
+ refId: is_disconnecting
type: threshold
- expression: B
+ expression: disconnects_now
conditions:
- evaluator:
type: gt
diff --git a/docker/telemetry/grafana/provisioning/alerting/templates.yaml b/docker/telemetry/grafana/provisioning/alerting/templates.yaml
index 071fd7499a..8d158318f3 100644
--- a/docker/telemetry/grafana/provisioning/alerting/templates.yaml
+++ b/docker/telemetry/grafana/provisioning/alerting/templates.yaml
@@ -7,23 +7,24 @@
# says. These templates replace that body with the rule's own prose.
#
# ---------------------------------------------------------------------------
-# Slack and email need SEPARATE bodies. Do not share one.
+# SLACK ONLY. Email deliberately has no template here.
# ---------------------------------------------------------------------------
-# Measured against a real email and a real Slack payload:
+# Measured, and the reason this file covers one channel rather than two:
#
-# * Email ESCAPES any HTML in the message: a literal `
` arrives as
-# `<br>`. So no tags, and no clickable anchors are possible here.
-# * Email DOES turn a newline in the template into a real `
`, so line
-# breaks are the only layout tool available.
-# * Slack markup is meaningless in email. Bold asterisks, backticks and
-# `:rotating_light:` all arrive as literal characters, and worst of all a
-# `` link cannot become a link, so it dumps the whole URL
-# inline. Four panel links then bury the prose in a wall of text.
+# * Email ESCAPES any HTML in the message body, so a literal `
` arrives
+# as `<br>` and no tag or anchor is possible. A custom email body is
+# therefore plain text and cannot reproduce Grafana's default layout, which
+# already gives section headings, a label table, clickable annotation links
+# and the View/Silence buttons. So email keeps the default body.
+# * Slack markup is meaningless in email anyway: bold asterisks, backticks
+# and `:rotating_light:` arrive as literal characters, and a ``
+# link cannot become a link, so it dumps the whole URL inline. Four panel
+# links bury the prose in a wall of text. That is what a shared body did.
# * Real emoji CHARACTERS render in both channels; the `:shortcode:` form
# only works in Slack.
#
-# So the email body is plain text with one fact per line and each link on its
-# own labelled line, and the Slack body keeps mrkdwn.
+# The value line Grafana's default email body ends with is made readable by
+# NAMING each rule's query steps (see rules.yaml), not by replacing the body.
#
# ---------------------------------------------------------------------------
# What a template may and may not use
@@ -93,23 +94,3 @@ templates:
{{ .Annotations.description }}
{{ end }}{{ end }}
{{- end }}
-
- {{- /* EMAIL body: plain text only. Newlines are the sole layout tool,
- and each link sits on its own labelled line. */ -}}
- {{ define "xrpld.email.body" -}}
- {{ with .Alerts.Firing }}{{ range . }}
- FIRING - {{ .Labels.alertname }} on {{ template "xrpld.node" . }}
- {{ .Annotations.description }}
- {{- with .Annotations.action }}
- Why it matters: {{ . }}
- {{- end }}
- {{- range $key, $url := .Annotations }}
- {{- if match "^panel_" $key }}
- {{ title (reReplaceAll "_" " " (reReplaceAll "^panel_" "" $key)) }}: {{ $url }}
- {{- end }}{{ end }}
- {{ end }}{{ end }}
- {{- with .Alerts.Resolved }}{{ range . }}
- RESOLVED - {{ .Labels.alertname }} on {{ template "xrpld.node" . }}
- {{ .Annotations.description }}
- {{ end }}{{ end }}
- {{- end }}