mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-27 07:26:51 +00:00
Eighteen conflict regions across nine files. Resolved by asking, per region, which side is the better final state rather than by taking a branch wholesale. Telemetry.cpp keeps phase-9's two resource builders. phase-8 offered a single makeResource() with no node identity; phase-9 splits it into makeTracerResource() and makeMetricsResource() because the metrics provider is built in the constructor, before setNodeId() runs, so xrpl.node.id can only be stamped unconditionally on the tracer side. Collapsing them would have dropped that attribute, which is what keeps per-node traces from folding into one identity. Telemetry.h and the config test compose both sides: phase-9's nodeId member and its assertion, plus the renamed endpoint. xrpld-telemetry.cfg keeps phase-9's devnet identity and its metrics_endpoint, renames the traces key, and drops exporter=otlp_http. Nothing reads an `exporter` key on any branch in the chain: it was a real Setup member in the first phase-1b implementation, removed when only OTLP/HTTP was wired up, and already deleted from TESTING.md once on the same grounds. The cfg line was the last carrier. The docs keep phase-9's versions, which are both fuller and more accurate: the incoming runbook listed the consensus strategy values as "random" where the code compares against "attribute". OTelCollector.cpp had five comment-only regions in a file phase-7 owns, so those take the upstream side. MetricsRegistry.h's usage example named a member that no longer exists and the wrong arity; it now matches the real three-argument call and says where the endpoint comes from.
207 lines
8.8 KiB
YAML
207 lines
8.8 KiB
YAML
# Docker Compose stack for xrpld OpenTelemetry observability.
|
|
#
|
|
# Provides services for local development:
|
|
# - otel-collector: receives OTLP traces from xrpld, batches and
|
|
# forwards them to Tempo. Also tails xrpld log files
|
|
# via filelog receiver and exports to Loki. Listens on ports
|
|
# 4317 (gRPC) and 4318 (HTTP).
|
|
# - tempo: Grafana Tempo tracing backend, queryable via Grafana Explore
|
|
# on port 3000. Recommended for production (S3/GCS storage, TraceQL).
|
|
# - loki: Grafana Loki log aggregation backend for centralized log
|
|
# ingestion and log-trace correlation.
|
|
# - grafana: dashboards on port 3000, pre-configured with Tempo,
|
|
# Prometheus, and Loki datasources.
|
|
#
|
|
# Requires Docker Compose >= 2.24.0. The grafana service uses the long-form
|
|
# `env_file` mapping (`path:` / `required:`), which older Compose cannot parse —
|
|
# and it fails for the whole file, not just that service. The long form is
|
|
# needed: `.env.alerting` is gitignored and absent in a fresh clone, and the
|
|
# short form treats a missing env file as an error.
|
|
#
|
|
# Usage:
|
|
# docker compose -f docker/telemetry/docker-compose.yml up -d
|
|
#
|
|
# Configure xrpld to export traces by adding to xrpld.cfg:
|
|
# [telemetry]
|
|
# enabled=1
|
|
# traces_endpoint=http://localhost:4318/v1/traces
|
|
|
|
services:
|
|
# One-shot init for the collector's offset store. Docker creates a fresh
|
|
# named volume owned by root, but the collector image runs as 10001:10001
|
|
# and ships no writable directory, so the file_storage extension could not
|
|
# create its database and the collector would fail to start. Chown the
|
|
# volume once, then exit; the collector waits for this to complete.
|
|
#
|
|
# Reuses the Prometheus image purely because the stack already pulls it and
|
|
# it has a shell — this adds no new image dependency. The entrypoint is
|
|
# overridden since that image normally starts the Prometheus server.
|
|
otelcol-storage-init:
|
|
image: prom/prometheus:v3.13.2
|
|
user: "0:0"
|
|
entrypoint: ["sh", "-c"]
|
|
command: ["mkdir -p /data/file_storage && chown -R 10001:10001 /data"]
|
|
volumes:
|
|
- otelcol-storage:/data
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# OpenTelemetry Collector: receives spans from xrpld via OTLP protocol,
|
|
# batches them for efficiency, and forwards to Tempo for storage.
|
|
otel-collector:
|
|
image: otel/opentelemetry-collector-contrib:0.158.0
|
|
# Second --config layers filelog offset persistence on top of the shared
|
|
# base config; the collector deep-merges them. Only this stack keeps its
|
|
# logs across restarts, so only this stack needs it.
|
|
command:
|
|
[
|
|
"--config=/etc/otel-collector-config.yaml",
|
|
"--config=/etc/otel-collector-filestorage.yaml",
|
|
]
|
|
ports:
|
|
- "4317:4317" # OTLP gRPC
|
|
- "4318:4318" # OTLP HTTP (traces + native OTel metrics)
|
|
- "8889:8889" # Prometheus metrics (spanmetrics + OTLP)
|
|
# StatsD UDP port removed — beast::insight now uses native OTLP.
|
|
# Uncomment if using server=statsd fallback:
|
|
# - "8125:8125/udp"
|
|
volumes:
|
|
# Mount collector pipeline config (receivers → processors → exporters)
|
|
- ./otel-collector-config.yaml:/etc/otel-collector-config.yaml:ro
|
|
# Dev-only overlay: persist filelog read offsets across restarts
|
|
- ./otel-collector-filestorage.yaml:/etc/otel-collector-filestorage.yaml:ro
|
|
# Mount the xrpld log root for the filelog receiver. The telemetry
|
|
# configs write to docker/telemetry/data/logs/<network>/debug.log, so
|
|
# the default source is the repo-relative ./data/logs — user-owned and
|
|
# needing no root, so `docker compose up` works with no setup. Override
|
|
# XRPLD_LOG_DIR to point at another root (e.g. the integration test sets
|
|
# it to its own workdir). Mounted read-only so the collector only tails.
|
|
- ${XRPLD_LOG_DIR:-./data/logs}:/var/log/xrpld:ro
|
|
# Persisted filelog read offsets, so a collector restart resumes
|
|
# instead of re-reading every debug.log from the top.
|
|
- otelcol-storage:/var/lib/otelcol
|
|
depends_on:
|
|
tempo:
|
|
condition: service_started
|
|
loki:
|
|
condition: service_started
|
|
otelcol-storage-init:
|
|
condition: service_completed_successfully
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana Tempo: distributed tracing backend that stores and indexes
|
|
# spans. Queryable via TraceQL in Grafana Explore.
|
|
tempo:
|
|
image: grafana/tempo:2.9.4
|
|
command: ["-config.file=/etc/tempo.yaml"]
|
|
ports:
|
|
- "3200:3200" # Tempo HTTP API (health check, query)
|
|
volumes:
|
|
# Mount Tempo storage and ingestion config
|
|
- ./tempo.yaml:/etc/tempo.yaml:ro
|
|
# Persistent volume for trace data (WAL + blocks)
|
|
- tempo-data:/var/tempo
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana Loki for centralized log ingestion and log-trace
|
|
# correlation. Loki 3.x supports native OTLP ingestion, so the OTel
|
|
# Collector exports via otlphttp to Loki's /otlp endpoint.
|
|
# Query logs via Grafana Explore -> Loki at http://localhost:3000.
|
|
loki:
|
|
image: grafana/loki:3.7.6
|
|
ports:
|
|
- "3100:3100"
|
|
command: -config.file=/etc/loki/local-config.yaml
|
|
volumes:
|
|
- loki-data:/loki
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
prometheus:
|
|
# Pinned to an exact patch release for reproducible, config-stable runs.
|
|
image: prom/prometheus:v3.13.2
|
|
ports:
|
|
- "9090:9090"
|
|
volumes:
|
|
- ./prometheus.yml:/etc/prometheus/prometheus.yml:ro
|
|
- prometheus-data:/prometheus
|
|
depends_on:
|
|
- otel-collector
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana: visualization UI with Tempo pre-configured as a datasource.
|
|
# Anonymous admin access enabled for local development convenience.
|
|
grafana:
|
|
image: grafana/grafana:13.1.2
|
|
# Alerting secrets/addresses (Slack webhook, alert email) come from this
|
|
# gitignored file; Grafana expands the ${VARS} referenced in the alerting
|
|
# provisioning YAML. Absent file = unset vars = receivers have no live
|
|
# destination, which is fine for a local stack. See .env.alerting.example.
|
|
env_file:
|
|
- path: .env.alerting
|
|
required: false
|
|
environment:
|
|
- GF_AUTH_ANONYMOUS_ENABLED=true # No login required for local dev
|
|
- GF_AUTH_ANONYMOUS_ORG_ROLE=Admin # Full access without auth
|
|
# Remote image rendering: point Grafana at the renderer container.
|
|
# These belong on the grafana service (the server delegates renders);
|
|
# the callback URL is how the renderer fetches the panel from grafana.
|
|
- GF_RENDERING_SERVER_URL=http://renderer:8081/render
|
|
- GF_RENDERING_CALLBACK_URL=http://grafana:3000/
|
|
# Grafana 13 turns the renderAuthJWT feature toggle on by default, which
|
|
# makes the renderer reject requests unless both sides share a token that
|
|
# is not the "-" default. Must match AUTH_TOKEN on the renderer service.
|
|
- GF_RENDERING_RENDERER_TOKEN=${GF_RENDERING_RENDERER_TOKEN:-xrpld-local-render}
|
|
# SMTP for the critical-tier email receiver and the Slack webhook / email
|
|
# address for the contact points all come from the env_file above, which
|
|
# injects them straight into the container environment for Grafana to
|
|
# expand. They are deliberately NOT duplicated here: a compose-level
|
|
# `environment:` entry is interpolated from the host shell and would
|
|
# override (and blank out) the env_file values.
|
|
ports:
|
|
- "3000:3000" # Grafana web UI
|
|
volumes:
|
|
# Auto-provision Tempo datasource and search filters on startup
|
|
- ./grafana/provisioning:/etc/grafana/provisioning:ro
|
|
- ./grafana/dashboards:/var/lib/grafana/dashboards:ro
|
|
depends_on:
|
|
- tempo
|
|
- prometheus
|
|
- loki
|
|
- renderer
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana image renderer: a sidecar that renders panels/dashboards to PNG
|
|
# for image export and alerting. Grafana calls it at http://renderer:8081.
|
|
renderer:
|
|
# Pinned like every other image in this stack so a stack that worked
|
|
# yesterday still works today. Upstream maintains only the newest release,
|
|
# which is expected to serve all supported Grafana versions.
|
|
image: grafana/grafana-image-renderer:v5.12.0
|
|
environment:
|
|
# Shared secret for the JWT-authenticated render requests Grafana 13
|
|
# sends. Must match GF_RENDERING_RENDERER_TOKEN on the grafana service.
|
|
- AUTH_TOKEN=${GF_RENDERING_RENDERER_TOKEN:-xrpld-local-render}
|
|
ports:
|
|
- "8081:8081" # Renderer HTTP endpoint (called by grafana)
|
|
networks:
|
|
- xrpld-telemetry
|
|
# Named volume for Tempo trace storage (WAL and compacted blocks).
|
|
# Data persists across container restarts. Remove with:
|
|
# docker compose -f docker/telemetry/docker-compose.yml down -v
|
|
volumes:
|
|
tempo-data:
|
|
prometheus-data:
|
|
loki-data:
|
|
otelcol-storage:
|
|
|
|
# Isolated bridge network so services communicate by container name
|
|
# (e.g., the collector reaches Tempo at http://tempo:4317).
|
|
networks:
|
|
xrpld-telemetry:
|
|
driver: bridge
|