mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-28 07:48:01 +00:00
Addresses the open review findings on this branch. The log root was never delivered at all. Docker creates a missing bind-mount source as root, Config::getDebugLogFile() only warns when it cannot create the network subdirectory inside it, and Application carries on. The node therefore looked healthy while writing no debug.log, and Loki stayed empty with no error at any layer. docker/telemetry/data/logs has in fact been root-owned in a working checkout since it was first created. A one-shot xrpld-logdir-init service now creates the directory and hands it to XRPLD_UID/XRPLD_GID, following the pattern the storage-init service already uses. Ingested logs carried no node identity, so a multi-node stack collapsed into one indistinguishable stream while every dashboard filters on service_instance_id. The receiver now sets include_file_path and lifts the per-node directory onto the resource attribute service.instance.id, which is on the allow-list Loki promotes to an indexed stream label. A record attribute would only become structured metadata and could not be used in a selector. For that to join anything the directory name has to equal the emitter's service_instance_id, so the node directories are renamed to match: node$i becomes Node-$i, and the standalone config writes to logs/xrpld-standalone. The integration test aborted before reporting. Under set -o pipefail the grep | head -1 pipeline is killed by SIGPIPE once the log exceeds the pipe buffer, so the run exited 141 somewhere past a few hundred matching lines and read as a flaky test. grep -m1 stops on its own. The test also verified the local file and Tempo but never that a line reached Loki, which is the one hop this branch adds, so a bounded Loki assertion is added alongside a readiness wait. Documentation fixes: the Tempo cross-check counted .data, but Tempo returns OTLP shape so the array is batches and one trace can span several; the Loki step used the instant /query endpoint, which rejects a bare log selector with HTTP 400 and a text/plain body, so jq could never parse it and the step never printed a number even when ingestion worked. The filelog comment claimed six fractional digits where the node always emits nine. The two flowcharts used <br/>, carried no legend, and advertised GetSpan(), which Log.cpp deliberately avoids in favour of reading the thread-local context directly. Finally, rename the deprecated collector component names: the pinned collector warns on every start that otlphttp and filelog are aliases for otlp_http and file_log. Alloy's otelcol.exporter.otlphttp and otelcol.receiver.filelog are that product's own component names and are not deprecated, so they are left alone.
190 lines
7.3 KiB
YAML
190 lines
7.3 KiB
YAML
# Docker Compose stack for xrpld OpenTelemetry observability.
|
|
#
|
|
# Provides services for local development:
|
|
# - otel-collector: receives OTLP traces from xrpld, batches and
|
|
# forwards them to Tempo. Also tails xrpld log files
|
|
# via file_log receiver and exports to Loki. Listens on ports
|
|
# 4317 (gRPC) and 4318 (HTTP).
|
|
# - tempo: Grafana Tempo tracing backend, queryable via Grafana Explore
|
|
# on port 3000. Recommended for production (S3/GCS storage, TraceQL).
|
|
# - loki: Grafana Loki log aggregation backend for centralized log
|
|
# ingestion and log-trace correlation.
|
|
# - grafana: dashboards on port 3000, pre-configured with Tempo,
|
|
# Prometheus, and Loki datasources.
|
|
#
|
|
# Usage:
|
|
# docker compose -f docker/telemetry/docker-compose.yml up -d
|
|
#
|
|
# Configure xrpld to export traces by adding to xrpld.cfg:
|
|
# [telemetry]
|
|
# enabled=1
|
|
# traces_endpoint=http://localhost:4318/v1/traces
|
|
|
|
services:
|
|
# One-shot init for the collector's offset store. Docker creates a fresh
|
|
# named volume owned by root, but the collector image runs as 10001:10001
|
|
# and ships no writable directory, so the file_storage extension could not
|
|
# create its database and the collector would fail to start. Chown the
|
|
# volume once, then exit; the collector waits for this to complete.
|
|
#
|
|
# Reuses the Prometheus image purely because the stack already pulls it and
|
|
# it has a shell — this adds no new image dependency. The entrypoint is
|
|
# overridden since that image normally starts the Prometheus server.
|
|
otelcol-storage-init:
|
|
image: prom/prometheus:v3.13.2
|
|
user: "0:0"
|
|
entrypoint: ["sh", "-c"]
|
|
command: ["mkdir -p /data/file_storage && chown -R 10001:10001 /data"]
|
|
volumes:
|
|
- otelcol-storage:/data
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# One-shot init for the xrpld log root. Docker creates a missing bind-mount
|
|
# source as root, and xrpld then cannot create the <network> subdirectory
|
|
# inside it. Config::getDebugLogFile() only warns on that failure and carries
|
|
# on, so the node looks healthy while writing no debug.log at all and the
|
|
# whole log pipeline stays empty with no error at any layer. Create the
|
|
# directory here and hand it to the host user instead.
|
|
#
|
|
# XRPLD_UID/XRPLD_GID default to 1000, the first non-root user on a typical
|
|
# Linux host. Set them if `id -u` differs, or xrpld still cannot write.
|
|
# Reuses the Prometheus image for the same reason otelcol-storage-init does.
|
|
xrpld-logdir-init:
|
|
image: prom/prometheus:v3.13.2
|
|
user: "0:0"
|
|
entrypoint: ["sh", "-c"]
|
|
command:
|
|
[
|
|
"mkdir -p /data/logs && chown ${XRPLD_UID:-1000}:${XRPLD_GID:-1000} /data /data/logs",
|
|
]
|
|
volumes:
|
|
- ./data:/data
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# OpenTelemetry Collector: receives spans from xrpld via OTLP protocol,
|
|
# batches them for efficiency, and forwards to Tempo for storage.
|
|
otel-collector:
|
|
image: otel/opentelemetry-collector-contrib:0.158.0
|
|
# Second --config layers file_log offset persistence on top of the shared
|
|
# base config; the collector deep-merges them. Only this stack keeps its
|
|
# logs across restarts, so only this stack needs it.
|
|
command:
|
|
[
|
|
"--config=/etc/otel-collector-config.yaml",
|
|
"--config=/etc/otel-collector-filestorage.yaml",
|
|
]
|
|
ports:
|
|
- "4317:4317" # OTLP gRPC
|
|
- "4318:4318" # OTLP HTTP (traces + native OTel metrics)
|
|
- "8889:8889" # Prometheus metrics (spanmetrics + OTLP)
|
|
# StatsD UDP port removed — beast::insight now uses native OTLP.
|
|
# Uncomment if using server=statsd fallback:
|
|
# - "8125:8125/udp"
|
|
volumes:
|
|
# Mount collector pipeline config (receivers → processors → exporters)
|
|
- ./otel-collector-config.yaml:/etc/otel-collector-config.yaml:ro
|
|
# Dev-only overlay: persist file_log read offsets across restarts
|
|
- ./otel-collector-filestorage.yaml:/etc/otel-collector-filestorage.yaml:ro
|
|
# Mount the xrpld log root for the file_log receiver. The telemetry
|
|
# configs write to docker/telemetry/data/logs/<network>/debug.log, so
|
|
# the default source is the repo-relative ./data/logs, which
|
|
# xrpld-logdir-init has already created and handed to the host user.
|
|
# Override XRPLD_LOG_DIR to point at another root (e.g. the integration
|
|
# test sets it to its own workdir; that root is created by the test, so
|
|
# the init service is a no-op there). Mounted read-only so the collector
|
|
# only tails.
|
|
- ${XRPLD_LOG_DIR:-./data/logs}:/var/log/xrpld:ro
|
|
# Persisted file_log read offsets, so a collector restart resumes
|
|
# instead of re-reading every debug.log from the top.
|
|
- otelcol-storage:/var/lib/otelcol
|
|
depends_on:
|
|
tempo:
|
|
condition: service_started
|
|
loki:
|
|
condition: service_started
|
|
otelcol-storage-init:
|
|
condition: service_completed_successfully
|
|
xrpld-logdir-init:
|
|
condition: service_completed_successfully
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana Tempo: distributed tracing backend that stores and indexes
|
|
# spans. Queryable via TraceQL in Grafana Explore.
|
|
tempo:
|
|
image: grafana/tempo:2.9.4
|
|
command: ["-config.file=/etc/tempo.yaml"]
|
|
ports:
|
|
- "3200:3200" # Tempo HTTP API (health check, query)
|
|
volumes:
|
|
# Mount Tempo storage and ingestion config
|
|
- ./tempo.yaml:/etc/tempo.yaml:ro
|
|
# Persistent volume for trace data (WAL + blocks)
|
|
- tempo-data:/var/tempo
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana Loki for centralized log ingestion and log-trace
|
|
# correlation. Loki 3.x supports native OTLP ingestion, so the OTel
|
|
# Collector exports via otlp_http to Loki's /otlp endpoint.
|
|
# Query logs via Grafana Explore -> Loki at http://localhost:3000.
|
|
loki:
|
|
image: grafana/loki:3.4.2
|
|
ports:
|
|
- "3100:3100"
|
|
command: -config.file=/etc/loki/local-config.yaml
|
|
volumes:
|
|
- loki-data:/loki
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
prometheus:
|
|
# Pinned to an exact patch release for reproducible, config-stable runs.
|
|
image: prom/prometheus:v3.13.2
|
|
ports:
|
|
- "9090:9090"
|
|
volumes:
|
|
- ./prometheus.yml:/etc/prometheus/prometheus.yml:ro
|
|
- prometheus-data:/prometheus
|
|
depends_on:
|
|
- otel-collector
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Grafana: visualization UI with Tempo pre-configured as a datasource.
|
|
# Anonymous admin access enabled for local development convenience.
|
|
grafana:
|
|
image: grafana/grafana:13.1.2
|
|
environment:
|
|
- GF_AUTH_ANONYMOUS_ENABLED=true # No login required for local dev
|
|
- GF_AUTH_ANONYMOUS_ORG_ROLE=Admin # Full access without auth
|
|
ports:
|
|
- "3000:3000" # Grafana web UI
|
|
volumes:
|
|
# Auto-provision Tempo datasource and search filters on startup
|
|
- ./grafana/provisioning:/etc/grafana/provisioning:ro
|
|
- ./grafana/dashboards:/var/lib/grafana/dashboards:ro
|
|
depends_on:
|
|
- tempo
|
|
- prometheus
|
|
- loki
|
|
networks:
|
|
- xrpld-telemetry
|
|
|
|
# Named volume for Tempo trace storage (WAL and compacted blocks).
|
|
# Data persists across container restarts. Remove with:
|
|
# docker compose -f docker/telemetry/docker-compose.yml down -v
|
|
volumes:
|
|
tempo-data:
|
|
prometheus-data:
|
|
loki-data:
|
|
otelcol-storage:
|
|
|
|
# Isolated bridge network so services communicate by container name
|
|
# (e.g., the collector reaches Tempo at http://tempo:4317).
|
|
networks:
|
|
xrpld-telemetry:
|
|
driver: bridge
|