Files
rippled/docker/telemetry/xrpld-telemetry.cfg
Pratik Mankawde 3860c93db2 refactor(telemetry): route dashboards, runbook and collector work to phase-9
These changes were developed on the phase-10 branch but belong to content this
branch and its upstreams introduced. Carrying them on phase-10 made its PR diff
report churn in files phase-10 does not own, and left each PR claiming a scope
that did not match its contents.

Moved here from phase-10 (identical content, no functional change):

- Dashboards: all 14 existing boards plus the new log-derived-insights board.
- Docs: telemetry-runbook.md (minus the workload/benchmark sections, which
  describe phase-10 tooling) and the new telemetry-glossary.md.
- Grafana Cloud + Alloy export path: collector config, compose override, the
  two .env examples and alloy/config.alloy.
- Local stack: otel-collector-config.yaml gains sub-millisecond and
  second-scale spanmetrics buckets, pins unit=ms, and promotes
  close_time_correct; integration-test.sh and TESTING.md follow.
- Node configs: exported_instance -> service_instance_id in comments; the
  mainnet sample now logs at warning to bound log volume.
- Metrics code: Telemetry.cpp builds the metrics pipeline in the constructor
  via initMetrics() so the global MeterProvider is published before any
  subsystem creates a beast::insight instrument, and the histogram view keeps
  each instrument's own name instead of collapsing them under one series.
  MetricsRegistry gains a last_close_time gauge and skips negative job-queue
  durations. OTelCollector drops an unused accessor.
- Naming CI: xrpl_work_item joins EXTERNAL_INFRA_LABELS and Rule E accepts the
  dotted perf-iac resource-attribute form. This must travel with the
  dashboards and runbook that reference those labels, or the rules fail.
- Doxygen input glob no longer recurses dot-directories.

Sections describing phase-10 tooling stay on phase-10 and keep their
"Future Enhancement" / "Planned, not yet implemented" markers here; phase-10
removes those markers when it lands the tooling.
2026-08-04 16:10:04 +01:00

137 lines
3.1 KiB
INI

# xrpld configuration for Devnet with full OpenTelemetry tracing.
#
# Connects to the XRP Ledger Devnet and exercises ALL instrumented
# workflows: RPC, transactions, consensus, peer overlay, ledger ops,
# and pathfinding.
#
# Usage:
# 1. Start the observability stack:
# docker compose -f docker/telemetry/docker-compose.yml up -d
# 2. Run xrpld:
# ./xrpld --conf docker/telemetry/xrpld-telemetry.cfg
# 3. Wait for sync (server_state=full), then exercise workflows:
# curl -s http://localhost:5005 -d '{"method":"server_info"}'
# 4. View traces in Grafana Explore -> Tempo: http://localhost:3000
# --- Server ports -----------------------------------------------------------
[server]
port_rpc_admin_local
port_ws_admin_local
port_ws_public
port_peer
[port_rpc_admin_local]
port = 5005
ip = 127.0.0.1
admin = 127.0.0.1
protocol = http
[port_ws_admin_local]
port = 6006
ip = 127.0.0.1
admin = 127.0.0.1
protocol = ws
[port_ws_public]
port = 6005
ip = 0.0.0.0
protocol = ws
[port_peer]
port = 51235
ip = 0.0.0.0
protocol = peer
# --- Network ----------------------------------------------------------------
[network_id]
devnet
[ips]
s.devnet.rippletest.net 51235
[validators_file]
validators-devnet.txt
[peer_private]
0
[peers_max]
21
# --- Pathfinding (exercises ripple_path_find / path_find workflows) ---------
[path_search]
7
[path_search_fast]
2
[path_search_max]
10
# --- Signing (allows sign/sign_for RPC for test tx submission) --------------
[signing_support]
true
# --- Database ---------------------------------------------------------------
[node_db]
type=NuDB
path=docker/telemetry/data/nudb
online_delete=2000
advisory_delete=0
[database_path]
docker/telemetry/data
[ledger_history]
1000
# --- Logging ----------------------------------------------------------------
# Path is resolved relative to this config file's directory (docker/telemetry),
# so this writes to docker/telemetry/data/logs/devnet/debug.log — the same
# dir the compose stack bind-mounts into the collector as /var/log/xrpld.
[debug_logfile]
data/logs/devnet/debug.log
[rpc_startup]
{ "command": "log_level", "severity": "debug" }
# --- SSL --------------------------------------------------------------------
[ssl_verify]
0
# --- Insight (native OTel metrics via beast::insight) -----------------------
[insight]
server=otel
endpoint=http://localhost:4318/v1/metrics
prefix=xrpld
# Sets the OTel service.instance.id resource attribute, which Prometheus
# exposes as the `service_instance_id` label. Dashboards filter on it via the
# $node template variable, so without this every insight-backed panel is
# empty. Matches [telemetry] service_instance_id for a single node identity.
service_instance_id=xrpld-devnet
# --- OpenTelemetry tracing --------------------------------------------------
[telemetry]
enabled=1
service_instance_id=xrpld-devnet
endpoint=http://localhost:4318/v1/traces
metrics_endpoint=http://localhost:4318/v1/metrics
exporter=otlp_http
batch_size=512
batch_delay_ms=5000
max_queue_size=2048
trace_rpc=1
trace_transactions=1
trace_consensus=1
trace_peer=1
trace_ledger=1