mirror of
https://github.com/XRPLF/rippled.git
synced 2026-09-27 23:38:08 +00:00
ff8629bb11 dropped prefix=xrpld as inert and misleading, but four OTel-path
sites still set it, so the branch contradicted itself.
OTelCollector routes every instrument name through a static formatName()
that only lowercases and maps '.'/space to '_'; the sole read of prefix_ is
the startup log line at OTelCollector.cpp:810. All four instrument factories
funnel through formatName(), so no prefix can reach an exported name.
StatsDCollector does prepend it (StatsDCollector.cpp:551/592/640/715), so the
StatsD example legitimately keeps it.
Removed from the 09 reference's OTel config block and from both
quick-reference setups, and from the cfg integration-test.sh generates. The
StatsD example is unchanged and now states why it keeps the key.
Also corrected run-full-validation.sh: [insight] endpoint was described as
"already matches the built-in default", implying it would matter if it
differed. CollectorManager reads it and hands it to OTelCollector, which also
only logs it; the exporter URL is built in Telemetry::initMetrics() from
[telemetry] endpoint. It is as inert as prefix was.
663 lines
25 KiB
Bash
Executable File
663 lines
25 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# run-full-validation.sh — Orchestrates the full telemetry validation pipeline.
|
|
#
|
|
# Sequence:
|
|
# 1. Start the observability stack (OTel Collector, Tempo, Prometheus, Loki, Grafana)
|
|
# 2. Start a multi-node rippled cluster with full telemetry enabled
|
|
# 3. Wait for consensus
|
|
# 4. Run workload orchestrator (RPC load, TX submission, propagation wait)
|
|
# 5. Run the telemetry validation suite
|
|
# 6. Capture OTel timings and compare against committed baseline
|
|
# 7. (Optional) Run the performance overhead benchmark
|
|
#
|
|
# Usage:
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld --with-benchmark
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld --skip-regression
|
|
# ./run-full-validation.sh --cleanup
|
|
#
|
|
# Exit codes:
|
|
# 0 — All validation checks and the regression gate passed
|
|
# 1 — Validation checks failed OR the regression gate detected a regression
|
|
# OR the benchmark exceeded its overhead thresholds
|
|
# 2 — Infrastructure error (cluster/stack failed to start, workload
|
|
# orchestration failed, timing capture failed, overhead could not be
|
|
# measured)
|
|
#
|
|
# Every step below records its status and folds it into FINAL_EXIT; the first
|
|
# non-zero status in pipeline order is the one returned, so the earliest
|
|
# failure — the one that explains the later ones — is what the caller sees.
|
|
|
|
set -euo pipefail
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Colored output helpers
|
|
# ---------------------------------------------------------------------------
|
|
log() { printf "\033[1;34m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
ok() { printf "\033[1;32m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
warn() { printf "\033[1;33m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
fail() { printf "\033[1;31m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
die() {
|
|
printf "\033[1;31m[VALIDATE]\033[0m %s\n" "$*" >&2
|
|
exit 2
|
|
}
|
|
|
|
# Overall run status, folded step by step (see the exit-code table above).
|
|
FINAL_EXIT=0
|
|
|
|
# fold_exit STATUS — record a step's status in FINAL_EXIT.
|
|
# First non-zero wins, so FINAL_EXIT names the earliest failing step.
|
|
fold_exit() {
|
|
if [ "$1" -ne 0 ] && [ "$FINAL_EXIT" -eq 0 ]; then
|
|
FINAL_EXIT="$1"
|
|
fi
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Configuration
|
|
# ---------------------------------------------------------------------------
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
TELEMETRY_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
|
|
REPO_ROOT="$(cd "$TELEMETRY_DIR/../.." && pwd)"
|
|
COMPOSE_FILE="$TELEMETRY_DIR/docker-compose.workload.yaml"
|
|
WORKDIR="/tmp/xrpld-validation"
|
|
|
|
XRPLD="${XRPLD:-$REPO_ROOT/.build/xrpld}"
|
|
NUM_NODES=5
|
|
RPC_PORT_BASE=5005
|
|
WS_PORT_BASE=6006
|
|
PEER_PORT_BASE=51235
|
|
# Inert: parsed from --rpc-rate/--rpc-duration/--tx-tps/--tx-duration and never
|
|
# read again. Load shape comes from the workload profile instead. Kept because
|
|
# the CI workflow still passes the four flags.
|
|
RPC_RATE=50
|
|
RPC_DURATION=120
|
|
TX_TPS=5
|
|
TX_DURATION=120
|
|
WITH_BENCHMARK=false
|
|
SKIP_LOKI=false
|
|
SKIP_REGRESSION=false
|
|
WORKLOAD_PROFILE="full-validation"
|
|
REPORT_DIR="$WORKDIR/reports"
|
|
# Rate window handed to Prometheus `rate()` when capturing timings. Keep
|
|
# this close to the active workload duration so histogram buckets cover
|
|
# the measurement window; longer windows dilute short-lived regressions.
|
|
REGRESSION_WINDOW="${REGRESSION_WINDOW:-3m}"
|
|
BASELINE_FILE="${BASELINE_FILE:-$SCRIPT_DIR/baselines/baseline-timings.json}"
|
|
THRESHOLDS_FILE="${THRESHOLDS_FILE:-$SCRIPT_DIR/regression-thresholds.json}"
|
|
METRICS_FILE="${METRICS_FILE:-$SCRIPT_DIR/regression-metrics.json}"
|
|
|
|
GENESIS_ACCOUNT="rHb9CJAWyB4rj91VRWn96DkukG4bwdtyTh"
|
|
GENESIS_SEED="snoPBrXtMeMyMHUVTgbuqAfg1SUTb"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Argument parsing
|
|
# ---------------------------------------------------------------------------
|
|
usage() {
|
|
echo "Usage: $0 [OPTIONS]"
|
|
echo ""
|
|
echo "Options:"
|
|
echo " --xrpld PATH Path to xrpld binary"
|
|
echo " --nodes NUM Number of validator nodes (default: 5)"
|
|
echo " --profile NAME Workload profile (default: full-validation)"
|
|
echo " --with-benchmark Also run performance overhead benchmark (telemetry off vs on)"
|
|
echo " --skip-loki Skip Loki log-trace correlation checks"
|
|
echo " --skip-regression Skip the OTel-baseline regression gate"
|
|
echo " --cleanup Tear down everything and exit"
|
|
echo " -h, --help Show this help"
|
|
echo ""
|
|
echo "Accepted but INERT (parsed for compatibility, then ignored):"
|
|
echo " --rpc-rate RPS no effect"
|
|
echo " --rpc-duration SECS no effect"
|
|
echo " --tx-tps TPS no effect"
|
|
echo " --tx-duration SECS no effect"
|
|
echo ""
|
|
echo " Load shape comes from the workload profile (--profile), which sets"
|
|
echo " the rate and duration of every phase in workload-profiles.json."
|
|
echo " These four flags stay accepted because the CI workflow passes them."
|
|
exit 0
|
|
}
|
|
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--xrpld)
|
|
XRPLD="$2"
|
|
shift 2
|
|
;;
|
|
--nodes)
|
|
NUM_NODES="$2"
|
|
shift 2
|
|
;;
|
|
# The next four are inert — see the RPC_RATE default above.
|
|
--rpc-rate)
|
|
RPC_RATE="$2"
|
|
shift 2
|
|
;;
|
|
--rpc-duration)
|
|
RPC_DURATION="$2"
|
|
shift 2
|
|
;;
|
|
--tx-tps)
|
|
TX_TPS="$2"
|
|
shift 2
|
|
;;
|
|
--tx-duration)
|
|
TX_DURATION="$2"
|
|
shift 2
|
|
;;
|
|
--profile)
|
|
WORKLOAD_PROFILE="$2"
|
|
shift 2
|
|
;;
|
|
--with-benchmark)
|
|
WITH_BENCHMARK=true
|
|
shift
|
|
;;
|
|
--skip-loki)
|
|
SKIP_LOKI=true
|
|
shift
|
|
;;
|
|
--skip-regression)
|
|
SKIP_REGRESSION=true
|
|
shift
|
|
;;
|
|
--cleanup) # Cleanup mode
|
|
log "Cleaning up..."
|
|
# Match the node config path, not the bare workdir: a plain
|
|
# "$WORKDIR" pattern also matches any shell, editor or log tail
|
|
# whose command line merely mentions that path.
|
|
pkill -f "$WORKDIR/node[0-9]+/xrpld\.cfg" 2>/dev/null || true
|
|
docker compose -f "$COMPOSE_FILE" down 2>/dev/null || true
|
|
rm -rf "$WORKDIR"
|
|
ok "Cleanup complete."
|
|
exit 0
|
|
;;
|
|
-h | --help) usage ;;
|
|
*) die "Unknown option: $1" ;;
|
|
esac
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Prerequisites
|
|
# ---------------------------------------------------------------------------
|
|
log "Checking prerequisites..."
|
|
[ -x "$XRPLD" ] || die "xrpld binary not found: $XRPLD"
|
|
command -v docker >/dev/null 2>&1 || die "docker not found"
|
|
docker compose version >/dev/null 2>&1 || die "docker compose (v2) not found"
|
|
command -v python3 >/dev/null 2>&1 || die "python3 not found"
|
|
command -v curl >/dev/null 2>&1 || die "curl not found"
|
|
command -v jq >/dev/null 2>&1 || die "jq not found"
|
|
[ -f "$COMPOSE_FILE" ] || die "docker-compose.workload.yaml not found"
|
|
|
|
# Install Python dependencies.
|
|
log "Installing Python dependencies..."
|
|
pip3 install -q -r "$SCRIPT_DIR/requirements.txt" 2>/dev/null ||
|
|
pip install -q -r "$SCRIPT_DIR/requirements.txt" 2>/dev/null ||
|
|
warn "Could not install Python dependencies — they may already be present"
|
|
|
|
ok "Prerequisites verified."
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Cleanup previous run
|
|
# ---------------------------------------------------------------------------
|
|
log "Cleaning up previous run..."
|
|
# Narrowed for the same reason as the --cleanup branch above.
|
|
pkill -f "$WORKDIR/node[0-9]+/xrpld\.cfg" 2>/dev/null || true
|
|
sleep 2
|
|
rm -rf "$WORKDIR"
|
|
mkdir -p "$WORKDIR" "$REPORT_DIR"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 1: Start observability stack
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 1: Starting observability stack..."
|
|
# Point the collector's log mount at this run's workdir so the filelog
|
|
# receiver tails the per-node debug.log files generated below.
|
|
XRPLD_LOG_DIR="$WORKDIR" docker compose -f "$COMPOSE_FILE" up -d
|
|
|
|
log "Waiting for OTel Collector..."
|
|
for attempt in $(seq 1 30); do
|
|
status=$(curl -so /dev/null -w '%{http_code}' http://localhost:4318/ 2>/dev/null || echo 000)
|
|
if [ "$status" != "000" ]; then
|
|
ok "OTel Collector ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "OTel Collector not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
log "Waiting for Tempo..."
|
|
for attempt in $(seq 1 30); do
|
|
if curl -sf "http://localhost:3200/ready" >/dev/null 2>&1; then
|
|
ok "Tempo ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "Tempo not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
log "Waiting for Prometheus..."
|
|
for attempt in $(seq 1 30); do
|
|
if curl -sf "http://localhost:9090/-/healthy" >/dev/null 2>&1; then
|
|
ok "Prometheus ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "Prometheus not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 2: Generate validator keys and start cluster
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 2: Starting $NUM_NODES-node validator cluster..."
|
|
|
|
bash "$SCRIPT_DIR/generate-validator-keys.sh" "$XRPLD" "$NUM_NODES" "$WORKDIR"
|
|
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
NODE_DIR="$WORKDIR/node$i"
|
|
mkdir -p "$NODE_DIR/nudb" "$NODE_DIR/db"
|
|
|
|
RPC_PORT=$((RPC_PORT_BASE + i - 1))
|
|
WS_PORT=$((WS_PORT_BASE + i - 1))
|
|
PEER_PORT=$((PEER_PORT_BASE + i - 1))
|
|
SEED=$(jq -r ".[$((i - 1))].seed" "$WORKDIR/validator-keys.json")
|
|
|
|
# Build ips_fixed.
|
|
IPS_FIXED=""
|
|
for j in $(seq 1 "$NUM_NODES"); do
|
|
if [ "$j" -ne "$i" ]; then
|
|
IPS_FIXED="${IPS_FIXED}127.0.0.1 $((PEER_PORT_BASE + j - 1))
|
|
"
|
|
fi
|
|
done
|
|
|
|
cat >"$NODE_DIR/xrpld.cfg" <<EOCFG
|
|
[server]
|
|
port_rpc
|
|
port_ws
|
|
port_peer
|
|
|
|
[port_rpc]
|
|
port = $RPC_PORT
|
|
ip = 127.0.0.1
|
|
admin = 127.0.0.1
|
|
protocol = http
|
|
|
|
[port_ws]
|
|
port = $WS_PORT
|
|
ip = 127.0.0.1
|
|
admin = 127.0.0.1
|
|
protocol = ws
|
|
|
|
[port_peer]
|
|
port = $PEER_PORT
|
|
ip = 0.0.0.0
|
|
protocol = peer
|
|
|
|
[node_db]
|
|
type=NuDB
|
|
path=$NODE_DIR/nudb
|
|
online_delete=256
|
|
|
|
[database_path]
|
|
$NODE_DIR/db
|
|
|
|
[debug_logfile]
|
|
$NODE_DIR/debug.log
|
|
|
|
[validation_seed]
|
|
$SEED
|
|
|
|
[validators_file]
|
|
$WORKDIR/validators.txt
|
|
|
|
[ips]
|
|
${IPS_FIXED}
|
|
|
|
[telemetry]
|
|
enabled=1
|
|
service_instance_id=validator-${i}
|
|
endpoint=http://localhost:4318/v1/traces
|
|
batch_size=512
|
|
batch_delay_ms=2000
|
|
max_queue_size=2048
|
|
trace_rpc=1
|
|
trace_transactions=1
|
|
trace_consensus=1
|
|
trace_peer=1
|
|
trace_ledger=1
|
|
|
|
[insight]
|
|
# Native OTel metrics via OTLP/HTTP. The collector has no StatsD receiver
|
|
# (its metrics pipeline is [otlp, spanmetrics]), so beast::insight must export
|
|
# over OTLP for system metrics to reach Prometheus at all. server=otel is the
|
|
# only load-bearing key here -- it selects the OTel collector.
|
|
#
|
|
# endpoint below is parsed but inert, exactly like prefix: CollectorManager
|
|
# reads it and hands it to OTelCollector, which uses it only in its startup log
|
|
# line. The URL the metric exporter actually posts to is derived in
|
|
# Telemetry::initMetrics() from [telemetry] endpoint, by swapping the trailing
|
|
# /v1/traces for /v1/metrics. It is kept here so the log line names the right
|
|
# URL; changing it would not redirect a single metric.
|
|
#
|
|
# No prefix is set on purpose. It would be inert: OTelCollector applies no
|
|
# prefix to instrument names, so exported names are the lowercased raw names
|
|
# (jobq_job_count, rpc_requests_total, total_bytes_in). The service is
|
|
# identified by the OTel resource service.name, not by a name prefix.
|
|
server=otel
|
|
endpoint=http://localhost:4318/v1/metrics
|
|
|
|
[rpc_startup]
|
|
# info is required here, not cosmetic, and is the minimum that works. The
|
|
# log-trace correlation checks need a log line emitted while a span is current
|
|
# on the thread. A span becomes current either as a ScopedSpanGuard or by
|
|
# activating a plain SpanGuard via activate()/activateIfLive(); a plain
|
|
# SpanGuard that is never activated yields no trace_id.
|
|
#
|
|
# The guaranteed correlated line at info is the consensus accept pair in
|
|
# RCLConsensus.cpp (an if/else, so exactly one fires every accepted round).
|
|
# doAccept activates the accept span as ambient over its whole body, so both
|
|
# branches carry a trace_id. At ~4s per round that is dozens of correlated
|
|
# lines per run.
|
|
#
|
|
# Do NOT raise this to debug to get more coverage. debug puts synchronous log
|
|
# I/O inside ledger.build, consensus.accept and tx.apply -- the same spans
|
|
# whose p50/p95/p99 latencies regression-metrics.json gates -- so a baseline
|
|
# captured at debug would bake log I/O into the numbers permanently.
|
|
{ "command": "log_level", "severity": "info" }
|
|
|
|
[signing_support]
|
|
true
|
|
|
|
[ssl_verify]
|
|
0
|
|
EOCFG
|
|
|
|
"$XRPLD" --conf "$NODE_DIR/xrpld.cfg" --start >"$NODE_DIR/stdout.log" 2>&1 &
|
|
echo $! >"$NODE_DIR/xrpld.pid"
|
|
log " Node $i: RPC=$RPC_PORT WS=$WS_PORT Peer=$PEER_PORT PID=$!"
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 3: Wait for consensus
|
|
# ---------------------------------------------------------------------------
|
|
# Report whether a node process is still alive.
|
|
#
|
|
# A child that has exited but not yet been waited on still answers `kill -0`,
|
|
# because the zombie keeps its pid until someone collects it. Checking only
|
|
# `kill -0` therefore reads a dead node as alive for the whole readiness
|
|
# window, which is how a crashed node came to look like a slow one.
|
|
node_running() {
|
|
local pid="$1" state
|
|
kill -0 "$pid" 2>/dev/null || return 1
|
|
if [ -r "/proc/$pid/stat" ]; then
|
|
state=$(awk '{print $3}' "/proc/$pid/stat" 2>/dev/null || echo "?")
|
|
[ "$state" != "Z" ] || return 1
|
|
fi
|
|
return 0
|
|
}
|
|
|
|
# Print why each stopped node stopped: its wait status, then its last output.
|
|
#
|
|
# The status is the discriminator this harness was missing -- 137 is SIGKILL
|
|
# (the kernel reclaiming memory), 139 a segfault, 134 an abort, anything under
|
|
# 128 a deliberate exit. The nodes are direct children of this script, so their
|
|
# status is still retrievable until something waits on them.
|
|
#
|
|
# stdout is printed inline rather than left to the artifact upload because a
|
|
# node that dies before its debug log opens writes nothing else, and a
|
|
# cancelled run uploads nothing at all.
|
|
report_stopped_nodes() {
|
|
local i pid status
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
pid=$(cat "$WORKDIR/node$i/xrpld.pid" 2>/dev/null || echo "")
|
|
[ -n "$pid" ] || continue
|
|
node_running "$pid" && continue
|
|
status=0
|
|
wait "$pid" 2>/dev/null || status=$?
|
|
warn "node$i (pid $pid) is not running — wait status $status"
|
|
if [ -s "$WORKDIR/node$i/stdout.log" ]; then
|
|
warn "node$i last output:"
|
|
tail -n 15 "$WORKDIR/node$i/stdout.log" | sed 's/^/ /' >&2
|
|
else
|
|
warn "node$i wrote no stdout at all"
|
|
fi
|
|
done
|
|
}
|
|
|
|
log "Step 3: Waiting for consensus..."
|
|
for attempt in $(seq 1 120); do
|
|
ready=0
|
|
# Reset each attempt so a timeout reports the final state, not a history.
|
|
laggards=""
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
port=$((RPC_PORT_BASE + i - 1))
|
|
state=$(curl -sf "http://localhost:$port" \
|
|
-d '{"method":"server_info"}' 2>/dev/null |
|
|
jq -r '.result.info.server_state' 2>/dev/null || echo "")
|
|
if [ "$state" = "proposing" ]; then
|
|
ready=$((ready + 1))
|
|
else
|
|
# Name the node and what it last reported. A bare count says a
|
|
# node is missing but not which one, which leaves nothing to grep
|
|
# for in the artifacts. An empty state means the RPC port did not
|
|
# answer at all, which usually means the process is gone.
|
|
laggards="$laggards node$i=${state:-unreachable}"
|
|
fi
|
|
done
|
|
if [ "$ready" -ge "$NUM_NODES" ]; then
|
|
ok "All $NUM_NODES nodes proposing (attempt $attempt)"
|
|
break
|
|
fi
|
|
# A stopped process will never reach proposing. Waiting out the rest of the
|
|
# window only delays the same failure and buries its cause under two
|
|
# minutes of progress output.
|
|
stopped=0
|
|
for n in $(seq 1 "$NUM_NODES"); do
|
|
p=$(cat "$WORKDIR/node$n/xrpld.pid" 2>/dev/null || echo "")
|
|
if [ -n "$p" ] && ! node_running "$p"; then
|
|
stopped=$((stopped + 1))
|
|
fi
|
|
done
|
|
if [ "$stopped" -gt 0 ]; then
|
|
echo ""
|
|
report_stopped_nodes
|
|
die "$stopped of $NUM_NODES node(s) stopped during startup; only $ready reached proposing. Not proposing:${laggards}. Per-node status is above, then '$0 --cleanup'."
|
|
fi
|
|
if [ "$attempt" -eq 120 ]; then
|
|
# Fatal, not a warning. A partial cluster still answers queries, so the
|
|
# run would complete and report unrelated span/metric failures: series
|
|
# counts scale with the number of live nodes, and spans that need a
|
|
# quorum are simply never emitted. One infrastructure error here is
|
|
# worth more than a pile of misleading assertion failures later.
|
|
echo ""
|
|
# Every node is still running but not proposing, so this is a genuine
|
|
# convergence problem rather than a crash. Run the reporter anyway: it
|
|
# is a no-op when nothing stopped, and it costs nothing to be sure.
|
|
report_stopped_nodes
|
|
die "Consensus timeout — only $ready/$NUM_NODES nodes proposing after ${attempt}s. Not proposing:${laggards}. Check $WORKDIR/node*/debug.log and $WORKDIR/node*/stdout.log (a node that died before its log sink opened writes only the latter), then '$0 --cleanup'."
|
|
fi
|
|
printf "\r %d/%d nodes proposing..." "$ready" "$NUM_NODES"
|
|
sleep 1
|
|
done
|
|
echo ""
|
|
|
|
# Wait for first validated ledger.
|
|
log "Waiting for validated ledger..."
|
|
for attempt in $(seq 1 60); do
|
|
val_seq=$(curl -sf "http://localhost:$RPC_PORT_BASE" \
|
|
-d '{"method":"server_info"}' 2>/dev/null |
|
|
jq -r '.result.info.validated_ledger.seq // 0' 2>/dev/null || echo 0)
|
|
if [ "$val_seq" -gt 2 ] 2>/dev/null; then
|
|
ok "Validated ledger: seq $val_seq"
|
|
break
|
|
fi
|
|
# Fatal for the same reason as the consensus timeout above, and because
|
|
# several assertions are gated on a validated ledger existing at all:
|
|
# ledger_economy{metric="base_fee_xrp"} is only observed from a validated
|
|
# ledger, and complete_ledgers stays absent while the range is empty.
|
|
if [ "$attempt" -eq 60 ]; then
|
|
die "No validated ledger after ${attempt}s (last seq: $val_seq). Check $WORKDIR/node*/debug.log, then '$0 --cleanup'."
|
|
fi
|
|
sleep 1
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 4: Run workload orchestrator
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 4: Running workload orchestrator (profile: $WORKLOAD_PROFILE)..."
|
|
|
|
WS_ENDPOINTS=""
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
WS_ENDPOINTS="$WS_ENDPOINTS ws://localhost:$((WS_PORT_BASE + i - 1))"
|
|
done
|
|
|
|
ORCHESTRATOR_EXIT=0
|
|
python3 "$SCRIPT_DIR/workload_orchestrator.py" \
|
|
--profile "$WORKLOAD_PROFILE" \
|
|
--endpoints $WS_ENDPOINTS \
|
|
--report "$REPORT_DIR/workload-report.json" \
|
|
--report-dir "$REPORT_DIR" || ORCHESTRATOR_EXIT=$?
|
|
|
|
if [ "$ORCHESTRATOR_EXIT" -eq 0 ]; then
|
|
ok "Workload orchestration complete."
|
|
else
|
|
# Treated as an infrastructure error: the span and metric assertions below
|
|
# would be graded against traffic that was never generated.
|
|
fail "Workload orchestrator failed (exit $ORCHESTRATOR_EXIT) — the checks below run against incomplete traffic"
|
|
fold_exit 2
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 5: Run telemetry validation suite
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 5: Running telemetry validation suite..."
|
|
|
|
VALIDATION_ARGS="--report $REPORT_DIR/validation-report.json"
|
|
if [ "$SKIP_LOKI" = true ]; then
|
|
VALIDATION_ARGS="$VALIDATION_ARGS --skip-loki"
|
|
fi
|
|
|
|
VALIDATION_EXIT=0
|
|
python3 "$SCRIPT_DIR/validate_telemetry.py" $VALIDATION_ARGS || VALIDATION_EXIT=$?
|
|
|
|
if [ "$VALIDATION_EXIT" -eq 0 ]; then
|
|
ok "All telemetry validation checks passed!"
|
|
else
|
|
fail "Some telemetry validation checks failed (exit $VALIDATION_EXIT)"
|
|
fi
|
|
fold_exit "$VALIDATION_EXIT"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 6: Capture OTel timings and run the regression comparison
|
|
# ---------------------------------------------------------------------------
|
|
# This step ALWAYS captures timings (so CI always has an artifact from which
|
|
# to bootstrap/refresh the committed baseline). The comparator then either:
|
|
# - prints the paste-me JSON when the baseline is a placeholder, or
|
|
# - enforces thresholds and fails the run on regression.
|
|
# Use --skip-regression to opt out (e.g. for ad-hoc local exploration).
|
|
TIMINGS_FILE="$REPORT_DIR/timings.json"
|
|
REGRESSION_REPORT="$REPORT_DIR/regression-report.json"
|
|
REGRESSION_EXIT=0
|
|
|
|
if [ "$SKIP_REGRESSION" != true ]; then
|
|
log "Step 6: Capturing OTel timings from Prometheus..."
|
|
if python3 "$SCRIPT_DIR/capture_timings.py" \
|
|
--prometheus "http://localhost:9090" \
|
|
--metrics "$METRICS_FILE" \
|
|
--output "$TIMINGS_FILE" \
|
|
--window "$REGRESSION_WINDOW" \
|
|
--profile "$WORKLOAD_PROFILE"; then
|
|
ok "Timings captured: $TIMINGS_FILE"
|
|
else
|
|
fail "Failed to capture timings — skipping regression comparison."
|
|
REGRESSION_EXIT=2
|
|
SKIP_REGRESSION=true
|
|
fi
|
|
fi
|
|
|
|
if [ "$SKIP_REGRESSION" != true ]; then
|
|
log "Comparing against baseline $BASELINE_FILE..."
|
|
python3 "$SCRIPT_DIR/compare_to_baseline.py" \
|
|
--timings "$TIMINGS_FILE" \
|
|
--baseline "$BASELINE_FILE" \
|
|
--thresholds "$THRESHOLDS_FILE" \
|
|
--report "$REGRESSION_REPORT" || REGRESSION_EXIT=$?
|
|
if [ "$REGRESSION_EXIT" -eq 0 ]; then
|
|
ok "Regression gate passed (or baseline placeholder — paste JSON printed above)."
|
|
elif [ "$REGRESSION_EXIT" -eq 1 ]; then
|
|
fail "Regression detected — see $REGRESSION_REPORT"
|
|
else
|
|
fail "Regression comparator internal error (exit $REGRESSION_EXIT)"
|
|
fi
|
|
else
|
|
warn "Regression gate skipped."
|
|
fi
|
|
fold_exit "$REGRESSION_EXIT"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 7: (Optional) Run overhead benchmark
|
|
# ---------------------------------------------------------------------------
|
|
BENCHMARK_EXIT=0
|
|
if [ "$WITH_BENCHMARK" = true ]; then
|
|
log "Step 7: Running performance benchmark..."
|
|
bash "$SCRIPT_DIR/benchmark.sh" \
|
|
--xrpld "$XRPLD" \
|
|
--duration 120 \
|
|
--nodes 3 \
|
|
--output "$REPORT_DIR" || BENCHMARK_EXIT=$?
|
|
|
|
if [ "$BENCHMARK_EXIT" -eq 0 ]; then
|
|
ok "Benchmark within overhead thresholds."
|
|
elif [ "$BENCHMARK_EXIT" -eq 1 ]; then
|
|
# A measured threshold breach — same class as a failed check.
|
|
fail "Benchmark exceeded overhead thresholds (exit 1)"
|
|
fold_exit 1
|
|
else
|
|
# benchmark.sh could not produce a usable measurement (e.g. incomplete
|
|
# system metrics). Reported as an infrastructure error, not a perf
|
|
# regression: nothing was measured, so nothing was breached.
|
|
fail "Benchmark could not measure overhead (exit $BENCHMARK_EXIT) — treated as an infrastructure error"
|
|
fold_exit 2
|
|
fi
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Summary
|
|
# ---------------------------------------------------------------------------
|
|
echo ""
|
|
echo "==========================================================="
|
|
echo " FULL VALIDATION RESULTS"
|
|
echo "==========================================================="
|
|
echo ""
|
|
echo " Reports directory: $REPORT_DIR"
|
|
echo ""
|
|
ls -la "$REPORT_DIR/" 2>/dev/null || true
|
|
echo ""
|
|
echo " Observability stack is running:"
|
|
echo " Tempo: http://localhost:3200"
|
|
echo " Grafana: http://localhost:3000"
|
|
echo " Prometheus: http://localhost:9090"
|
|
echo ""
|
|
echo " xrpld nodes ($NUM_NODES) are running:"
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
rpc=$((RPC_PORT_BASE + i - 1))
|
|
ws=$((WS_PORT_BASE + i - 1))
|
|
pid=$(cat "$WORKDIR/node$i/xrpld.pid" 2>/dev/null || echo 'unknown')
|
|
echo " Node $i: RPC=$rpc WS=$ws PID=$pid"
|
|
done
|
|
echo ""
|
|
echo " To tear down:"
|
|
echo " $0 --cleanup"
|
|
echo ""
|
|
echo " Step statuses (0 = ok):"
|
|
echo " Workload orchestration: $ORCHESTRATOR_EXIT"
|
|
echo " Telemetry validation: $VALIDATION_EXIT"
|
|
echo " Regression gate: $REGRESSION_EXIT"
|
|
echo " Overhead benchmark: $BENCHMARK_EXIT"
|
|
echo ""
|
|
echo "==========================================================="
|
|
|
|
# FINAL_EXIT already holds the first non-zero step status (see fold_exit).
|
|
exit "$FINAL_EXIT"
|