mirror of
https://github.com/XRPLF/rippled.git
synced 2026-08-21 14:20:56 +00:00
Fixes the review findings on this PR that belong to files it owns, plus several defects found while verifying those fixes. Findings in files owned by upstream branches are routed there and left untouched here. Correctness: - tx_submitter: advance the account sequence only on results that actually consume one (tes*, tec*, terQUEUED). tem*/tef*/tel* never reach the ledger, so advancing left a permanent gap that every later submit from that account inherited. Add a re-fetch hatch so a repeated non-consuming failure cannot livelock on the same sequence, and gate the account check on funded-ness rather than list length. - validate_telemetry: filter spans by name before collecting attributes, so a per-span attribute contract can no longer be satisfied by a sibling span; require exact name equality for non-wildcard children and glob matching for wildcards; bounds-check every returned series instead of only the first. - collect_system_metrics: select xrpld by argv[0] rather than a substring match on the whole command line, which averaged in unrelated processes and reported their RSS as xrpld's. Count genuine 0.0 CPU readings, use a clamped nearest-rank p99 index, and record RPC latency only on success. - benchmark: return each verdict through a named variable instead of a command substitution, so the pass/fail counters survive and the exit gate can fire. Scale before dividing in the percentage math, which truncated a 1.26% impact to 1.00% and cleared a 1% threshold. - compare_to_baseline: fall back to the absolute bound when the baseline is not positive, so a 0 -> 500 ms jump is no longer "within bounds". - rpc_load_generator: bound each connection to one in-flight recv(), drain in-flight requests before closing, use a nearest-rank percentile, and report delivery shortfall so an under-delivered run cannot pass with a 0% error rate. Fail loudly instead of silently: - run-full-validation: treat a consensus timeout and a missing validated ledger as fatal infrastructure errors, and fold the orchestrator and benchmark exit codes into the final status. A degraded cluster previously ran a full validation pass and reported misleading downstream failures. - collect_system_metrics: warn per empty measurement source, emit metrics_complete, and exit non-zero instead of substituting zeros that pass every threshold. Require GNU date with %N rather than falling back to a per-sample python3 fork that costs more than the threshold it is measured against. - benchmark: distinguish "could not measure" from "exceeded thresholds", install a cleanup trap so a failure cannot leak nodes and ports, and report an unusable baseline as inconclusive. - workload_orchestrator: bound subprocess communicate() and fail the exit gate on per-phase errors. Also pins the workload compose images to the versions the sibling stack already uses, hash-pins the Python dependencies, restricts the validator config template to loopback, corrects the dashboard and metric counts in the reference docs, drops a span from the regression gate that cannot fire under a WebSocket-only workload, and narrows the teardown pkill pattern so it no longer matches processes that merely mention the work directory. Verified with a full harness run against a local five-node cluster: 158 of 158 checks passed with no regressions detected.
565 lines
20 KiB
Bash
Executable File
565 lines
20 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# run-full-validation.sh — Orchestrates the full telemetry validation pipeline.
|
|
#
|
|
# Sequence:
|
|
# 1. Start the observability stack (OTel Collector, Tempo, Prometheus, Loki, Grafana)
|
|
# 2. Start a multi-node rippled cluster with full telemetry enabled
|
|
# 3. Wait for consensus
|
|
# 4. Run workload orchestrator (RPC load, TX submission, propagation wait)
|
|
# 5. Run the telemetry validation suite
|
|
# 6. Capture OTel timings and compare against committed baseline
|
|
# 7. (Optional) Run the performance overhead benchmark
|
|
#
|
|
# Usage:
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld --with-benchmark
|
|
# ./run-full-validation.sh --xrpld /path/to/xrpld --skip-regression
|
|
# ./run-full-validation.sh --cleanup
|
|
#
|
|
# Exit codes:
|
|
# 0 — All validation checks and the regression gate passed
|
|
# 1 — Validation checks failed OR the regression gate detected a regression
|
|
# OR the benchmark exceeded its overhead thresholds
|
|
# 2 — Infrastructure error (cluster/stack failed to start, workload
|
|
# orchestration failed, timing capture failed, overhead could not be
|
|
# measured)
|
|
#
|
|
# Every step below records its status and folds it into FINAL_EXIT; the first
|
|
# non-zero status in pipeline order is the one returned, so the earliest
|
|
# failure — the one that explains the later ones — is what the caller sees.
|
|
|
|
set -euo pipefail
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Colored output helpers
|
|
# ---------------------------------------------------------------------------
|
|
log() { printf "\033[1;34m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
ok() { printf "\033[1;32m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
warn() { printf "\033[1;33m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
fail() { printf "\033[1;31m[VALIDATE]\033[0m %s\n" "$*"; }
|
|
die() {
|
|
printf "\033[1;31m[VALIDATE]\033[0m %s\n" "$*" >&2
|
|
exit 2
|
|
}
|
|
|
|
# Overall run status, folded step by step (see the exit-code table above).
|
|
FINAL_EXIT=0
|
|
|
|
# fold_exit STATUS — record a step's status in FINAL_EXIT.
|
|
# First non-zero wins, so FINAL_EXIT names the earliest failing step.
|
|
fold_exit() {
|
|
if [ "$1" -ne 0 ] && [ "$FINAL_EXIT" -eq 0 ]; then
|
|
FINAL_EXIT="$1"
|
|
fi
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Configuration
|
|
# ---------------------------------------------------------------------------
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
TELEMETRY_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
|
|
REPO_ROOT="$(cd "$TELEMETRY_DIR/../.." && pwd)"
|
|
COMPOSE_FILE="$TELEMETRY_DIR/docker-compose.workload.yaml"
|
|
WORKDIR="/tmp/xrpld-validation"
|
|
|
|
XRPLD="${XRPLD:-$REPO_ROOT/.build/xrpld}"
|
|
NUM_NODES=5
|
|
RPC_PORT_BASE=5005
|
|
WS_PORT_BASE=6006
|
|
PEER_PORT_BASE=51235
|
|
# Inert: parsed from --rpc-rate/--rpc-duration/--tx-tps/--tx-duration and never
|
|
# read again. Load shape comes from the workload profile instead. Kept because
|
|
# the CI workflow still passes the four flags.
|
|
RPC_RATE=50
|
|
RPC_DURATION=120
|
|
TX_TPS=5
|
|
TX_DURATION=120
|
|
WITH_BENCHMARK=false
|
|
SKIP_LOKI=false
|
|
SKIP_REGRESSION=false
|
|
WORKLOAD_PROFILE="full-validation"
|
|
REPORT_DIR="$WORKDIR/reports"
|
|
# Rate window handed to Prometheus `rate()` when capturing timings. Keep
|
|
# this close to the active workload duration so histogram buckets cover
|
|
# the measurement window; longer windows dilute short-lived regressions.
|
|
REGRESSION_WINDOW="${REGRESSION_WINDOW:-3m}"
|
|
BASELINE_FILE="${BASELINE_FILE:-$SCRIPT_DIR/baselines/baseline-timings.json}"
|
|
THRESHOLDS_FILE="${THRESHOLDS_FILE:-$SCRIPT_DIR/regression-thresholds.json}"
|
|
METRICS_FILE="${METRICS_FILE:-$SCRIPT_DIR/regression-metrics.json}"
|
|
|
|
GENESIS_ACCOUNT="rHb9CJAWyB4rj91VRWn96DkukG4bwdtyTh"
|
|
GENESIS_SEED="snoPBrXtMeMyMHUVTgbuqAfg1SUTb"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Argument parsing
|
|
# ---------------------------------------------------------------------------
|
|
usage() {
|
|
echo "Usage: $0 [OPTIONS]"
|
|
echo ""
|
|
echo "Options:"
|
|
echo " --xrpld PATH Path to xrpld binary"
|
|
echo " --nodes NUM Number of validator nodes (default: 5)"
|
|
echo " --profile NAME Workload profile (default: full-validation)"
|
|
echo " --with-benchmark Also run performance overhead benchmark (telemetry off vs on)"
|
|
echo " --skip-loki Skip Loki log-trace correlation checks"
|
|
echo " --skip-regression Skip the OTel-baseline regression gate"
|
|
echo " --cleanup Tear down everything and exit"
|
|
echo " -h, --help Show this help"
|
|
echo ""
|
|
echo "Accepted but INERT (parsed for compatibility, then ignored):"
|
|
echo " --rpc-rate RPS no effect"
|
|
echo " --rpc-duration SECS no effect"
|
|
echo " --tx-tps TPS no effect"
|
|
echo " --tx-duration SECS no effect"
|
|
echo ""
|
|
echo " Load shape comes from the workload profile (--profile), which sets"
|
|
echo " the rate and duration of every phase in workload-profiles.json."
|
|
echo " These four flags stay accepted because the CI workflow passes them."
|
|
exit 0
|
|
}
|
|
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
--xrpld)
|
|
XRPLD="$2"
|
|
shift 2
|
|
;;
|
|
--nodes)
|
|
NUM_NODES="$2"
|
|
shift 2
|
|
;;
|
|
# The next four are inert — see the RPC_RATE default above.
|
|
--rpc-rate)
|
|
RPC_RATE="$2"
|
|
shift 2
|
|
;;
|
|
--rpc-duration)
|
|
RPC_DURATION="$2"
|
|
shift 2
|
|
;;
|
|
--tx-tps)
|
|
TX_TPS="$2"
|
|
shift 2
|
|
;;
|
|
--tx-duration)
|
|
TX_DURATION="$2"
|
|
shift 2
|
|
;;
|
|
--profile)
|
|
WORKLOAD_PROFILE="$2"
|
|
shift 2
|
|
;;
|
|
--with-benchmark)
|
|
WITH_BENCHMARK=true
|
|
shift
|
|
;;
|
|
--skip-loki)
|
|
SKIP_LOKI=true
|
|
shift
|
|
;;
|
|
--skip-regression)
|
|
SKIP_REGRESSION=true
|
|
shift
|
|
;;
|
|
--cleanup) # Cleanup mode
|
|
log "Cleaning up..."
|
|
# Match the node config path, not the bare workdir: a plain
|
|
# "$WORKDIR" pattern also matches any shell, editor or log tail
|
|
# whose command line merely mentions that path.
|
|
pkill -f "$WORKDIR/node[0-9]+/xrpld\.cfg" 2>/dev/null || true
|
|
docker compose -f "$COMPOSE_FILE" down 2>/dev/null || true
|
|
rm -rf "$WORKDIR"
|
|
ok "Cleanup complete."
|
|
exit 0
|
|
;;
|
|
-h | --help) usage ;;
|
|
*) die "Unknown option: $1" ;;
|
|
esac
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Prerequisites
|
|
# ---------------------------------------------------------------------------
|
|
log "Checking prerequisites..."
|
|
[ -x "$XRPLD" ] || die "xrpld binary not found: $XRPLD"
|
|
command -v docker >/dev/null 2>&1 || die "docker not found"
|
|
docker compose version >/dev/null 2>&1 || die "docker compose (v2) not found"
|
|
command -v python3 >/dev/null 2>&1 || die "python3 not found"
|
|
command -v curl >/dev/null 2>&1 || die "curl not found"
|
|
command -v jq >/dev/null 2>&1 || die "jq not found"
|
|
[ -f "$COMPOSE_FILE" ] || die "docker-compose.workload.yaml not found"
|
|
|
|
# Install Python dependencies.
|
|
log "Installing Python dependencies..."
|
|
pip3 install -q -r "$SCRIPT_DIR/requirements.txt" 2>/dev/null ||
|
|
pip install -q -r "$SCRIPT_DIR/requirements.txt" 2>/dev/null ||
|
|
warn "Could not install Python dependencies — they may already be present"
|
|
|
|
ok "Prerequisites verified."
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Cleanup previous run
|
|
# ---------------------------------------------------------------------------
|
|
log "Cleaning up previous run..."
|
|
# Narrowed for the same reason as the --cleanup branch above.
|
|
pkill -f "$WORKDIR/node[0-9]+/xrpld\.cfg" 2>/dev/null || true
|
|
sleep 2
|
|
rm -rf "$WORKDIR"
|
|
mkdir -p "$WORKDIR" "$REPORT_DIR"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 1: Start observability stack
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 1: Starting observability stack..."
|
|
# Point the collector's log mount at this run's workdir so the filelog
|
|
# receiver tails the per-node debug.log files generated below.
|
|
XRPLD_LOG_DIR="$WORKDIR" docker compose -f "$COMPOSE_FILE" up -d
|
|
|
|
log "Waiting for OTel Collector..."
|
|
for attempt in $(seq 1 30); do
|
|
status=$(curl -so /dev/null -w '%{http_code}' http://localhost:4318/ 2>/dev/null || echo 000)
|
|
if [ "$status" != "000" ]; then
|
|
ok "OTel Collector ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "OTel Collector not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
log "Waiting for Tempo..."
|
|
for attempt in $(seq 1 30); do
|
|
if curl -sf "http://localhost:3200/ready" >/dev/null 2>&1; then
|
|
ok "Tempo ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "Tempo not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
log "Waiting for Prometheus..."
|
|
for attempt in $(seq 1 30); do
|
|
if curl -sf "http://localhost:9090/-/healthy" >/dev/null 2>&1; then
|
|
ok "Prometheus ready (attempt $attempt)"
|
|
break
|
|
fi
|
|
[ "$attempt" -eq 30 ] && die "Prometheus not ready after 30s"
|
|
sleep 1
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 2: Generate validator keys and start cluster
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 2: Starting $NUM_NODES-node validator cluster..."
|
|
|
|
bash "$SCRIPT_DIR/generate-validator-keys.sh" "$XRPLD" "$NUM_NODES" "$WORKDIR"
|
|
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
NODE_DIR="$WORKDIR/node$i"
|
|
mkdir -p "$NODE_DIR/nudb" "$NODE_DIR/db"
|
|
|
|
RPC_PORT=$((RPC_PORT_BASE + i - 1))
|
|
WS_PORT=$((WS_PORT_BASE + i - 1))
|
|
PEER_PORT=$((PEER_PORT_BASE + i - 1))
|
|
SEED=$(jq -r ".[$((i - 1))].seed" "$WORKDIR/validator-keys.json")
|
|
|
|
# Build ips_fixed.
|
|
IPS_FIXED=""
|
|
for j in $(seq 1 "$NUM_NODES"); do
|
|
if [ "$j" -ne "$i" ]; then
|
|
IPS_FIXED="${IPS_FIXED}127.0.0.1 $((PEER_PORT_BASE + j - 1))
|
|
"
|
|
fi
|
|
done
|
|
|
|
cat >"$NODE_DIR/xrpld.cfg" <<EOCFG
|
|
[server]
|
|
port_rpc
|
|
port_ws
|
|
port_peer
|
|
|
|
[port_rpc]
|
|
port = $RPC_PORT
|
|
ip = 127.0.0.1
|
|
admin = 127.0.0.1
|
|
protocol = http
|
|
|
|
[port_ws]
|
|
port = $WS_PORT
|
|
ip = 127.0.0.1
|
|
admin = 127.0.0.1
|
|
protocol = ws
|
|
|
|
[port_peer]
|
|
port = $PEER_PORT
|
|
ip = 0.0.0.0
|
|
protocol = peer
|
|
|
|
[node_db]
|
|
type=NuDB
|
|
path=$NODE_DIR/nudb
|
|
online_delete=256
|
|
|
|
[database_path]
|
|
$NODE_DIR/db
|
|
|
|
[debug_logfile]
|
|
$NODE_DIR/debug.log
|
|
|
|
[validation_seed]
|
|
$SEED
|
|
|
|
[validators_file]
|
|
$WORKDIR/validators.txt
|
|
|
|
[ips]
|
|
${IPS_FIXED}
|
|
|
|
[telemetry]
|
|
enabled=1
|
|
service_instance_id=validator-${i}
|
|
endpoint=http://localhost:4318/v1/traces
|
|
batch_size=512
|
|
batch_delay_ms=2000
|
|
max_queue_size=2048
|
|
trace_rpc=1
|
|
trace_transactions=1
|
|
trace_consensus=1
|
|
trace_peer=1
|
|
trace_ledger=1
|
|
|
|
[insight]
|
|
# Native OTel metrics via OTLP/HTTP. The collector has no StatsD receiver
|
|
# (metrics pipeline is [otlp, spanmetrics]), so beast::insight must export
|
|
# over OTLP for system metrics to reach Prometheus. prefix=xrpld matches the
|
|
# OTel resource service name and the metric names the dashboards query.
|
|
server=otel
|
|
endpoint=http://localhost:4318/v1/metrics
|
|
prefix=xrpld
|
|
|
|
[rpc_startup]
|
|
{ "command": "log_level", "severity": "warning" }
|
|
|
|
[signing_support]
|
|
true
|
|
|
|
[ssl_verify]
|
|
0
|
|
EOCFG
|
|
|
|
"$XRPLD" --conf "$NODE_DIR/xrpld.cfg" --start >"$NODE_DIR/stdout.log" 2>&1 &
|
|
echo $! >"$NODE_DIR/xrpld.pid"
|
|
log " Node $i: RPC=$RPC_PORT WS=$WS_PORT Peer=$PEER_PORT PID=$!"
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 3: Wait for consensus
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 3: Waiting for consensus..."
|
|
for attempt in $(seq 1 120); do
|
|
ready=0
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
port=$((RPC_PORT_BASE + i - 1))
|
|
state=$(curl -sf "http://localhost:$port" \
|
|
-d '{"method":"server_info"}' 2>/dev/null |
|
|
jq -r '.result.info.server_state' 2>/dev/null || echo "")
|
|
if [ "$state" = "proposing" ]; then
|
|
ready=$((ready + 1))
|
|
fi
|
|
done
|
|
if [ "$ready" -ge "$NUM_NODES" ]; then
|
|
ok "All $NUM_NODES nodes proposing (attempt $attempt)"
|
|
break
|
|
fi
|
|
if [ "$attempt" -eq 120 ]; then
|
|
# Fatal, not a warning. A partial cluster still answers queries, so the
|
|
# run would complete and report unrelated span/metric failures: series
|
|
# counts scale with the number of live nodes, and spans that need a
|
|
# quorum are simply never emitted. One infrastructure error here is
|
|
# worth more than a pile of misleading assertion failures later.
|
|
echo ""
|
|
die "Consensus timeout — only $ready/$NUM_NODES nodes proposing after ${attempt}s. Check $WORKDIR/node*/debug.log, then '$0 --cleanup'."
|
|
fi
|
|
printf "\r %d/%d nodes proposing..." "$ready" "$NUM_NODES"
|
|
sleep 1
|
|
done
|
|
echo ""
|
|
|
|
# Wait for first validated ledger.
|
|
log "Waiting for validated ledger..."
|
|
for attempt in $(seq 1 60); do
|
|
val_seq=$(curl -sf "http://localhost:$RPC_PORT_BASE" \
|
|
-d '{"method":"server_info"}' 2>/dev/null |
|
|
jq -r '.result.info.validated_ledger.seq // 0' 2>/dev/null || echo 0)
|
|
if [ "$val_seq" -gt 2 ] 2>/dev/null; then
|
|
ok "Validated ledger: seq $val_seq"
|
|
break
|
|
fi
|
|
# Fatal for the same reason as the consensus timeout above, and because
|
|
# several assertions are gated on a validated ledger existing at all:
|
|
# ledger_economy{metric="base_fee_xrp"} is only observed from a validated
|
|
# ledger, and complete_ledgers stays absent while the range is empty.
|
|
if [ "$attempt" -eq 60 ]; then
|
|
die "No validated ledger after ${attempt}s (last seq: $val_seq). Check $WORKDIR/node*/debug.log, then '$0 --cleanup'."
|
|
fi
|
|
sleep 1
|
|
done
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 4: Run workload orchestrator
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 4: Running workload orchestrator (profile: $WORKLOAD_PROFILE)..."
|
|
|
|
WS_ENDPOINTS=""
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
WS_ENDPOINTS="$WS_ENDPOINTS ws://localhost:$((WS_PORT_BASE + i - 1))"
|
|
done
|
|
|
|
ORCHESTRATOR_EXIT=0
|
|
python3 "$SCRIPT_DIR/workload_orchestrator.py" \
|
|
--profile "$WORKLOAD_PROFILE" \
|
|
--endpoints $WS_ENDPOINTS \
|
|
--report "$REPORT_DIR/workload-report.json" \
|
|
--report-dir "$REPORT_DIR" || ORCHESTRATOR_EXIT=$?
|
|
|
|
if [ "$ORCHESTRATOR_EXIT" -eq 0 ]; then
|
|
ok "Workload orchestration complete."
|
|
else
|
|
# Treated as an infrastructure error: the span and metric assertions below
|
|
# would be graded against traffic that was never generated.
|
|
fail "Workload orchestrator failed (exit $ORCHESTRATOR_EXIT) — the checks below run against incomplete traffic"
|
|
fold_exit 2
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 5: Run telemetry validation suite
|
|
# ---------------------------------------------------------------------------
|
|
log "Step 5: Running telemetry validation suite..."
|
|
|
|
VALIDATION_ARGS="--report $REPORT_DIR/validation-report.json"
|
|
if [ "$SKIP_LOKI" = true ]; then
|
|
VALIDATION_ARGS="$VALIDATION_ARGS --skip-loki"
|
|
fi
|
|
|
|
VALIDATION_EXIT=0
|
|
python3 "$SCRIPT_DIR/validate_telemetry.py" $VALIDATION_ARGS || VALIDATION_EXIT=$?
|
|
|
|
if [ "$VALIDATION_EXIT" -eq 0 ]; then
|
|
ok "All telemetry validation checks passed!"
|
|
else
|
|
fail "Some telemetry validation checks failed (exit $VALIDATION_EXIT)"
|
|
fi
|
|
fold_exit "$VALIDATION_EXIT"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 6: Capture OTel timings and run the regression comparison
|
|
# ---------------------------------------------------------------------------
|
|
# This step ALWAYS captures timings (so CI always has an artifact from which
|
|
# to bootstrap/refresh the committed baseline). The comparator then either:
|
|
# - prints the paste-me JSON when the baseline is a placeholder, or
|
|
# - enforces thresholds and fails the run on regression.
|
|
# Use --skip-regression to opt out (e.g. for ad-hoc local exploration).
|
|
TIMINGS_FILE="$REPORT_DIR/timings.json"
|
|
REGRESSION_REPORT="$REPORT_DIR/regression-report.json"
|
|
REGRESSION_EXIT=0
|
|
|
|
if [ "$SKIP_REGRESSION" != true ]; then
|
|
log "Step 6: Capturing OTel timings from Prometheus..."
|
|
if python3 "$SCRIPT_DIR/capture_timings.py" \
|
|
--prometheus "http://localhost:9090" \
|
|
--metrics "$METRICS_FILE" \
|
|
--output "$TIMINGS_FILE" \
|
|
--window "$REGRESSION_WINDOW" \
|
|
--profile "$WORKLOAD_PROFILE"; then
|
|
ok "Timings captured: $TIMINGS_FILE"
|
|
else
|
|
fail "Failed to capture timings — skipping regression comparison."
|
|
REGRESSION_EXIT=2
|
|
SKIP_REGRESSION=true
|
|
fi
|
|
fi
|
|
|
|
if [ "$SKIP_REGRESSION" != true ]; then
|
|
log "Comparing against baseline $BASELINE_FILE..."
|
|
python3 "$SCRIPT_DIR/compare_to_baseline.py" \
|
|
--timings "$TIMINGS_FILE" \
|
|
--baseline "$BASELINE_FILE" \
|
|
--thresholds "$THRESHOLDS_FILE" \
|
|
--report "$REGRESSION_REPORT" || REGRESSION_EXIT=$?
|
|
if [ "$REGRESSION_EXIT" -eq 0 ]; then
|
|
ok "Regression gate passed (or baseline placeholder — paste JSON printed above)."
|
|
elif [ "$REGRESSION_EXIT" -eq 1 ]; then
|
|
fail "Regression detected — see $REGRESSION_REPORT"
|
|
else
|
|
fail "Regression comparator internal error (exit $REGRESSION_EXIT)"
|
|
fi
|
|
else
|
|
warn "Regression gate skipped."
|
|
fi
|
|
fold_exit "$REGRESSION_EXIT"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Step 7: (Optional) Run overhead benchmark
|
|
# ---------------------------------------------------------------------------
|
|
BENCHMARK_EXIT=0
|
|
if [ "$WITH_BENCHMARK" = true ]; then
|
|
log "Step 7: Running performance benchmark..."
|
|
bash "$SCRIPT_DIR/benchmark.sh" \
|
|
--xrpld "$XRPLD" \
|
|
--duration 120 \
|
|
--nodes 3 \
|
|
--output "$REPORT_DIR" || BENCHMARK_EXIT=$?
|
|
|
|
if [ "$BENCHMARK_EXIT" -eq 0 ]; then
|
|
ok "Benchmark within overhead thresholds."
|
|
elif [ "$BENCHMARK_EXIT" -eq 1 ]; then
|
|
# A measured threshold breach — same class as a failed check.
|
|
fail "Benchmark exceeded overhead thresholds (exit 1)"
|
|
fold_exit 1
|
|
else
|
|
# benchmark.sh could not produce a usable measurement (e.g. incomplete
|
|
# system metrics). Reported as an infrastructure error, not a perf
|
|
# regression: nothing was measured, so nothing was breached.
|
|
fail "Benchmark could not measure overhead (exit $BENCHMARK_EXIT) — treated as an infrastructure error"
|
|
fold_exit 2
|
|
fi
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Summary
|
|
# ---------------------------------------------------------------------------
|
|
echo ""
|
|
echo "==========================================================="
|
|
echo " FULL VALIDATION RESULTS"
|
|
echo "==========================================================="
|
|
echo ""
|
|
echo " Reports directory: $REPORT_DIR"
|
|
echo ""
|
|
ls -la "$REPORT_DIR/" 2>/dev/null || true
|
|
echo ""
|
|
echo " Observability stack is running:"
|
|
echo " Tempo: http://localhost:3200"
|
|
echo " Grafana: http://localhost:3000"
|
|
echo " Prometheus: http://localhost:9090"
|
|
echo ""
|
|
echo " xrpld nodes ($NUM_NODES) are running:"
|
|
for i in $(seq 1 "$NUM_NODES"); do
|
|
rpc=$((RPC_PORT_BASE + i - 1))
|
|
ws=$((WS_PORT_BASE + i - 1))
|
|
pid=$(cat "$WORKDIR/node$i/xrpld.pid" 2>/dev/null || echo 'unknown')
|
|
echo " Node $i: RPC=$rpc WS=$ws PID=$pid"
|
|
done
|
|
echo ""
|
|
echo " To tear down:"
|
|
echo " $0 --cleanup"
|
|
echo ""
|
|
echo " Step statuses (0 = ok):"
|
|
echo " Workload orchestration: $ORCHESTRATOR_EXIT"
|
|
echo " Telemetry validation: $VALIDATION_EXIT"
|
|
echo " Regression gate: $REGRESSION_EXIT"
|
|
echo " Overhead benchmark: $BENCHMARK_EXIT"
|
|
echo ""
|
|
echo "==========================================================="
|
|
|
|
# FINAL_EXIT already holds the first non-zero step status (see fold_exit).
|
|
exit "$FINAL_EXIT"
|