mirror of
https://github.com/Xahau/xahaud.git
synced 2026-09-28 16:08:05 +00:00
test(testnet): exercise manifest lifecycle live across mixed binaries
Three scenarios on the new scripts primitives: mid-run rotation to sequence 2 through an old relay (supersede, admission, repair re-arm); a config-injected revocation seeded by the old node with the remaining four of five validators holding exact quorum; and wallet-wipe recovery through an old upstream with resumed paired forwarding. Waits are event-driven on terminal log facts with heartbeat logging.
This commit is contained in:
96
.testnet/scenarios/manifest_revocation_live.py
Normal file
96
.testnet/scenarios/manifest_revocation_live.py
Normal file
@@ -0,0 +1,96 @@
|
||||
"""Revoke one of five validators live and watch the terminal update propagate.
|
||||
|
||||
Five validators are used deliberately: revoking one leaves four of five —
|
||||
exactly the 80% quorum — so the network keeps closing ledgers and the
|
||||
scenario can assert liveness with the revoked member excluded, not merely a
|
||||
halt. (On a three-node UNL the same revocation stops the network: small-net
|
||||
quorums round up to everyone.)
|
||||
|
||||
The revocation is installed by config on the OLD release node deliberately:
|
||||
under the new transport slice a config-loaded revocation on an upgraded node
|
||||
sits in its durable cache and is never gossiped (there is no connect-time
|
||||
dump and no ambient manifest relay), while the old binary still dumps its
|
||||
manifest cache on every fresh connection and relays what it accepts. The old
|
||||
node is therefore the only in-topology seeder for a config-injected
|
||||
revocation — that asymmetry is itself part of the documented durability
|
||||
boundary.
|
||||
"""
|
||||
|
||||
from xahaud_scripts.testnet.scenario import (
|
||||
AssertionError as ScenarioAssertion,
|
||||
)
|
||||
|
||||
|
||||
async def scenario(ctx, log):
|
||||
nodes = [0, 1, 2, 3, 4, 5]
|
||||
# Validators 0-4 mesh through 0; the old release node 5 hangs off the
|
||||
# mesh and is the revocation's config seeder.
|
||||
expected = ctx.topology_edges(
|
||||
[(0, 1), (0, 2), (0, 3), (0, 4), (1, 2), (3, 4), (4, 5)]
|
||||
)
|
||||
|
||||
await ctx.apply_topology(expected, nodes=nodes, exact=False)
|
||||
|
||||
# Normal operation first: everyone validates and validator 4's manifest
|
||||
# is durably known, so the revocation supersedes real retained state.
|
||||
await ctx.wait_for_ledgers(2, node_id=0, timeout=180)
|
||||
|
||||
revoked = ctx.mark("revoked")
|
||||
result = await ctx.revoke_validator(4, 5)
|
||||
log(
|
||||
"revoked validator n4 master via the old relay n5: "
|
||||
f"{result['public_key']}"
|
||||
)
|
||||
|
||||
# The restarted old node loads the revocation and seeds it through its
|
||||
# legacy connect-time dump and accept-relay; its peer n4 spreads it into
|
||||
# the mesh through the normal revocation relay lane. The old node's
|
||||
# reconnect interval dominates the latency, so wait on the log fact.
|
||||
deadline = 36
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log("Revoked", since=revoked, nodes=[4])
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-revocation-arrival: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-revocation-arrival")
|
||||
|
||||
# Terminal manifest applied and relayed onward by upgraded nodes: first
|
||||
# at the old seeder's direct peer, then across the mesh.
|
||||
ctx.assert_log(
|
||||
"manifest_revocation accepted_for_relay",
|
||||
since=revoked,
|
||||
nodes=[4],
|
||||
)
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log("Revoked", since=revoked, nodes=[0])
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-mesh-revocation: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-mesh-revocation")
|
||||
ctx.assert_log(
|
||||
"manifest_revocation accepted_for_relay",
|
||||
since=revoked,
|
||||
nodes=[0],
|
||||
)
|
||||
|
||||
# Liveness with the revoked member excluded: four of five is exactly
|
||||
# quorum, so ledgers keep closing.
|
||||
await ctx.wait_for_ledgers(2, node_id=0, timeout=180)
|
||||
|
||||
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
|
||||
ctx.assert_not_log("Validation: Too small", nodes=nodes)
|
||||
|
||||
log(
|
||||
"PASS: a config-injected revocation seeded by the old relay reached"
|
||||
" the mesh, applied terminally on upgraded nodes, relayed onward,"
|
||||
" and the remaining four validators kept the network live at exact"
|
||||
" quorum"
|
||||
)
|
||||
110
.testnet/scenarios/manifest_rotation_propagation.py
Normal file
110
.testnet/scenarios/manifest_rotation_propagation.py
Normal file
@@ -0,0 +1,110 @@
|
||||
"""Rotate a validator's manifest mid-run and watch the bump propagate.
|
||||
|
||||
"Heals on a sequence bump" is the load-bearing healing claim of the
|
||||
manifest-before-validation transport slice. This exercises it live for the
|
||||
first time: the validator mints sequence 2 and restarts; its validations then
|
||||
carry the new manifest; an old release relay forwards the existing envelopes;
|
||||
the upgraded observer supersedes its retained knowledge and admits sequence 2;
|
||||
and the observer's repair lane re-arms at the newer sequence when the old
|
||||
relay's later validations arrive naked.
|
||||
"""
|
||||
|
||||
from xahaud_scripts.testnet.scenario import (
|
||||
AssertionError as ScenarioAssertion,
|
||||
)
|
||||
|
||||
|
||||
|
||||
async def scenario(ctx, log):
|
||||
nodes = [0, 1, 2]
|
||||
expected = ctx.topology_edges([(0, 1), (1, 2)])
|
||||
|
||||
await ctx.apply_topology(expected, nodes=nodes, exact=False)
|
||||
|
||||
# Baseline: sequence 1 propagates through the old relay and is admitted
|
||||
# by the upgraded observer before we rotate anything. The validator can
|
||||
# race several ledgers ahead of propagation under fast bootstrap, so
|
||||
# wait on the log fact itself.
|
||||
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
|
||||
deadline = 24
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log(
|
||||
"manifest_validation single_manifest_processed .*sequence=1",
|
||||
nodes=[2],
|
||||
)
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-baseline-admission: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-baseline-admission")
|
||||
|
||||
rotated = ctx.mark("rotated")
|
||||
rotation = await ctx.rotate_validator_manifest(0)
|
||||
assert rotation["sequence"] == 2, rotation
|
||||
log(f"rotated validator n0 to manifest sequence {rotation['sequence']}")
|
||||
|
||||
# The restarted validator re-joins and validates under the new signing
|
||||
# key; always-send carries the sequence-2 prerequisite with each
|
||||
# validation, and the old relay's one-shot manifest forward plus later
|
||||
# naked relays exercise both the supersede and the repair re-arm. Wait
|
||||
# on the terminal log fact rather than ledger counts: in this topology
|
||||
# only the validator advances its ledger, and its own restart closed the
|
||||
# node-0 WebSocket ledger feed. The repair re-arm is the last event in
|
||||
# the causal chain, so everything else must precede it.
|
||||
deadline = 36 # polls at 5s => 180s budget at ~16s consensus rounds
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log(
|
||||
"manifest_validation repair_sent .*sequence=2",
|
||||
since=rotated,
|
||||
nodes=[2],
|
||||
)
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-repair-rearm: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-repair-rearm")
|
||||
|
||||
ctx.assert_log(
|
||||
"manifest_validation pair_enqueued .*sequence=2",
|
||||
since=rotated,
|
||||
nodes=[0],
|
||||
)
|
||||
ctx.assert_log(
|
||||
"manifest_validation candidate_staged .*sequence=2",
|
||||
since=rotated,
|
||||
nodes=[2],
|
||||
)
|
||||
ctx.assert_log(
|
||||
"manifest_validation candidate_matched",
|
||||
since=rotated,
|
||||
nodes=[2],
|
||||
)
|
||||
ctx.assert_log(
|
||||
"manifest_validation single_manifest_processed .*sequence=2"
|
||||
" .*disposition=accepted",
|
||||
since=rotated,
|
||||
nodes=[2],
|
||||
)
|
||||
ctx.assert_log_order(
|
||||
[
|
||||
"manifest_validation single_manifest_processed .*sequence=2",
|
||||
"manifest_validation repair_sent .*sequence=2",
|
||||
],
|
||||
since=rotated,
|
||||
nodes=[2],
|
||||
)
|
||||
|
||||
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
|
||||
ctx.assert_not_log("Validation: Too small", nodes=nodes)
|
||||
|
||||
log(
|
||||
"PASS: mid-run rotation to sequence 2 propagated through an old"
|
||||
" relay; the upgraded observer superseded and admitted the new"
|
||||
" manifest and re-armed its repair lane at the new sequence"
|
||||
)
|
||||
@@ -20,3 +20,35 @@ tests:
|
||||
log_levels:
|
||||
Protocol: debug
|
||||
Validations: debug
|
||||
|
||||
- name: manifest_rotation_propagation
|
||||
script: .testnet/scenarios/manifest_rotation_propagation.py
|
||||
network:
|
||||
validators: 1
|
||||
node_binaries:
|
||||
1: "@release-3350"
|
||||
log_levels:
|
||||
Protocol: debug
|
||||
Validations: debug
|
||||
|
||||
- name: manifest_revocation_live
|
||||
script: .testnet/scenarios/manifest_revocation_live.py
|
||||
network:
|
||||
node_count: 6
|
||||
validators: 5
|
||||
node_binaries:
|
||||
5: "@release-3350"
|
||||
log_levels:
|
||||
Protocol: debug
|
||||
Validations: debug
|
||||
|
||||
- name: manifest_wipe_recovery
|
||||
script: .testnet/scenarios/manifest_wipe_recovery.py
|
||||
network:
|
||||
node_count: 4
|
||||
validators: 1
|
||||
node_binaries:
|
||||
1: "@release-3350"
|
||||
log_levels:
|
||||
Protocol: debug
|
||||
Validations: debug
|
||||
|
||||
97
.testnet/scenarios/manifest_wipe_recovery.py
Normal file
97
.testnet/scenarios/manifest_wipe_recovery.py
Normal file
@@ -0,0 +1,97 @@
|
||||
"""Wipe a mid-chain node's wallet and watch both generations heal it.
|
||||
|
||||
The forever cache's informal replicated history is gone from new nodes, so
|
||||
this pins what replaces it. A wiped upgraded node reconnects with empty
|
||||
durable state; its old-release upstream re-seeds it through the legacy
|
||||
connect-time dump lane, its own validation traffic re-pairs through
|
||||
always-send, and it resumes forwarding paired prerequisites downstream. The
|
||||
recovery, not a starvation, is the honest live boundary: every reachable
|
||||
topology heals, because old peers dump at connect and new peers re-send the
|
||||
prerequisite with every validation.
|
||||
"""
|
||||
|
||||
from xahaud_scripts.testnet.scenario import (
|
||||
AssertionError as ScenarioAssertion,
|
||||
)
|
||||
|
||||
|
||||
|
||||
async def scenario(ctx, log):
|
||||
nodes = [0, 1, 2, 3]
|
||||
expected = ctx.topology_edges([(0, 1), (1, 2), (2, 3)])
|
||||
|
||||
await ctx.apply_topology(expected, nodes=nodes, exact=False)
|
||||
|
||||
# Baseline: knowledge reaches the end of the chain. The tail observer
|
||||
# receives paired traffic from the upgraded mid-chain node. Admission at
|
||||
# the tail implies the whole upstream chain, so wait on that log fact —
|
||||
# four hops can trail the validator's fast-bootstrap ledger count.
|
||||
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
|
||||
deadline = 24
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log(
|
||||
"manifest_validation single_manifest_processed .*sequence=1",
|
||||
nodes=[3],
|
||||
)
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-baseline-chain: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-baseline-chain")
|
||||
ctx.assert_log(
|
||||
"manifest_validation single_manifest_processed .*sequence=1",
|
||||
nodes=[2],
|
||||
)
|
||||
|
||||
wiped = ctx.mark("wiped")
|
||||
await ctx.restart_node(2, wipe_wallet_db=True)
|
||||
log("restarted n2 with a wiped wallet database")
|
||||
|
||||
await ctx.wait_for_ledgers(3, node_id=0, timeout=180)
|
||||
|
||||
# Wait on the terminal log fact: resumed paired forwarding downstream is
|
||||
# the last event in the recovery chain, so re-learning and re-admission
|
||||
# must precede it.
|
||||
deadline = 36
|
||||
for attempt in range(deadline):
|
||||
try:
|
||||
ctx.assert_log(
|
||||
"manifest_validation send_prerequisite .*sequence=1",
|
||||
since=wiped,
|
||||
nodes=[2],
|
||||
)
|
||||
break
|
||||
except ScenarioAssertion:
|
||||
if attempt == deadline - 1:
|
||||
raise
|
||||
if attempt % 6 == 5:
|
||||
log(f"await-resumed-forwarding: attempt {attempt + 1}/{deadline}")
|
||||
await ctx.sleep(5, name="await-resumed-forwarding")
|
||||
|
||||
# Recovery: the wiped node re-learned the validator identity from live
|
||||
# traffic (the old upstream's connect-time dump arrives as a singleton
|
||||
# candidate; the next validation proves it) and admitted it durably
|
||||
# again.
|
||||
ctx.assert_log(
|
||||
"manifest_validation candidate_staged .*sequence=1",
|
||||
since=wiped,
|
||||
nodes=[2],
|
||||
)
|
||||
ctx.assert_log(
|
||||
"manifest_validation single_manifest_processed .*sequence=1"
|
||||
" .*disposition=accepted",
|
||||
since=wiped,
|
||||
nodes=[2],
|
||||
)
|
||||
|
||||
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
|
||||
ctx.assert_not_log("Validation: Too small", nodes=nodes)
|
||||
|
||||
log(
|
||||
"PASS: a wallet-wiped mid-chain node re-learned the validator"
|
||||
" identity from live traffic through an old upstream, re-admitted it"
|
||||
" durably, and resumed paired forwarding to the tail observer"
|
||||
)
|
||||
Reference in New Issue
Block a user