test(testnet): exercise manifest lifecycle live across mixed binaries

Three scenarios on the new scripts primitives: mid-run rotation to
sequence 2 through an old relay (supersede, admission, repair re-arm);
a config-injected revocation seeded by the old node with the remaining
four of five validators holding exact quorum; and wallet-wipe recovery
through an old upstream with resumed paired forwarding. Waits are
event-driven on terminal log facts with heartbeat logging.
This commit is contained in:
Nicholas Dudfield
2026-08-29 19:09:37 +07:00
parent c20e8ac2ba
commit 39629c4cf5
4 changed files with 335 additions and 0 deletions

View File

@@ -0,0 +1,96 @@
"""Revoke one of five validators live and watch the terminal update propagate.
Five validators are used deliberately: revoking one leaves four of five —
exactly the 80% quorum — so the network keeps closing ledgers and the
scenario can assert liveness with the revoked member excluded, not merely a
halt. (On a three-node UNL the same revocation stops the network: small-net
quorums round up to everyone.)
The revocation is installed by config on the OLD release node deliberately:
under the new transport slice a config-loaded revocation on an upgraded node
sits in its durable cache and is never gossiped (there is no connect-time
dump and no ambient manifest relay), while the old binary still dumps its
manifest cache on every fresh connection and relays what it accepts. The old
node is therefore the only in-topology seeder for a config-injected
revocation — that asymmetry is itself part of the documented durability
boundary.
"""
from xahaud_scripts.testnet.scenario import (
AssertionError as ScenarioAssertion,
)
async def scenario(ctx, log):
nodes = [0, 1, 2, 3, 4, 5]
# Validators 0-4 mesh through 0; the old release node 5 hangs off the
# mesh and is the revocation's config seeder.
expected = ctx.topology_edges(
[(0, 1), (0, 2), (0, 3), (0, 4), (1, 2), (3, 4), (4, 5)]
)
await ctx.apply_topology(expected, nodes=nodes, exact=False)
# Normal operation first: everyone validates and validator 4's manifest
# is durably known, so the revocation supersedes real retained state.
await ctx.wait_for_ledgers(2, node_id=0, timeout=180)
revoked = ctx.mark("revoked")
result = await ctx.revoke_validator(4, 5)
log(
"revoked validator n4 master via the old relay n5: "
f"{result['public_key']}"
)
# The restarted old node loads the revocation and seeds it through its
# legacy connect-time dump and accept-relay; its peer n4 spreads it into
# the mesh through the normal revocation relay lane. The old node's
# reconnect interval dominates the latency, so wait on the log fact.
deadline = 36
for attempt in range(deadline):
try:
ctx.assert_log("Revoked", since=revoked, nodes=[4])
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-revocation-arrival: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-revocation-arrival")
# Terminal manifest applied and relayed onward by upgraded nodes: first
# at the old seeder's direct peer, then across the mesh.
ctx.assert_log(
"manifest_revocation accepted_for_relay",
since=revoked,
nodes=[4],
)
for attempt in range(deadline):
try:
ctx.assert_log("Revoked", since=revoked, nodes=[0])
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-mesh-revocation: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-mesh-revocation")
ctx.assert_log(
"manifest_revocation accepted_for_relay",
since=revoked,
nodes=[0],
)
# Liveness with the revoked member excluded: four of five is exactly
# quorum, so ledgers keep closing.
await ctx.wait_for_ledgers(2, node_id=0, timeout=180)
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
ctx.assert_not_log("Validation: Too small", nodes=nodes)
log(
"PASS: a config-injected revocation seeded by the old relay reached"
" the mesh, applied terminally on upgraded nodes, relayed onward,"
" and the remaining four validators kept the network live at exact"
" quorum"
)

View File

@@ -0,0 +1,110 @@
"""Rotate a validator's manifest mid-run and watch the bump propagate.
"Heals on a sequence bump" is the load-bearing healing claim of the
manifest-before-validation transport slice. This exercises it live for the
first time: the validator mints sequence 2 and restarts; its validations then
carry the new manifest; an old release relay forwards the existing envelopes;
the upgraded observer supersedes its retained knowledge and admits sequence 2;
and the observer's repair lane re-arms at the newer sequence when the old
relay's later validations arrive naked.
"""
from xahaud_scripts.testnet.scenario import (
AssertionError as ScenarioAssertion,
)
async def scenario(ctx, log):
nodes = [0, 1, 2]
expected = ctx.topology_edges([(0, 1), (1, 2)])
await ctx.apply_topology(expected, nodes=nodes, exact=False)
# Baseline: sequence 1 propagates through the old relay and is admitted
# by the upgraded observer before we rotate anything. The validator can
# race several ledgers ahead of propagation under fast bootstrap, so
# wait on the log fact itself.
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
deadline = 24
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-baseline-admission: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-baseline-admission")
rotated = ctx.mark("rotated")
rotation = await ctx.rotate_validator_manifest(0)
assert rotation["sequence"] == 2, rotation
log(f"rotated validator n0 to manifest sequence {rotation['sequence']}")
# The restarted validator re-joins and validates under the new signing
# key; always-send carries the sequence-2 prerequisite with each
# validation, and the old relay's one-shot manifest forward plus later
# naked relays exercise both the supersede and the repair re-arm. Wait
# on the terminal log fact rather than ledger counts: in this topology
# only the validator advances its ledger, and its own restart closed the
# node-0 WebSocket ledger feed. The repair re-arm is the last event in
# the causal chain, so everything else must precede it.
deadline = 36 # polls at 5s => 180s budget at ~16s consensus rounds
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation repair_sent .*sequence=2",
since=rotated,
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-repair-rearm: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-repair-rearm")
ctx.assert_log(
"manifest_validation pair_enqueued .*sequence=2",
since=rotated,
nodes=[0],
)
ctx.assert_log(
"manifest_validation candidate_staged .*sequence=2",
since=rotated,
nodes=[2],
)
ctx.assert_log(
"manifest_validation candidate_matched",
since=rotated,
nodes=[2],
)
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=2"
" .*disposition=accepted",
since=rotated,
nodes=[2],
)
ctx.assert_log_order(
[
"manifest_validation single_manifest_processed .*sequence=2",
"manifest_validation repair_sent .*sequence=2",
],
since=rotated,
nodes=[2],
)
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
ctx.assert_not_log("Validation: Too small", nodes=nodes)
log(
"PASS: mid-run rotation to sequence 2 propagated through an old"
" relay; the upgraded observer superseded and admitted the new"
" manifest and re-armed its repair lane at the new sequence"
)

View File

@@ -20,3 +20,35 @@ tests:
log_levels:
Protocol: debug
Validations: debug
- name: manifest_rotation_propagation
script: .testnet/scenarios/manifest_rotation_propagation.py
network:
validators: 1
node_binaries:
1: "@release-3350"
log_levels:
Protocol: debug
Validations: debug
- name: manifest_revocation_live
script: .testnet/scenarios/manifest_revocation_live.py
network:
node_count: 6
validators: 5
node_binaries:
5: "@release-3350"
log_levels:
Protocol: debug
Validations: debug
- name: manifest_wipe_recovery
script: .testnet/scenarios/manifest_wipe_recovery.py
network:
node_count: 4
validators: 1
node_binaries:
1: "@release-3350"
log_levels:
Protocol: debug
Validations: debug

View File

@@ -0,0 +1,97 @@
"""Wipe a mid-chain node's wallet and watch both generations heal it.
The forever cache's informal replicated history is gone from new nodes, so
this pins what replaces it. A wiped upgraded node reconnects with empty
durable state; its old-release upstream re-seeds it through the legacy
connect-time dump lane, its own validation traffic re-pairs through
always-send, and it resumes forwarding paired prerequisites downstream. The
recovery, not a starvation, is the honest live boundary: every reachable
topology heals, because old peers dump at connect and new peers re-send the
prerequisite with every validation.
"""
from xahaud_scripts.testnet.scenario import (
AssertionError as ScenarioAssertion,
)
async def scenario(ctx, log):
nodes = [0, 1, 2, 3]
expected = ctx.topology_edges([(0, 1), (1, 2), (2, 3)])
await ctx.apply_topology(expected, nodes=nodes, exact=False)
# Baseline: knowledge reaches the end of the chain. The tail observer
# receives paired traffic from the upgraded mid-chain node. Admission at
# the tail implies the whole upstream chain, so wait on that log fact —
# four hops can trail the validator's fast-bootstrap ledger count.
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
deadline = 24
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[3],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-baseline-chain: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-baseline-chain")
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[2],
)
wiped = ctx.mark("wiped")
await ctx.restart_node(2, wipe_wallet_db=True)
log("restarted n2 with a wiped wallet database")
await ctx.wait_for_ledgers(3, node_id=0, timeout=180)
# Wait on the terminal log fact: resumed paired forwarding downstream is
# the last event in the recovery chain, so re-learning and re-admission
# must precede it.
deadline = 36
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation send_prerequisite .*sequence=1",
since=wiped,
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-resumed-forwarding: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-resumed-forwarding")
# Recovery: the wiped node re-learned the validator identity from live
# traffic (the old upstream's connect-time dump arrives as a singleton
# candidate; the next validation proves it) and admitted it durably
# again.
ctx.assert_log(
"manifest_validation candidate_staged .*sequence=1",
since=wiped,
nodes=[2],
)
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1"
" .*disposition=accepted",
since=wiped,
nodes=[2],
)
ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes)
ctx.assert_not_log("Validation: Too small", nodes=nodes)
log(
"PASS: a wallet-wiped mid-chain node re-learned the validator"
" identity from live traffic through an old upstream, re-admitted it"
" durably, and resumed paired forwarding to the tail observer"
)