From 39629c4cf51a46c3d333d4396002f286eac2dd31 Mon Sep 17 00:00:00 2001 From: Nicholas Dudfield Date: Sat, 29 Aug 2026 19:09:37 +0700 Subject: [PATCH] test(testnet): exercise manifest lifecycle live across mixed binaries Three scenarios on the new scripts primitives: mid-run rotation to sequence 2 through an old relay (supersede, admission, repair re-arm); a config-injected revocation seeded by the old node with the remaining four of five validators holding exact quorum; and wallet-wipe recovery through an old upstream with resumed paired forwarding. Waits are event-driven on terminal log facts with heartbeat logging. --- .../scenarios/manifest_revocation_live.py | 96 +++++++++++++++ .../manifest_rotation_propagation.py | 110 ++++++++++++++++++ .../scenarios/manifest_validation_order.yml | 32 +++++ .testnet/scenarios/manifest_wipe_recovery.py | 97 +++++++++++++++ 4 files changed, 335 insertions(+) create mode 100644 .testnet/scenarios/manifest_revocation_live.py create mode 100644 .testnet/scenarios/manifest_rotation_propagation.py create mode 100644 .testnet/scenarios/manifest_wipe_recovery.py diff --git a/.testnet/scenarios/manifest_revocation_live.py b/.testnet/scenarios/manifest_revocation_live.py new file mode 100644 index 0000000000..829a5b3223 --- /dev/null +++ b/.testnet/scenarios/manifest_revocation_live.py @@ -0,0 +1,96 @@ +"""Revoke one of five validators live and watch the terminal update propagate. + +Five validators are used deliberately: revoking one leaves four of five — +exactly the 80% quorum — so the network keeps closing ledgers and the +scenario can assert liveness with the revoked member excluded, not merely a +halt. (On a three-node UNL the same revocation stops the network: small-net +quorums round up to everyone.) + +The revocation is installed by config on the OLD release node deliberately: +under the new transport slice a config-loaded revocation on an upgraded node +sits in its durable cache and is never gossiped (there is no connect-time +dump and no ambient manifest relay), while the old binary still dumps its +manifest cache on every fresh connection and relays what it accepts. The old +node is therefore the only in-topology seeder for a config-injected +revocation — that asymmetry is itself part of the documented durability +boundary. +""" + +from xahaud_scripts.testnet.scenario import ( + AssertionError as ScenarioAssertion, +) + + +async def scenario(ctx, log): + nodes = [0, 1, 2, 3, 4, 5] + # Validators 0-4 mesh through 0; the old release node 5 hangs off the + # mesh and is the revocation's config seeder. + expected = ctx.topology_edges( + [(0, 1), (0, 2), (0, 3), (0, 4), (1, 2), (3, 4), (4, 5)] + ) + + await ctx.apply_topology(expected, nodes=nodes, exact=False) + + # Normal operation first: everyone validates and validator 4's manifest + # is durably known, so the revocation supersedes real retained state. + await ctx.wait_for_ledgers(2, node_id=0, timeout=180) + + revoked = ctx.mark("revoked") + result = await ctx.revoke_validator(4, 5) + log( + "revoked validator n4 master via the old relay n5: " + f"{result['public_key']}" + ) + + # The restarted old node loads the revocation and seeds it through its + # legacy connect-time dump and accept-relay; its peer n4 spreads it into + # the mesh through the normal revocation relay lane. The old node's + # reconnect interval dominates the latency, so wait on the log fact. + deadline = 36 + for attempt in range(deadline): + try: + ctx.assert_log("Revoked", since=revoked, nodes=[4]) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-revocation-arrival: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-revocation-arrival") + + # Terminal manifest applied and relayed onward by upgraded nodes: first + # at the old seeder's direct peer, then across the mesh. + ctx.assert_log( + "manifest_revocation accepted_for_relay", + since=revoked, + nodes=[4], + ) + for attempt in range(deadline): + try: + ctx.assert_log("Revoked", since=revoked, nodes=[0]) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-mesh-revocation: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-mesh-revocation") + ctx.assert_log( + "manifest_revocation accepted_for_relay", + since=revoked, + nodes=[0], + ) + + # Liveness with the revoked member excluded: four of five is exactly + # quorum, so ledgers keep closing. + await ctx.wait_for_ledgers(2, node_id=0, timeout=180) + + ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes) + ctx.assert_not_log("Validation: Too small", nodes=nodes) + + log( + "PASS: a config-injected revocation seeded by the old relay reached" + " the mesh, applied terminally on upgraded nodes, relayed onward," + " and the remaining four validators kept the network live at exact" + " quorum" + ) diff --git a/.testnet/scenarios/manifest_rotation_propagation.py b/.testnet/scenarios/manifest_rotation_propagation.py new file mode 100644 index 0000000000..75956d896c --- /dev/null +++ b/.testnet/scenarios/manifest_rotation_propagation.py @@ -0,0 +1,110 @@ +"""Rotate a validator's manifest mid-run and watch the bump propagate. + +"Heals on a sequence bump" is the load-bearing healing claim of the +manifest-before-validation transport slice. This exercises it live for the +first time: the validator mints sequence 2 and restarts; its validations then +carry the new manifest; an old release relay forwards the existing envelopes; +the upgraded observer supersedes its retained knowledge and admits sequence 2; +and the observer's repair lane re-arms at the newer sequence when the old +relay's later validations arrive naked. +""" + +from xahaud_scripts.testnet.scenario import ( + AssertionError as ScenarioAssertion, +) + + + +async def scenario(ctx, log): + nodes = [0, 1, 2] + expected = ctx.topology_edges([(0, 1), (1, 2)]) + + await ctx.apply_topology(expected, nodes=nodes, exact=False) + + # Baseline: sequence 1 propagates through the old relay and is admitted + # by the upgraded observer before we rotate anything. The validator can + # race several ledgers ahead of propagation under fast bootstrap, so + # wait on the log fact itself. + await ctx.wait_for_ledgers(2, node_id=0, timeout=120) + deadline = 24 + for attempt in range(deadline): + try: + ctx.assert_log( + "manifest_validation single_manifest_processed .*sequence=1", + nodes=[2], + ) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-baseline-admission: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-baseline-admission") + + rotated = ctx.mark("rotated") + rotation = await ctx.rotate_validator_manifest(0) + assert rotation["sequence"] == 2, rotation + log(f"rotated validator n0 to manifest sequence {rotation['sequence']}") + + # The restarted validator re-joins and validates under the new signing + # key; always-send carries the sequence-2 prerequisite with each + # validation, and the old relay's one-shot manifest forward plus later + # naked relays exercise both the supersede and the repair re-arm. Wait + # on the terminal log fact rather than ledger counts: in this topology + # only the validator advances its ledger, and its own restart closed the + # node-0 WebSocket ledger feed. The repair re-arm is the last event in + # the causal chain, so everything else must precede it. + deadline = 36 # polls at 5s => 180s budget at ~16s consensus rounds + for attempt in range(deadline): + try: + ctx.assert_log( + "manifest_validation repair_sent .*sequence=2", + since=rotated, + nodes=[2], + ) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-repair-rearm: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-repair-rearm") + + ctx.assert_log( + "manifest_validation pair_enqueued .*sequence=2", + since=rotated, + nodes=[0], + ) + ctx.assert_log( + "manifest_validation candidate_staged .*sequence=2", + since=rotated, + nodes=[2], + ) + ctx.assert_log( + "manifest_validation candidate_matched", + since=rotated, + nodes=[2], + ) + ctx.assert_log( + "manifest_validation single_manifest_processed .*sequence=2" + " .*disposition=accepted", + since=rotated, + nodes=[2], + ) + ctx.assert_log_order( + [ + "manifest_validation single_manifest_processed .*sequence=2", + "manifest_validation repair_sent .*sequence=2", + ], + since=rotated, + nodes=[2], + ) + + ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes) + ctx.assert_not_log("Validation: Too small", nodes=nodes) + + log( + "PASS: mid-run rotation to sequence 2 propagated through an old" + " relay; the upgraded observer superseded and admitted the new" + " manifest and re-armed its repair lane at the new sequence" + ) diff --git a/.testnet/scenarios/manifest_validation_order.yml b/.testnet/scenarios/manifest_validation_order.yml index 48a6d56e80..1644d07d6d 100644 --- a/.testnet/scenarios/manifest_validation_order.yml +++ b/.testnet/scenarios/manifest_validation_order.yml @@ -20,3 +20,35 @@ tests: log_levels: Protocol: debug Validations: debug + + - name: manifest_rotation_propagation + script: .testnet/scenarios/manifest_rotation_propagation.py + network: + validators: 1 + node_binaries: + 1: "@release-3350" + log_levels: + Protocol: debug + Validations: debug + + - name: manifest_revocation_live + script: .testnet/scenarios/manifest_revocation_live.py + network: + node_count: 6 + validators: 5 + node_binaries: + 5: "@release-3350" + log_levels: + Protocol: debug + Validations: debug + + - name: manifest_wipe_recovery + script: .testnet/scenarios/manifest_wipe_recovery.py + network: + node_count: 4 + validators: 1 + node_binaries: + 1: "@release-3350" + log_levels: + Protocol: debug + Validations: debug diff --git a/.testnet/scenarios/manifest_wipe_recovery.py b/.testnet/scenarios/manifest_wipe_recovery.py new file mode 100644 index 0000000000..118261e945 --- /dev/null +++ b/.testnet/scenarios/manifest_wipe_recovery.py @@ -0,0 +1,97 @@ +"""Wipe a mid-chain node's wallet and watch both generations heal it. + +The forever cache's informal replicated history is gone from new nodes, so +this pins what replaces it. A wiped upgraded node reconnects with empty +durable state; its old-release upstream re-seeds it through the legacy +connect-time dump lane, its own validation traffic re-pairs through +always-send, and it resumes forwarding paired prerequisites downstream. The +recovery, not a starvation, is the honest live boundary: every reachable +topology heals, because old peers dump at connect and new peers re-send the +prerequisite with every validation. +""" + +from xahaud_scripts.testnet.scenario import ( + AssertionError as ScenarioAssertion, +) + + + +async def scenario(ctx, log): + nodes = [0, 1, 2, 3] + expected = ctx.topology_edges([(0, 1), (1, 2), (2, 3)]) + + await ctx.apply_topology(expected, nodes=nodes, exact=False) + + # Baseline: knowledge reaches the end of the chain. The tail observer + # receives paired traffic from the upgraded mid-chain node. Admission at + # the tail implies the whole upstream chain, so wait on that log fact — + # four hops can trail the validator's fast-bootstrap ledger count. + await ctx.wait_for_ledgers(2, node_id=0, timeout=120) + deadline = 24 + for attempt in range(deadline): + try: + ctx.assert_log( + "manifest_validation single_manifest_processed .*sequence=1", + nodes=[3], + ) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-baseline-chain: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-baseline-chain") + ctx.assert_log( + "manifest_validation single_manifest_processed .*sequence=1", + nodes=[2], + ) + + wiped = ctx.mark("wiped") + await ctx.restart_node(2, wipe_wallet_db=True) + log("restarted n2 with a wiped wallet database") + + await ctx.wait_for_ledgers(3, node_id=0, timeout=180) + + # Wait on the terminal log fact: resumed paired forwarding downstream is + # the last event in the recovery chain, so re-learning and re-admission + # must precede it. + deadline = 36 + for attempt in range(deadline): + try: + ctx.assert_log( + "manifest_validation send_prerequisite .*sequence=1", + since=wiped, + nodes=[2], + ) + break + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"await-resumed-forwarding: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name="await-resumed-forwarding") + + # Recovery: the wiped node re-learned the validator identity from live + # traffic (the old upstream's connect-time dump arrives as a singleton + # candidate; the next validation proves it) and admitted it durably + # again. + ctx.assert_log( + "manifest_validation candidate_staged .*sequence=1", + since=wiped, + nodes=[2], + ) + ctx.assert_log( + "manifest_validation single_manifest_processed .*sequence=1" + " .*disposition=accepted", + since=wiped, + nodes=[2], + ) + + ctx.assert_not_log("Validation forwarded by peer is invalid", nodes=nodes) + ctx.assert_not_log("Validation: Too small", nodes=nodes) + + log( + "PASS: a wallet-wiped mid-chain node re-learned the validator" + " identity from live traffic through an old upstream, re-admitted it" + " durably, and resumed paired forwarding to the tail observer" + )