From 54b59d8017f7d449ec1adfa8502cea5afdcab72b Mon Sep 17 00:00:00 2001 From: Nicholas Dudfield Date: Sat, 29 Aug 2026 19:19:18 +0700 Subject: [PATCH] test(testnet): share log waits and simplify revocation overlay --- .../scenarios/manifest_revocation_live.py | 60 +++++++++++------- .../manifest_rotation_propagation.py | 61 +++++++++--------- .testnet/scenarios/manifest_wipe_recovery.py | 62 ++++++++++--------- 3 files changed, 100 insertions(+), 83 deletions(-) diff --git a/.testnet/scenarios/manifest_revocation_live.py b/.testnet/scenarios/manifest_revocation_live.py index 829a5b3223..b7d716750a 100644 --- a/.testnet/scenarios/manifest_revocation_live.py +++ b/.testnet/scenarios/manifest_revocation_live.py @@ -21,12 +21,27 @@ from xahaud_scripts.testnet.scenario import ( ) +async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36): + # Catch the runner's ScenarioAssertion, not the builtin. The scenario + # module shadows AssertionError without subclassing it. + for attempt in range(deadline): + try: + ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {})) + return + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"{name}: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name=name) + + async def scenario(ctx, log): nodes = [0, 1, 2, 3, 4, 5] - # Validators 0-4 mesh through 0; the old release node 5 hangs off the - # mesh and is the revocation's config seeder. + # Validators 0-4 meet at n0. The old release node n5 is a pendant on n4 + # so it can seed the revocation without joining the UNL mesh. expected = ctx.topology_edges( - [(0, 1), (0, 2), (0, 3), (0, 4), (1, 2), (3, 4), (4, 5)] + [(0, 1), (0, 2), (0, 3), (0, 4), (4, 5)] ) await ctx.apply_topology(expected, nodes=nodes, exact=False) @@ -46,17 +61,15 @@ async def scenario(ctx, log): # legacy connect-time dump and accept-relay; its peer n4 spreads it into # the mesh through the normal revocation relay lane. The old node's # reconnect interval dominates the latency, so wait on the log fact. - deadline = 36 - for attempt in range(deadline): - try: - ctx.assert_log("Revoked", since=revoked, nodes=[4]) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-revocation-arrival: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-revocation-arrival") + await _await_log( + ctx, + log, + "Revoked", + nodes=[4], + name="await-revocation-arrival", + since=revoked, + deadline=36, + ) # Terminal manifest applied and relayed onward by upgraded nodes: first # at the old seeder's direct peer, then across the mesh. @@ -65,16 +78,15 @@ async def scenario(ctx, log): since=revoked, nodes=[4], ) - for attempt in range(deadline): - try: - ctx.assert_log("Revoked", since=revoked, nodes=[0]) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-mesh-revocation: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-mesh-revocation") + await _await_log( + ctx, + log, + "Revoked", + nodes=[0], + name="await-mesh-revocation", + since=revoked, + deadline=36, + ) ctx.assert_log( "manifest_revocation accepted_for_relay", since=revoked, diff --git a/.testnet/scenarios/manifest_rotation_propagation.py b/.testnet/scenarios/manifest_rotation_propagation.py index 75956d896c..5ea278bb35 100644 --- a/.testnet/scenarios/manifest_rotation_propagation.py +++ b/.testnet/scenarios/manifest_rotation_propagation.py @@ -14,6 +14,20 @@ from xahaud_scripts.testnet.scenario import ( ) +async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36): + # Catch the runner's ScenarioAssertion, not the builtin. The scenario + # module shadows AssertionError without subclassing it. + for attempt in range(deadline): + try: + ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {})) + return + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"{name}: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name=name) + async def scenario(ctx, log): nodes = [0, 1, 2] @@ -26,20 +40,14 @@ async def scenario(ctx, log): # race several ledgers ahead of propagation under fast bootstrap, so # wait on the log fact itself. await ctx.wait_for_ledgers(2, node_id=0, timeout=120) - deadline = 24 - for attempt in range(deadline): - try: - ctx.assert_log( - "manifest_validation single_manifest_processed .*sequence=1", - nodes=[2], - ) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-baseline-admission: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-baseline-admission") + await _await_log( + ctx, + log, + "manifest_validation single_manifest_processed .*sequence=1", + nodes=[2], + name="await-baseline-admission", + deadline=24, + ) rotated = ctx.mark("rotated") rotation = await ctx.rotate_validator_manifest(0) @@ -54,21 +62,16 @@ async def scenario(ctx, log): # only the validator advances its ledger, and its own restart closed the # node-0 WebSocket ledger feed. The repair re-arm is the last event in # the causal chain, so everything else must precede it. - deadline = 36 # polls at 5s => 180s budget at ~16s consensus rounds - for attempt in range(deadline): - try: - ctx.assert_log( - "manifest_validation repair_sent .*sequence=2", - since=rotated, - nodes=[2], - ) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-repair-rearm: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-repair-rearm") + # polls at 5s => 180s budget at ~16s consensus rounds + await _await_log( + ctx, + log, + "manifest_validation repair_sent .*sequence=2", + nodes=[2], + name="await-repair-rearm", + since=rotated, + deadline=36, + ) ctx.assert_log( "manifest_validation pair_enqueued .*sequence=2", diff --git a/.testnet/scenarios/manifest_wipe_recovery.py b/.testnet/scenarios/manifest_wipe_recovery.py index 118261e945..5e039308bb 100644 --- a/.testnet/scenarios/manifest_wipe_recovery.py +++ b/.testnet/scenarios/manifest_wipe_recovery.py @@ -15,6 +15,20 @@ from xahaud_scripts.testnet.scenario import ( ) +async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36): + # Catch the runner's ScenarioAssertion, not the builtin. The scenario + # module shadows AssertionError without subclassing it. + for attempt in range(deadline): + try: + ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {})) + return + except ScenarioAssertion: + if attempt == deadline - 1: + raise + if attempt % 6 == 5: + log(f"{name}: attempt {attempt + 1}/{deadline}") + await ctx.sleep(5, name=name) + async def scenario(ctx, log): nodes = [0, 1, 2, 3] @@ -25,22 +39,16 @@ async def scenario(ctx, log): # Baseline: knowledge reaches the end of the chain. The tail observer # receives paired traffic from the upgraded mid-chain node. Admission at # the tail implies the whole upstream chain, so wait on that log fact — - # four hops can trail the validator's fast-bootstrap ledger count. + # three hops can trail the validator's fast-bootstrap ledger count. await ctx.wait_for_ledgers(2, node_id=0, timeout=120) - deadline = 24 - for attempt in range(deadline): - try: - ctx.assert_log( - "manifest_validation single_manifest_processed .*sequence=1", - nodes=[3], - ) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-baseline-chain: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-baseline-chain") + await _await_log( + ctx, + log, + "manifest_validation single_manifest_processed .*sequence=1", + nodes=[3], + name="await-baseline-chain", + deadline=24, + ) ctx.assert_log( "manifest_validation single_manifest_processed .*sequence=1", nodes=[2], @@ -55,21 +63,15 @@ async def scenario(ctx, log): # Wait on the terminal log fact: resumed paired forwarding downstream is # the last event in the recovery chain, so re-learning and re-admission # must precede it. - deadline = 36 - for attempt in range(deadline): - try: - ctx.assert_log( - "manifest_validation send_prerequisite .*sequence=1", - since=wiped, - nodes=[2], - ) - break - except ScenarioAssertion: - if attempt == deadline - 1: - raise - if attempt % 6 == 5: - log(f"await-resumed-forwarding: attempt {attempt + 1}/{deadline}") - await ctx.sleep(5, name="await-resumed-forwarding") + await _await_log( + ctx, + log, + "manifest_validation send_prerequisite .*sequence=1", + nodes=[2], + name="await-resumed-forwarding", + since=wiped, + deadline=36, + ) # Recovery: the wiped node re-learned the validator identity from live # traffic (the old upstream's connect-time dump arrives as a singleton