test(testnet): share log waits and simplify revocation overlay

This commit is contained in:
Nicholas Dudfield
2026-08-29 19:19:18 +07:00
parent 39629c4cf5
commit 54b59d8017
3 changed files with 100 additions and 83 deletions

View File

@@ -21,12 +21,27 @@ from xahaud_scripts.testnet.scenario import (
)
async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36):
# Catch the runner's ScenarioAssertion, not the builtin. The scenario
# module shadows AssertionError without subclassing it.
for attempt in range(deadline):
try:
ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {}))
return
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"{name}: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name=name)
async def scenario(ctx, log):
nodes = [0, 1, 2, 3, 4, 5]
# Validators 0-4 mesh through 0; the old release node 5 hangs off the
# mesh and is the revocation's config seeder.
# Validators 0-4 meet at n0. The old release node n5 is a pendant on n4
# so it can seed the revocation without joining the UNL mesh.
expected = ctx.topology_edges(
[(0, 1), (0, 2), (0, 3), (0, 4), (1, 2), (3, 4), (4, 5)]
[(0, 1), (0, 2), (0, 3), (0, 4), (4, 5)]
)
await ctx.apply_topology(expected, nodes=nodes, exact=False)
@@ -46,17 +61,15 @@ async def scenario(ctx, log):
# legacy connect-time dump and accept-relay; its peer n4 spreads it into
# the mesh through the normal revocation relay lane. The old node's
# reconnect interval dominates the latency, so wait on the log fact.
deadline = 36
for attempt in range(deadline):
try:
ctx.assert_log("Revoked", since=revoked, nodes=[4])
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-revocation-arrival: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-revocation-arrival")
await _await_log(
ctx,
log,
"Revoked",
nodes=[4],
name="await-revocation-arrival",
since=revoked,
deadline=36,
)
# Terminal manifest applied and relayed onward by upgraded nodes: first
# at the old seeder's direct peer, then across the mesh.
@@ -65,16 +78,15 @@ async def scenario(ctx, log):
since=revoked,
nodes=[4],
)
for attempt in range(deadline):
try:
ctx.assert_log("Revoked", since=revoked, nodes=[0])
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-mesh-revocation: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-mesh-revocation")
await _await_log(
ctx,
log,
"Revoked",
nodes=[0],
name="await-mesh-revocation",
since=revoked,
deadline=36,
)
ctx.assert_log(
"manifest_revocation accepted_for_relay",
since=revoked,

View File

@@ -14,6 +14,20 @@ from xahaud_scripts.testnet.scenario import (
)
async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36):
# Catch the runner's ScenarioAssertion, not the builtin. The scenario
# module shadows AssertionError without subclassing it.
for attempt in range(deadline):
try:
ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {}))
return
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"{name}: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name=name)
async def scenario(ctx, log):
nodes = [0, 1, 2]
@@ -26,20 +40,14 @@ async def scenario(ctx, log):
# race several ledgers ahead of propagation under fast bootstrap, so
# wait on the log fact itself.
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
deadline = 24
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-baseline-admission: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-baseline-admission")
await _await_log(
ctx,
log,
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[2],
name="await-baseline-admission",
deadline=24,
)
rotated = ctx.mark("rotated")
rotation = await ctx.rotate_validator_manifest(0)
@@ -54,21 +62,16 @@ async def scenario(ctx, log):
# only the validator advances its ledger, and its own restart closed the
# node-0 WebSocket ledger feed. The repair re-arm is the last event in
# the causal chain, so everything else must precede it.
deadline = 36 # polls at 5s => 180s budget at ~16s consensus rounds
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation repair_sent .*sequence=2",
since=rotated,
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-repair-rearm: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-repair-rearm")
# polls at 5s => 180s budget at ~16s consensus rounds
await _await_log(
ctx,
log,
"manifest_validation repair_sent .*sequence=2",
nodes=[2],
name="await-repair-rearm",
since=rotated,
deadline=36,
)
ctx.assert_log(
"manifest_validation pair_enqueued .*sequence=2",

View File

@@ -15,6 +15,20 @@ from xahaud_scripts.testnet.scenario import (
)
async def _await_log(ctx, log, pattern, *, nodes, name, since=None, deadline=36):
# Catch the runner's ScenarioAssertion, not the builtin. The scenario
# module shadows AssertionError without subclassing it.
for attempt in range(deadline):
try:
ctx.assert_log(pattern, nodes=nodes, **({"since": since} if since else {}))
return
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"{name}: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name=name)
async def scenario(ctx, log):
nodes = [0, 1, 2, 3]
@@ -25,22 +39,16 @@ async def scenario(ctx, log):
# Baseline: knowledge reaches the end of the chain. The tail observer
# receives paired traffic from the upgraded mid-chain node. Admission at
# the tail implies the whole upstream chain, so wait on that log fact —
# four hops can trail the validator's fast-bootstrap ledger count.
# three hops can trail the validator's fast-bootstrap ledger count.
await ctx.wait_for_ledgers(2, node_id=0, timeout=120)
deadline = 24
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[3],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-baseline-chain: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-baseline-chain")
await _await_log(
ctx,
log,
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[3],
name="await-baseline-chain",
deadline=24,
)
ctx.assert_log(
"manifest_validation single_manifest_processed .*sequence=1",
nodes=[2],
@@ -55,21 +63,15 @@ async def scenario(ctx, log):
# Wait on the terminal log fact: resumed paired forwarding downstream is
# the last event in the recovery chain, so re-learning and re-admission
# must precede it.
deadline = 36
for attempt in range(deadline):
try:
ctx.assert_log(
"manifest_validation send_prerequisite .*sequence=1",
since=wiped,
nodes=[2],
)
break
except ScenarioAssertion:
if attempt == deadline - 1:
raise
if attempt % 6 == 5:
log(f"await-resumed-forwarding: attempt {attempt + 1}/{deadline}")
await ctx.sleep(5, name="await-resumed-forwarding")
await _await_log(
ctx,
log,
"manifest_validation send_prerequisite .*sequence=1",
nodes=[2],
name="await-resumed-forwarding",
since=wiped,
deadline=36,
)
# Recovery: the wiped node re-learned the validator identity from live
# traffic (the old upstream's connect-time dump arrives as a singleton