mirror of
https://github.com/Xahau/xahaud.git
synced 2026-09-28 07:58:05 +00:00
fix(test): restore current virtual time before threaded node restart
Fix issue 2029: live clocks advance while a stopped node has no TimeKeeper. Keep the threaded runner network-time cursor and apply it before startup on restart or late add, preserving existing per-tick conversion and stepping clock semantics. A 25-second offline regression reconnects through production handshake checks; restoring the old stop-time snapshot makes it fail with the observed clock exception.
This commit is contained in:
@@ -860,6 +860,84 @@ class ThreadedExtensions_test : public beast::unit_test::suite
|
||||
BEAST_EXPECT(consensus.maxAcceptLockHoldNs() > 0);
|
||||
}
|
||||
|
||||
void
|
||||
testRestartClock()
|
||||
{
|
||||
testcase("threaded restart catches up 25 seconds before relinking");
|
||||
using namespace std::chrono_literals;
|
||||
MultiNode net(*this, /*virtualClock=*/true, /*stepping=*/false);
|
||||
auto const a = ValidatorKey::fromPassphrase("restart-clock-a");
|
||||
auto const b = ValidatorKey::fromPassphrase("restart-clock-b");
|
||||
std::vector<std::string> const unl{a.pubKey, b.pubKey};
|
||||
auto const factory = [](Application& app) {
|
||||
return std::unique_ptr<Overlay>(std::make_unique<SimOverlay>(app));
|
||||
};
|
||||
auto const configure = [](Config& cfg) { cfg.NETWORK_ID = networkID; };
|
||||
for (auto const& seed : {a.seed, b.seed, std::string{}})
|
||||
net.add(TrustConfig{seed, unl}, factory, {}, false, configure);
|
||||
if (!BEAST_EXPECT(net.allUp()))
|
||||
return;
|
||||
for (std::size_t i = 0; i < 3; ++i)
|
||||
for (std::size_t j = i + 1; j < 3; ++j)
|
||||
if (!BEAST_EXPECT(net.simConnect(i, j) != nullptr))
|
||||
return;
|
||||
if (!BEAST_EXPECT(net.waitForPeers(2, 5s)))
|
||||
return;
|
||||
for (int i = 0; i < 120 && net.minValidated() < 3; ++i)
|
||||
(void)net.threadedTick(1s);
|
||||
if (!BEAST_EXPECT(net.minValidated() >= 3))
|
||||
return;
|
||||
|
||||
constexpr std::size_t returning = 2;
|
||||
bool saved = false;
|
||||
for (int i = 0; i < 60; ++i)
|
||||
{
|
||||
auto const row =
|
||||
net[returning].app().getRelationalDatabase().getMaxLedgerSeq();
|
||||
if (row && *row >= 3)
|
||||
{
|
||||
saved = true;
|
||||
break;
|
||||
}
|
||||
(void)net.threadedTick(1s);
|
||||
}
|
||||
if (!BEAST_EXPECT(saved))
|
||||
return;
|
||||
|
||||
auto const stoppedAt = net[returning].clock().now();
|
||||
net.stopNode(returning);
|
||||
for (int i = 0; i < 25; ++i)
|
||||
(void)net.threadedTick(1s);
|
||||
auto const expected = net[0].clock().now();
|
||||
BEAST_EXPECT(expected == stoppedAt + 25s);
|
||||
if (!BEAST_EXPECT(net.restartNode(returning).isUp()))
|
||||
return;
|
||||
log << " restart clock: stopped="
|
||||
<< stoppedAt.time_since_epoch().count()
|
||||
<< " current=" << expected.time_since_epoch().count()
|
||||
<< " restored="
|
||||
<< net[returning].clock().now().time_since_epoch().count()
|
||||
<< std::endl;
|
||||
BEAST_EXPECT(net[returning].clock().now() == expected);
|
||||
|
||||
// Do not tick between restart and relink. Production handshake
|
||||
// verification must already see the current clock, not the saved one.
|
||||
bool linked = true;
|
||||
try
|
||||
{
|
||||
for (std::size_t i = 0; i < returning; ++i)
|
||||
linked = net.simConnect(i, returning) != nullptr && linked;
|
||||
}
|
||||
catch (std::exception const& ex)
|
||||
{
|
||||
log << " restart clock relink: " << ex.what() << std::endl;
|
||||
linked = false;
|
||||
}
|
||||
if (!BEAST_EXPECT(linked))
|
||||
return;
|
||||
BEAST_EXPECT(net.waitForPeers(2, 5s));
|
||||
}
|
||||
|
||||
void
|
||||
testAcceptLockWaits()
|
||||
{
|
||||
@@ -1021,8 +1099,14 @@ public:
|
||||
void
|
||||
run() override
|
||||
{
|
||||
if (arg() == "restart-clock-only")
|
||||
{
|
||||
testRestartClock();
|
||||
return;
|
||||
}
|
||||
testSimulateReenters();
|
||||
testAcceptLockWaits();
|
||||
testRestartClock();
|
||||
realThreads();
|
||||
}
|
||||
};
|
||||
|
||||
@@ -245,8 +245,8 @@ struct NodeSpec
|
||||
// the LAST gated timer, virtualized). Empty -> production nullptr:
|
||||
// PeerImp keeps its raw asio member untouched.
|
||||
TimeoutCounterTimerFactory peerTimerFactory;
|
||||
// NetClock at stop. Applied to the new keeper before the app runs so a
|
||||
// loaded ledger does not sit ahead of a keeper that starts at the epoch.
|
||||
// Startup NetClock override, applied before the app runs. Threaded virtual
|
||||
// nodes use the runner's current time, including time spent stopped.
|
||||
std::optional<NetClock::time_point> restoredNetClock;
|
||||
};
|
||||
|
||||
@@ -608,6 +608,10 @@ class MultiNode
|
||||
// callback maps scheduler virtual time onto each node's NetClock from this
|
||||
// base.
|
||||
NetClock::time_point netBase_{};
|
||||
// Current network time in non-stepping virtual mode, owned by the test
|
||||
// thread. Unlike live node clocks, it advances while every node is stopped.
|
||||
// Keep the same per-tick whole-second conversion as the existing driver.
|
||||
std::optional<NetClock::time_point> threadedNetTime_;
|
||||
struct NodeSlot
|
||||
{
|
||||
TempDir dbDir;
|
||||
@@ -841,7 +845,11 @@ public:
|
||||
/*injectedPrng=*/slots_[id]->prng.get(),
|
||||
std::move(timerFactory),
|
||||
std::move(peerTimerFactory),
|
||||
/*restoredNetClock=*/std::nullopt}));
|
||||
/*restoredNetClock=*/threadedNetTime_}));
|
||||
|
||||
if (steadyClock_ && !stepper_ && !threadedNetTime_ &&
|
||||
nodes_.back()->isUp())
|
||||
threadedNetTime_ = nodes_.back()->clock().now();
|
||||
|
||||
// Capture the genesis NetClock base from the first node for
|
||||
// syncClocks(). Gate on isUp(): a setup failure resets app_ (destroying
|
||||
@@ -853,7 +861,8 @@ public:
|
||||
// at genesis close time while the network's virtual clocks are far
|
||||
// ahead — the real handshake rejects that skew ("Peer clock is too
|
||||
// far off"). Sync every clock to scheduler time, exactly as
|
||||
// restartNode does; a no-op for the normal t=0 bring-up.
|
||||
// restartNode does; a no-op for the normal t=0 bring-up. Threaded
|
||||
// virtual nodes receive threadedNetTime_ before app startup above.
|
||||
if (stepper_ && nodes_.back()->isUp())
|
||||
syncClocks(stepper_->now());
|
||||
return *nodes_.back();
|
||||
@@ -980,7 +989,8 @@ private:
|
||||
/*injectedPrng=*/slots_[i]->prng.get(),
|
||||
std::move(timerFactory),
|
||||
std::move(peerTimerFactory),
|
||||
slots_[i]->savedNetClock});
|
||||
threadedNetTime_ ? threadedNetTime_
|
||||
: slots_[i]->savedNetClock});
|
||||
if (stepper_ && nodes_[i]->isUp())
|
||||
syncClocks(stepper_->now());
|
||||
return *nodes_[i];
|
||||
@@ -1547,13 +1557,9 @@ public:
|
||||
stats.beat = ++threadedBeat_;
|
||||
stats.transportStart = simActivitySnapshot();
|
||||
|
||||
// 1) steady clock (elapsed-time source for openTime / round duration).
|
||||
steadyClock_->advance(dt);
|
||||
// 2) NetClock in lockstep (truncates to whole seconds; pass dt >= 1s).
|
||||
auto const netDt = duration_cast<NetClock::duration>(dt);
|
||||
for (auto& n : nodes_)
|
||||
if (n)
|
||||
n->clock().set(n->clock().now() + netDt);
|
||||
// Advance both domains once, including the runner's current NetClock
|
||||
// used when a stopped node returns before the next heartbeat.
|
||||
advanceInjectedClocks(dt);
|
||||
|
||||
struct Signal
|
||||
{
|
||||
@@ -2195,6 +2201,8 @@ private:
|
||||
|
||||
steadyClock_->advance(dt);
|
||||
auto const netDt = std::chrono::duration_cast<NetClock::duration>(dt);
|
||||
if (threadedNetTime_)
|
||||
*threadedNetTime_ += netDt;
|
||||
for (auto& n : nodes_)
|
||||
if (n)
|
||||
n->clock().set(n->clock().now() + netDt);
|
||||
|
||||
Reference in New Issue
Block a user