mirror of
https://github.com/XRPLF/rippled.git
synced 2026-08-29 02:00:56 +00:00
Compare commits
354 Commits
develop
...
ximinez/on
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
814d14ad2b | ||
|
|
e7cfc46275 | ||
|
|
c3c555c260 | ||
|
|
3cc27a4a67 | ||
|
|
17bec9e4d4 | ||
|
|
0e54845db0 | ||
|
|
6f1f53a152 | ||
|
|
080ad9fe34 | ||
|
|
2620ae59ae | ||
|
|
d3d5f39787 | ||
|
|
77793df659 | ||
|
|
6abcc9fe1a | ||
|
|
e67e9f6660 | ||
|
|
6718b3276b | ||
|
|
642dbad885 | ||
|
|
833433b25e | ||
|
|
84e32ed33c | ||
|
|
23369e114a | ||
|
|
d3d0c6366a | ||
|
|
c15c7bfd54 | ||
|
|
af61140cd9 | ||
|
|
4e1de43939 | ||
|
|
dc71651429 | ||
|
|
e372201499 | ||
|
|
8b2a7f6253 | ||
|
|
6211dbd1cc | ||
|
|
d89063acae | ||
|
|
0e328d8277 | ||
|
|
7d55bb7d8e | ||
|
|
14c190fd31 | ||
|
|
efe843d6b2 | ||
|
|
41e422d2c1 | ||
|
|
7f10f3938f | ||
|
|
9c8b45f0ed | ||
|
|
f7e96beba7 | ||
|
|
5a7a715283 | ||
|
|
8f1974d0ec | ||
|
|
f84e07427c | ||
|
|
d375fa356f | ||
|
|
5d5e4d3405 | ||
|
|
1eacb33d72 | ||
|
|
fafc48bda3 | ||
|
|
9a2b6866a7 | ||
|
|
8a49f93b78 | ||
|
|
c5c85eef48 | ||
|
|
5aed870ff0 | ||
|
|
0fe2be63f4 | ||
|
|
4c93c8d5be | ||
|
|
5817d3547e | ||
|
|
7a8d86b0b1 | ||
|
|
1bbe54ad60 | ||
|
|
0bb9716784 | ||
|
|
295c71b007 | ||
|
|
78f3148e4c | ||
|
|
894263bb89 | ||
|
|
9de4e7e58d | ||
|
|
ec844351d9 | ||
|
|
41b4cf36e6 | ||
|
|
6f7ad510e3 | ||
|
|
5db10a4d3b | ||
|
|
919cfd5f49 | ||
|
|
476a64d5ce | ||
|
|
303952aeb4 | ||
|
|
85a6dfee1f | ||
|
|
621f876337 | ||
|
|
da2f36c8a6 | ||
|
|
fab69c28d2 | ||
|
|
afedf445f0 | ||
|
|
fe1b1c8f92 | ||
|
|
e61b134742 | ||
|
|
2ba3edaf14 | ||
|
|
7cf063b1d8 | ||
|
|
5f087f6446 | ||
|
|
a2446e8f99 | ||
|
|
6accc87a10 | ||
|
|
313d0d188b | ||
|
|
53dc3b4051 | ||
|
|
b589985dae | ||
|
|
180c979835 | ||
|
|
88665a6de0 | ||
|
|
66526a8a48 | ||
|
|
18c11097b6 | ||
|
|
bfe54849c9 | ||
|
|
831ac87627 | ||
|
|
1b7af3165d | ||
|
|
48ca6e0025 | ||
|
|
0047ed4db2 | ||
|
|
6dc73cb528 | ||
|
|
7f58c3d7d6 | ||
|
|
86f52c412c | ||
|
|
4dce2f22b6 | ||
|
|
fe8368b970 | ||
|
|
d68273c755 | ||
|
|
df489c7a6e | ||
|
|
b45038f70a | ||
|
|
44cc4e656f | ||
|
|
ac3435bf0a | ||
|
|
bd2f5dbd35 | ||
|
|
4644347d21 | ||
|
|
e34e05ebec | ||
|
|
9b22c6c80b | ||
|
|
d4e166bbdc | ||
|
|
b80776b0fe | ||
|
|
9af4fb521e | ||
|
|
7ee290674c | ||
|
|
68f050b186 | ||
|
|
2be13c18a1 | ||
|
|
df5fa5aade | ||
|
|
51ef3a7343 | ||
|
|
dd4eca22b7 | ||
|
|
1116dbfd0e | ||
|
|
258d5e1f1f | ||
|
|
4270367efc | ||
|
|
fad34e852d | ||
|
|
f5b4ac358f | ||
|
|
875bcc530e | ||
|
|
e75f5b101b | ||
|
|
95f74d61b2 | ||
|
|
4e12b787be | ||
|
|
8b6c80027b | ||
|
|
ace73678cf | ||
|
|
12bbba00b0 | ||
|
|
2affb66f50 | ||
|
|
d09e785c39 | ||
|
|
97fdf310f9 | ||
|
|
d3ce76825c | ||
|
|
1e7e274727 | ||
|
|
b272d71cc7 | ||
|
|
21268d9c36 | ||
|
|
932e22df7d | ||
|
|
652a7cd225 | ||
|
|
4a224dbfa4 | ||
|
|
d2a5981f87 | ||
|
|
a977836630 | ||
|
|
bf075200bb | ||
|
|
ee49e76a16 | ||
|
|
c1318990f3 | ||
|
|
69128294f8 | ||
|
|
149e884ebc | ||
|
|
d27353225c | ||
|
|
13a0f77eb5 | ||
|
|
549c093398 | ||
|
|
fb66ca7a2e | ||
|
|
e2af73b0a0 | ||
|
|
14c3c9a256 | ||
|
|
4b6851a287 | ||
|
|
d182673d24 | ||
|
|
bea609d805 | ||
|
|
56eeb20bc8 | ||
|
|
e293a3d918 | ||
|
|
3dff580d39 | ||
|
|
487f5a0fd3 | ||
|
|
d8134f98e9 | ||
|
|
450a623d4b | ||
|
|
9e517be4ce | ||
|
|
b8370438fb | ||
|
|
ecb5604d3b | ||
|
|
c4527e7b0f | ||
|
|
778e2b3ce8 | ||
|
|
295d03aec8 | ||
|
|
9262c2e624 | ||
|
|
b385a41aa5 | ||
|
|
4eb9726097 | ||
|
|
28a38ef1cc | ||
|
|
39f9380b2b | ||
|
|
239fcaceb4 | ||
|
|
8ab86f009e | ||
|
|
d4b58a74f4 | ||
|
|
ff900591ae | ||
|
|
a5b7471af6 | ||
|
|
e0734986dd | ||
|
|
8440f479e5 | ||
|
|
ac390622e0 | ||
|
|
5d807f0d6d | ||
|
|
b93294f26b | ||
|
|
ef95ace0f9 | ||
|
|
fc58bf6edf | ||
|
|
348555d5ba | ||
|
|
302802e42d | ||
|
|
5d881f87a3 | ||
|
|
a8a8035b32 | ||
|
|
45af14231f | ||
|
|
380bf274d0 | ||
|
|
460ec5eeea | ||
|
|
14be8ca4ea | ||
|
|
948264d44c | ||
|
|
9c816b2043 | ||
|
|
f68402acd1 | ||
|
|
5358e25eaa | ||
|
|
a34d1e5537 | ||
|
|
78a122943d | ||
|
|
06135e7203 | ||
|
|
f20425fa4f | ||
|
|
3e94546acc | ||
|
|
3b088ed0dc | ||
|
|
26182ed52e | ||
|
|
de2a3e10f5 | ||
|
|
e17f8554fc | ||
|
|
386a7192ba | ||
|
|
12cc6e424d | ||
|
|
c9deecf1b7 | ||
|
|
ddd1b49f38 | ||
|
|
c1b2a24005 | ||
|
|
86d88eca31 | ||
|
|
ef09eaea00 | ||
|
|
c504cfb291 | ||
|
|
0c217dfa2b | ||
|
|
b0198d2566 | ||
|
|
7eee8ca802 | ||
|
|
2a079a0154 | ||
|
|
40989c1178 | ||
|
|
addc831eb3 | ||
|
|
b4efc6d116 | ||
|
|
125d075d6e | ||
|
|
370a775479 | ||
|
|
1a2ee706eb | ||
|
|
2a981357ba | ||
|
|
1ae475e724 | ||
|
|
a3e9401fbc | ||
|
|
9091469f9e | ||
|
|
17fa54f1f9 | ||
|
|
8fb5347c2d | ||
|
|
6739bf998f | ||
|
|
6eea38ba67 | ||
|
|
e9cf88b359 | ||
|
|
645b203476 | ||
|
|
be2aff1f4c | ||
|
|
56ed237e82 | ||
|
|
fd7b0fd135 | ||
|
|
e700994891 | ||
|
|
c76f7029ac | ||
|
|
d535c5fb2a | ||
|
|
54f860463e | ||
|
|
950434b8ff | ||
|
|
ee365e876d | ||
|
|
c6c59834b9 | ||
|
|
63b47914b8 | ||
|
|
9e02e5be2e | ||
|
|
093cd70fa1 | ||
|
|
376d65a483 | ||
|
|
a0d9a2458e | ||
|
|
456f639cf7 | ||
|
|
2c559ec2f3 | ||
|
|
619c81f463 | ||
|
|
f1490df960 | ||
|
|
7bdf74de98 | ||
|
|
1743d6fb98 | ||
|
|
ca7a5bb926 | ||
|
|
ce8b1a3f1e | ||
|
|
486fa75a10 | ||
|
|
f8d68cd3d3 | ||
|
|
ef7a3f5606 | ||
|
|
4f84ed7490 | ||
|
|
d534103131 | ||
|
|
82dff3c2ce | ||
|
|
30d73eb5ba | ||
|
|
1b2754bac2 | ||
|
|
cf80cafc75 | ||
|
|
b8897d51de | ||
|
|
3ff25eeb65 | ||
|
|
2bbfc4e786 | ||
|
|
2b1eb052e6 | ||
|
|
360e214e54 | ||
|
|
2618afed94 | ||
|
|
698ba2c788 | ||
|
|
b614e99588 | ||
|
|
fe8e4af2fa | ||
|
|
0a897f1528 | ||
|
|
cf8a3f5779 | ||
|
|
db39a39868 | ||
|
|
37a03d28c2 | ||
|
|
19d275425a | ||
|
|
88e9045602 | ||
|
|
5adbc536b6 | ||
|
|
e27af94ba9 | ||
|
|
43fe1e7e9c | ||
|
|
f456a858c8 | ||
|
|
084c3aa88e | ||
|
|
34f9b63921 | ||
|
|
bd3de79817 | ||
|
|
304eee2259 | ||
|
|
9e729b7f59 | ||
|
|
dd141468c4 | ||
|
|
933147c21f | ||
|
|
9201a4f591 | ||
|
|
5adb1e9b8b | ||
|
|
4df84d7988 | ||
|
|
cd87c0968b | ||
|
|
8a8e7c90bf | ||
|
|
e806069065 | ||
|
|
ce948cbec0 | ||
|
|
6ed34b3294 | ||
|
|
7161a235ca | ||
|
|
71463810de | ||
|
|
e997219a85 | ||
|
|
895cc13fa6 | ||
|
|
8d3c3ca29a | ||
|
|
9829553807 | ||
|
|
e551f9731a | ||
|
|
fd827bf58b | ||
|
|
5a3baba34d | ||
|
|
c78f5b160f | ||
|
|
485f78761a | ||
|
|
23cd2f7b21 | ||
|
|
5753266c43 | ||
|
|
4722d2607d | ||
|
|
85b5b4f855 | ||
|
|
a16f492f0f | ||
|
|
3633dc632c | ||
|
|
b3b30c3a86 | ||
|
|
c78a7684f4 | ||
|
|
cf83d92630 | ||
|
|
a56b1274d8 | ||
|
|
ae4bdd0492 | ||
|
|
e90102dd3b | ||
|
|
71f0e8db3d | ||
|
|
638929373a | ||
|
|
8440654377 | ||
|
|
9fa66c4741 | ||
|
|
38a9235145 | ||
|
|
c7a3cc9108 | ||
|
|
248337908d | ||
|
|
3d003619fd | ||
|
|
f163dca12c | ||
|
|
6e0ce458e5 | ||
|
|
5fae8480f1 | ||
|
|
e6587d374a | ||
|
|
376cc404e0 | ||
|
|
9898ca638f | ||
|
|
34b46d8f7c | ||
|
|
fe7d0798a7 | ||
|
|
0cecc09d71 | ||
|
|
e091d55561 | ||
|
|
69cf18158b | ||
|
|
6513c53817 | ||
|
|
e13baa58a5 | ||
|
|
951056fe9b | ||
|
|
67700ea6bd | ||
|
|
e5442cf3f1 | ||
|
|
da68076f04 | ||
|
|
b24116a118 | ||
|
|
f67398c6bf | ||
|
|
43d3eb1a24 | ||
|
|
0993315ed5 | ||
|
|
0bc383ada9 | ||
|
|
1841ceca43 | ||
|
|
2714cebabd | ||
|
|
e184db4ce2 | ||
|
|
ac6dc6943c | ||
|
|
ddd53806df | ||
|
|
e629a1f70e | ||
|
|
68076d969c | ||
|
|
d3009d3e1c | ||
|
|
54f7f3c894 |
@@ -1094,8 +1094,8 @@
|
||||
# Default is 100.
|
||||
#
|
||||
# back_off_milliseconds
|
||||
# Number of milliseconds to wait between
|
||||
# online_delete batches to allow other functions
|
||||
# Number of milliseconds to wait between online_delete
|
||||
# SQL deletion batches to allow other functions
|
||||
# to catch up.
|
||||
# Default is 100.
|
||||
#
|
||||
@@ -1109,10 +1109,22 @@
|
||||
# The online delete process checks periodically
|
||||
# that xrpld is still in sync with the network,
|
||||
# and that the validated ledger is less than
|
||||
# 'age_threshold_seconds' old. If not, then continue
|
||||
# 'age_threshold_seconds' old, and that all
|
||||
# recent ledgers are available. If not, then continue
|
||||
# sleeping for this number of seconds and
|
||||
# checking until healthy.
|
||||
# Default is 5.
|
||||
# Default is 2.
|
||||
#
|
||||
# max_waiting_ledgers
|
||||
# The maximum number of ledgers that may be validated
|
||||
# while online deletion is waiting for the node to get
|
||||
# fully synced with the rest of the network. If more than
|
||||
# this number of ledgers are validated while waiting, then
|
||||
# online deletion gives up on the current ledger and tries
|
||||
# again later. Note this only affects situations that cause
|
||||
# rotation to wait, such as going out of sync, or missing
|
||||
# ledgers. Forward progress is not penalized. Minimum is 64.
|
||||
# Default is the online_delete value.
|
||||
#
|
||||
# Notes:
|
||||
# The 'node_db' entry configures the primary, persistent storage.
|
||||
|
||||
@@ -125,6 +125,7 @@ struct Keys
|
||||
static constexpr auto kMaximumTxnInLedger = "maximum_txn_in_ledger";
|
||||
static constexpr auto kMaximumTxnPerAccount = "maximum_txn_per_account";
|
||||
static constexpr auto kMemoryLevel = "memory_level";
|
||||
static constexpr auto kMaxWaitingLedgers = "max_waiting_ledgers";
|
||||
static constexpr auto kMinLedgersToComputeSizeLimit = "min_ledgers_to_compute_size_limit";
|
||||
static constexpr auto kMinimumEscalationMultiplier = "minimum_escalation_multiplier";
|
||||
static constexpr auto kMinimumLastLedgerBuffer = "minimum_last_ledger_buffer";
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include <xrpl/nodestore/Database.h>
|
||||
#include <xrpl/nodestore/Scheduler.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
@@ -54,6 +55,12 @@ public:
|
||||
*/
|
||||
virtual void
|
||||
setRotationInFlight(bool inFlight) = 0;
|
||||
virtual bool
|
||||
isRotationInFlight() const = 0;
|
||||
|
||||
[[nodiscard]]
|
||||
virtual std::uint64_t
|
||||
getAndResetDuplicationCount() = 0;
|
||||
};
|
||||
|
||||
} // namespace xrpl::node_store
|
||||
|
||||
@@ -72,19 +72,28 @@ public:
|
||||
|
||||
void
|
||||
setRotationInFlight(bool inFlight) override;
|
||||
bool
|
||||
isRotationInFlight() const override;
|
||||
|
||||
[[nodiscard]]
|
||||
std::uint64_t
|
||||
getAndResetDuplicationCount() override;
|
||||
|
||||
private:
|
||||
std::shared_ptr<Backend> writableBackend_;
|
||||
std::shared_ptr<Backend> archiveBackend_;
|
||||
mutable std::mutex mutex_;
|
||||
|
||||
// True between SHAMapStore starting the cache-freshen phase and the
|
||||
// completion of rotate(). While true, archive hits on ordinary
|
||||
// (duplicate == false) fetches are copied forward into the writable
|
||||
// backend; copyForwardCount_ tallies them per rotation for the
|
||||
// Set to true during the entire SHAMapStore rotation process.
|
||||
// While true, archive hits on ordinary (duplicate == false)
|
||||
// fetches are copied forward into the writable backend.
|
||||
// copyForwardCount_ tallies them per rotation for the
|
||||
// summary line logged at swap.
|
||||
std::atomic<bool> rotationInFlight_{false};
|
||||
std::atomic<std::uint64_t> copyForwardCount_{0};
|
||||
// Duplication count tracks the number of nodes that are directly duplicated because they're in
|
||||
// the target ledger or rescued from a cache.
|
||||
std::atomic<std::uint64_t> duplicationCount_{0};
|
||||
|
||||
std::shared_ptr<NodeObject>
|
||||
fetchNodeObject(uint256 const& hash, std::uint32_t, FetchReport& fetchReport, bool duplicate)
|
||||
|
||||
@@ -54,6 +54,7 @@ DatabaseRotatingImp::rotate(
|
||||
// deleted.
|
||||
std::shared_ptr<node_store::Backend> oldArchiveBackend;
|
||||
std::uint64_t copyForwards = 0;
|
||||
std::uint64_t duplications = 0;
|
||||
{
|
||||
std::scoped_lock const lock(mutex_);
|
||||
|
||||
@@ -66,6 +67,7 @@ DatabaseRotatingImp::rotate(
|
||||
writableBackend_ = std::move(newBackend);
|
||||
|
||||
copyForwards = copyForwardCount_.exchange(0, std::memory_order_relaxed);
|
||||
duplications = duplicationCount_.exchange(0, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
if (copyForwards > 0)
|
||||
@@ -74,6 +76,11 @@ DatabaseRotatingImp::rotate(
|
||||
<< " archive-served reads into the writable backend "
|
||||
"during the rotation window";
|
||||
}
|
||||
if (duplications > 0)
|
||||
{
|
||||
JLOG(j_.warn()) << "Rotating: duplicated " << duplications
|
||||
<< " nodes into the writable backend for the relevant cache.";
|
||||
}
|
||||
|
||||
f(newWritableBackendName, newArchiveBackendName);
|
||||
}
|
||||
@@ -86,6 +93,22 @@ DatabaseRotatingImp::setRotationInFlight(bool inFlight)
|
||||
<< (inFlight ? "enabled" : "disabled");
|
||||
}
|
||||
|
||||
bool
|
||||
DatabaseRotatingImp::isRotationInFlight() const
|
||||
{
|
||||
return rotationInFlight_.load(std::memory_order_acquire);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
std::uint64_t
|
||||
DatabaseRotatingImp::getAndResetDuplicationCount()
|
||||
{
|
||||
std::uint64_t duplications = 0;
|
||||
duplications = duplicationCount_.exchange(0, std::memory_order_relaxed);
|
||||
|
||||
return duplications;
|
||||
}
|
||||
|
||||
std::string
|
||||
DatabaseRotatingImp::getName() const
|
||||
{
|
||||
@@ -141,7 +164,7 @@ DatabaseRotatingImp::sweep()
|
||||
std::shared_ptr<NodeObject>
|
||||
DatabaseRotatingImp::fetchNodeObject(
|
||||
uint256 const& hash,
|
||||
std::uint32_t,
|
||||
std::uint32_t ledgerSeq,
|
||||
FetchReport& fetchReport,
|
||||
bool duplicate)
|
||||
{
|
||||
@@ -190,22 +213,29 @@ DatabaseRotatingImp::fetchNodeObject(
|
||||
nodeObject = fetch(archive);
|
||||
if (nodeObject)
|
||||
{
|
||||
{
|
||||
// Refresh the writable backend pointer
|
||||
std::scoped_lock const lock(mutex_);
|
||||
writable = writableBackend_;
|
||||
}
|
||||
|
||||
// Update writable backend with data from the archive backend.
|
||||
// While a rotation is in flight, ordinary (duplicate == false)
|
||||
// reads served by the archive are copied forward too: the
|
||||
// archive is about to be deleted, and a body canonicalized
|
||||
// into the cache after the freshen getKeys() snapshot would
|
||||
// otherwise survive only in RAM once the archive is dropped.
|
||||
if (duplicate || rotationInFlight_.load(std::memory_order_acquire))
|
||||
auto const inFlight = isRotationInFlight();
|
||||
if (duplicate || inFlight)
|
||||
{
|
||||
if (!duplicate)
|
||||
{
|
||||
// Refresh the writable backend pointer since we need to use it
|
||||
std::scoped_lock const lock(mutex_);
|
||||
writable = writableBackend_;
|
||||
}
|
||||
|
||||
if (duplicate)
|
||||
{
|
||||
duplicationCount_.fetch_add(1, std::memory_order_relaxed);
|
||||
}
|
||||
else
|
||||
{
|
||||
copyForwardCount_.fetch_add(1, std::memory_order_relaxed);
|
||||
}
|
||||
writable->store(nodeObject);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,21 @@
|
||||
#include <test/jtx/noop.h>
|
||||
|
||||
#include <xrpld/app/ledger/LedgerMaster.h>
|
||||
#include <xrpld/app/misc/SHAMapStore.h>
|
||||
#include <xrpld/core/Config.h>
|
||||
|
||||
#include <xrpl/basics/ToString.h>
|
||||
#include <xrpl/basics/base_uint.h>
|
||||
#include <xrpl/beast/unit_test/suite.h>
|
||||
#include <xrpl/protocol/Feature.h>
|
||||
#include <xrpl/protocol/Protocol.h>
|
||||
#include <xrpl/protocol/SField.h>
|
||||
#include <xrpl/protocol/STObject.h>
|
||||
#include <xrpl/protocol/STTx.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <sstream>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::test {
|
||||
@@ -111,6 +115,71 @@ class LedgerMaster_test : public beast::unit_test::Suite
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
testCompleteLedgerRange(FeatureBitset features)
|
||||
{
|
||||
// Note that this test is intentionally very similar to
|
||||
// SHAMapStore_test::testLedgerGaps, but has a different
|
||||
// focus.
|
||||
|
||||
testcase("Complete Ledger operations");
|
||||
|
||||
using namespace test::jtx;
|
||||
|
||||
auto const deleteInterval = 8;
|
||||
|
||||
Env env{*this, envconfig(onlineDelete, deleteInterval)};
|
||||
|
||||
auto const alice = Account("alice");
|
||||
env.fund(XRP(1000), alice);
|
||||
env.close();
|
||||
|
||||
auto& lm = env.app().getLedgerMaster();
|
||||
LedgerIndex minSeq = 2;
|
||||
LedgerIndex maxSeq = env.closed()->header().seq;
|
||||
auto& store = env.app().getSHAMapStore();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
LedgerIndex lastRotated = store.getLastRotated();
|
||||
BEAST_EXPECTS(maxSeq == 3, to_string(maxSeq));
|
||||
BEAST_EXPECTS(lm.getCompleteLedgers() == "2-3", lm.getCompleteLedgers());
|
||||
BEAST_EXPECTS(lastRotated == 3, to_string(lastRotated));
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq, maxSeq) == 0);
|
||||
BEAST_EXPECT(minSeq + 1 > maxSeq - 1);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 1, maxSeq + 1) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 2, maxSeq - 2) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq + 2, maxSeq + 2) == 2);
|
||||
|
||||
// Close enough ledgers to rotate a few times
|
||||
for (int i = 0; i < 24; ++i)
|
||||
{
|
||||
for (int t = 0; t < 3; ++t)
|
||||
{
|
||||
env(noop(alice));
|
||||
}
|
||||
env.close();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
++maxSeq;
|
||||
|
||||
if (maxSeq == lastRotated + deleteInterval)
|
||||
{
|
||||
minSeq = lastRotated;
|
||||
lastRotated = maxSeq;
|
||||
}
|
||||
BEAST_EXPECTS(
|
||||
env.closed()->header().seq == maxSeq, to_string(env.closed()->header().seq));
|
||||
BEAST_EXPECTS(store.getLastRotated() == lastRotated, to_string(store.getLastRotated()));
|
||||
std::stringstream expectedRange;
|
||||
expectedRange << minSeq << "-" << maxSeq;
|
||||
BEAST_EXPECTS(lm.getCompleteLedgers() == expectedRange.str(), lm.getCompleteLedgers());
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq, maxSeq) == 0);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq + 1, maxSeq - 1) == 0);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 1, maxSeq + 1) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 2, maxSeq - 2) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq + 2, maxSeq + 2) == 2);
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
void
|
||||
run() override
|
||||
@@ -124,6 +193,7 @@ public:
|
||||
testWithFeats(FeatureBitset features)
|
||||
{
|
||||
testTxnIdFromIndex(features);
|
||||
testCompleteLedgerRange(features);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
#include <test/jtx/Env.h>
|
||||
#include <test/jtx/amount.h>
|
||||
#include <test/jtx/envconfig.h>
|
||||
#include <test/jtx/noop.h>
|
||||
|
||||
#include <xrpld/app/ledger/LedgerMaster.h>
|
||||
#include <xrpld/app/main/Application.h>
|
||||
#include <xrpld/app/main/NodeStoreScheduler.h>
|
||||
#include <xrpld/app/misc/SHAMapStore.h>
|
||||
@@ -22,16 +24,21 @@
|
||||
#include <xrpl/protocol/Protocol.h>
|
||||
#include <xrpl/protocol/XRPAmount.h>
|
||||
#include <xrpl/protocol/jss.h>
|
||||
#include <xrpl/server/NetworkOPs.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <thread>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::test {
|
||||
|
||||
@@ -42,9 +49,8 @@ class SHAMapStore_test : public beast::unit_test::Suite
|
||||
static auto
|
||||
onlineDelete(std::unique_ptr<Config> cfg)
|
||||
{
|
||||
cfg->ledgerHistory = kDeleteInterval;
|
||||
auto& section = cfg->section(Sections::kNodeDatabase);
|
||||
section.set(Keys::kOnlineDelete, std::to_string(kDeleteInterval));
|
||||
cfg = jtx::onlineDelete(std::move(cfg), kDeleteInterval);
|
||||
cfg->section(Sections::kNodeDatabase).set(Keys::kRecoveryWaitSeconds, "1");
|
||||
return cfg;
|
||||
}
|
||||
|
||||
@@ -143,11 +149,11 @@ class SHAMapStore_test : public beast::unit_test::Suite
|
||||
auto& store = env.app().getSHAMapStore();
|
||||
|
||||
int ledgerSeq = 3;
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
BEAST_EXPECT(!store.getLastRotated());
|
||||
|
||||
env.close();
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
auto ledger = env.rpc("ledger", "validated");
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++)));
|
||||
@@ -227,7 +233,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(kDeleteInterval + 4)));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
BEAST_EXPECT(store.getLastRotated() == kDeleteInterval + 3);
|
||||
lastRotated = store.getLastRotated();
|
||||
@@ -254,7 +260,7 @@ public:
|
||||
!getHash(ledgers[i]).empty());
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
BEAST_EXPECT(store.getLastRotated() == kDeleteInterval + lastRotated);
|
||||
|
||||
@@ -292,7 +298,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
// The database will always have back to ledger 2,
|
||||
// regardless of lastRotated.
|
||||
@@ -307,7 +313,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - lastRotated, lastRotated);
|
||||
BEAST_EXPECT(lastRotated != store.getLastRotated());
|
||||
@@ -323,7 +329,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, kDeleteInterval + 1, lastRotated);
|
||||
BEAST_EXPECT(lastRotated != store.getLastRotated());
|
||||
@@ -362,7 +368,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - 2, 2);
|
||||
BEAST_EXPECT(lastRotated == store.getLastRotated());
|
||||
@@ -372,7 +378,7 @@ public:
|
||||
BEAST_EXPECT(!rpc::containsError(canDelete[jss::result]));
|
||||
BEAST_EXPECT(canDelete[jss::result][jss::can_delete] == ledgerSeq + (kDeleteInterval / 2));
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - 2, 2);
|
||||
BEAST_EXPECT(store.getLastRotated() == lastRotated);
|
||||
@@ -385,7 +391,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - lastRotated, lastRotated);
|
||||
|
||||
@@ -401,7 +407,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
BEAST_EXPECT(store.getLastRotated() == lastRotated);
|
||||
|
||||
@@ -413,7 +419,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - firstBatch, firstBatch);
|
||||
|
||||
@@ -435,7 +441,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
BEAST_EXPECT(store.getLastRotated() == lastRotated);
|
||||
|
||||
@@ -447,7 +453,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - lastRotated, lastRotated);
|
||||
|
||||
@@ -468,7 +474,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
BEAST_EXPECT(store.getLastRotated() == lastRotated);
|
||||
|
||||
@@ -480,7 +486,7 @@ public:
|
||||
BEAST_EXPECT(goodLedger(env, ledger, std::to_string(ledgerSeq++), true));
|
||||
}
|
||||
|
||||
store.rendezvous();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
ledgerCheck(env, ledgerSeq - lastRotated, lastRotated);
|
||||
|
||||
@@ -603,6 +609,302 @@ public:
|
||||
BEAST_EXPECT(dbr->getName() == "3");
|
||||
}
|
||||
|
||||
void
|
||||
testLedgerGaps()
|
||||
{
|
||||
// Note that this test is intentionally very similar to
|
||||
// LedgerMaster_test::testCompleteLedgerRange, but has a different
|
||||
// focus.
|
||||
|
||||
testcase("Wait for ledger gaps to fill in");
|
||||
|
||||
using namespace test::jtx;
|
||||
|
||||
Env env{*this, envconfig(onlineDelete)};
|
||||
|
||||
auto failureMessage = [&](char const* label, auto expected, auto actual) {
|
||||
std::stringstream ss;
|
||||
ss << label << ": Expected: " << expected << ", Got: " << actual;
|
||||
return ss.str();
|
||||
};
|
||||
|
||||
auto const alice = Account("alice");
|
||||
env.fund(XRP(1000), alice);
|
||||
env.close();
|
||||
|
||||
auto& lm = env.app().getLedgerMaster();
|
||||
LedgerIndex minSeq = 2;
|
||||
LedgerIndex maxSeq = env.closed()->header().seq;
|
||||
auto& store = env.app().getSHAMapStore();
|
||||
LedgerIndex lastRotated = store.getLastRotated();
|
||||
auto& netOPs = env.app().getOPs();
|
||||
while (lastRotated != 3)
|
||||
{
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
lastRotated = store.getLastRotated();
|
||||
}
|
||||
BEAST_EXPECTS(maxSeq == 3, std::to_string(maxSeq));
|
||||
BEAST_EXPECTS(lm.getCompleteLedgers() == "2-3", lm.getCompleteLedgers());
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq, maxSeq) == 0);
|
||||
BEAST_EXPECT(minSeq + 1 > maxSeq - 1);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 1, maxSeq + 1) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq - 2, maxSeq - 2) == 2);
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq + 2, maxSeq + 2) == 2);
|
||||
|
||||
auto expectedRange =
|
||||
[](LedgerIndex minSeq, std::vector<LedgerIndex> const& deleteSeqs, LedgerIndex maxSeq) {
|
||||
std::stringstream expectedRange;
|
||||
expectedRange << minSeq;
|
||||
auto lastDelete = minSeq - 1;
|
||||
for (auto deleteSeq : deleteSeqs)
|
||||
{
|
||||
if (deleteSeq <= lastDelete)
|
||||
continue;
|
||||
expectedRange << "-" << (deleteSeq - 1);
|
||||
if (deleteSeq + 1 <= maxSeq)
|
||||
expectedRange << "," << (deleteSeq + 1);
|
||||
lastDelete = deleteSeq;
|
||||
}
|
||||
if (lastDelete + 1 < maxSeq)
|
||||
{
|
||||
expectedRange << "-" << maxSeq;
|
||||
}
|
||||
return expectedRange.str();
|
||||
};
|
||||
|
||||
auto deleteLedgerSeq =
|
||||
[&lm, &store, &netOPs, &minSeq, &lastRotated, &expectedRange, &failureMessage, this](
|
||||
Env& env,
|
||||
LedgerIndex& maxSeq,
|
||||
std::vector<LedgerIndex>& deleteSeqs) -> LedgerIndex {
|
||||
using namespace std::chrono_literals;
|
||||
|
||||
// The next ledger will trigger a rotation. Delete the
|
||||
// current ledger from LedgerMaster.
|
||||
|
||||
netOPs.setMode(OperatingMode::CONNECTED);
|
||||
|
||||
LedgerIndex const deleteSeq = maxSeq;
|
||||
std::size_t iterations = 30;
|
||||
while (!lm.haveLedger(deleteSeq) && --iterations > 0)
|
||||
{
|
||||
std::this_thread::sleep_for(10ms);
|
||||
}
|
||||
// Even the slowest machines should be able to finalize deleteSeq within 10
|
||||
// loops (100ms). If this test ever actually fails feel free to lower this
|
||||
// cutoff. The intent of this test is to flag if the loop takes a very long
|
||||
// time, but still allow the rest of this function to finish.
|
||||
BEAST_EXPECTS(iterations > 20, std::to_string(iterations));
|
||||
if (!BEAST_EXPECT(lm.haveLedger(deleteSeq)))
|
||||
return 0;
|
||||
|
||||
// This test may be timing sensitive, because it's messing with server internals in ways
|
||||
// that they can't be messed with normally. Sleep a little bit to give the server time
|
||||
// to finish any internal work before we delete the ledger.
|
||||
std::this_thread::sleep_for(250ms);
|
||||
|
||||
lm.clearLedger(deleteSeq);
|
||||
deleteSeqs.push_back(deleteSeq);
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return 0;
|
||||
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
failureMessage(
|
||||
"Complete ledgers",
|
||||
expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
lm.getCompleteLedgers()));
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq, maxSeq) == deleteSeqs.size());
|
||||
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return 0;
|
||||
// Close another ledger, which will trigger a rotation, but the
|
||||
// rotation will be stuck until the missing ledger is filled in.
|
||||
env.close();
|
||||
// Do not call rendezvous() here without a timeout; it will block until the missing
|
||||
// ledger is backfilled. That will not happen automatically. It's a manual step that
|
||||
// is done later in this test.
|
||||
++maxSeq;
|
||||
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return 0;
|
||||
netOPs.setMode(OperatingMode::FULL);
|
||||
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return 0;
|
||||
BEAST_EXPECT(!store.rendezvous(10ms));
|
||||
BEAST_EXPECT(netOPs.getOperatingMode() == OperatingMode::FULL);
|
||||
|
||||
// Nothing has changed
|
||||
BEAST_EXPECTS(
|
||||
store.getLastRotated() == lastRotated,
|
||||
failureMessage("lastRotated", lastRotated, store.getLastRotated()));
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
failureMessage(
|
||||
"Complete ledgers",
|
||||
expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
lm.getCompleteLedgers()));
|
||||
|
||||
return deleteSeq;
|
||||
};
|
||||
|
||||
std::vector<LedgerIndex> deleteSeqs;
|
||||
|
||||
// Close enough ledgers to rotate a few times
|
||||
while (maxSeq < 40)
|
||||
{
|
||||
for (int t = 0; t < 3; ++t)
|
||||
{
|
||||
env(noop(alice));
|
||||
}
|
||||
env.close();
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
++maxSeq;
|
||||
|
||||
if (maxSeq + 1 == lastRotated + kDeleteInterval)
|
||||
{
|
||||
using namespace std::chrono_literals;
|
||||
|
||||
{
|
||||
// Trigger the circuit breaker in SHAMapStoreImp::healthWait() to ensure it
|
||||
// doesn't block forever.
|
||||
LedgerIndex const deleteSeq = deleteLedgerSeq(env, maxSeq, deleteSeqs);
|
||||
if (!BEAST_EXPECT(deleteSeq > 0))
|
||||
return;
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return;
|
||||
|
||||
// Close 7 more ledgers, waiting a little bit in between to
|
||||
// simulate the ledger making progress while online delete waits
|
||||
// for the missing ledger to be filled in.
|
||||
// After the 7th ledger, the circuit breaker will trigger and abort the attempt.
|
||||
while (maxSeq < lastRotated + (kDeleteInterval * 2) - 2)
|
||||
{
|
||||
env.close();
|
||||
++maxSeq;
|
||||
// Nothing has changed
|
||||
BEAST_EXPECTS(
|
||||
store.getLastRotated() == lastRotated,
|
||||
failureMessage("lastRotated", lastRotated, store.getLastRotated()));
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
failureMessage(
|
||||
"Complete Ledgers",
|
||||
expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
lm.getCompleteLedgers()));
|
||||
// The Store is "stuck" in healthWait() and won't finish the run() loop
|
||||
// until it's backfilled
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return;
|
||||
}
|
||||
|
||||
// Close one more ledger, which will NOT trigger the circuit breaker. Wait for
|
||||
// the full 1 second recovery wait timeout to ensure the circuit breaker is not
|
||||
// triggered.
|
||||
env.close();
|
||||
++maxSeq;
|
||||
// The Store is "stuck" in healthWait() and won't finish the run() loop
|
||||
// until it's backfilled
|
||||
BEAST_EXPECT(!store.rendezvous(1s));
|
||||
|
||||
// Close one more ledger, which will trigger the circuit breaker and abort the
|
||||
// attempt to rotate.
|
||||
env.close();
|
||||
++maxSeq;
|
||||
// Nothing has changed
|
||||
BEAST_EXPECTS(
|
||||
store.getLastRotated() == lastRotated,
|
||||
failureMessage("lastRotated", lastRotated, store.getLastRotated()));
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
failureMessage(
|
||||
"Complete Ledgers",
|
||||
expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
lm.getCompleteLedgers()));
|
||||
|
||||
// The circuit breaker has been triggered.
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
}
|
||||
{
|
||||
// Recover before the circuit breaker triggers, so the test can continue.
|
||||
LedgerIndex const deleteSeq = deleteLedgerSeq(env, maxSeq, deleteSeqs);
|
||||
if (!BEAST_EXPECT(deleteSeq > 0))
|
||||
return;
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return;
|
||||
|
||||
// Close 5 more ledgers, waiting a little bit in between to
|
||||
// simulate the ledger making progress while online delete waits
|
||||
// for the missing ledger to be filled in.
|
||||
// This ensures the healthWait check has time to run and
|
||||
// detect the gap.
|
||||
for (int l = 0; l < 5; ++l)
|
||||
{
|
||||
env.close();
|
||||
++maxSeq;
|
||||
// Nothing has changed
|
||||
BEAST_EXPECTS(
|
||||
store.getLastRotated() == lastRotated,
|
||||
failureMessage("lastRotated", lastRotated, store.getLastRotated()));
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
failureMessage(
|
||||
"Complete Ledgers",
|
||||
expectedRange(minSeq, deleteSeqs, maxSeq),
|
||||
lm.getCompleteLedgers()));
|
||||
if (!BEAST_EXPECT(!lm.haveLedger(deleteSeq)))
|
||||
return;
|
||||
}
|
||||
|
||||
// The Store is "stuck" in healthWait() and won't finish the run() loop
|
||||
// until it's backfilled
|
||||
// Wait for the full 1 second recovery wait timeout to ensure the circuit
|
||||
// breaker is not triggered, and this isn't some other timing fluke.
|
||||
BEAST_EXPECT(!store.rendezvous(1s));
|
||||
|
||||
// Put the missing ledger back in LedgerMaster
|
||||
lm.setLedgerRangePresent(deleteSeq, deleteSeq);
|
||||
BEAST_EXPECT(deleteSeqs.back() == deleteSeq);
|
||||
deleteSeqs.pop_back();
|
||||
|
||||
// Wait for the rotation to finish
|
||||
BEAST_EXPECT(store.rendezvous());
|
||||
|
||||
minSeq = lastRotated;
|
||||
while (deleteSeqs.front() < minSeq)
|
||||
{
|
||||
deleteSeqs.erase(deleteSeqs.begin());
|
||||
}
|
||||
lastRotated = deleteSeq + 1;
|
||||
}
|
||||
}
|
||||
BEAST_EXPECT(maxSeq != lastRotated + kDeleteInterval);
|
||||
BEAST_EXPECTS(
|
||||
env.closed()->header().seq == maxSeq,
|
||||
failureMessage("maxSeq", maxSeq, env.closed()->header().seq));
|
||||
BEAST_EXPECTS(
|
||||
store.getLastRotated() == lastRotated,
|
||||
failureMessage("lastRotated", lastRotated, store.getLastRotated()));
|
||||
{
|
||||
auto const expected = expectedRange(minSeq, deleteSeqs, maxSeq);
|
||||
BEAST_EXPECTS(
|
||||
lm.getCompleteLedgers() == expected,
|
||||
failureMessage("CompleteLedgers", expected, lm.getCompleteLedgers()));
|
||||
}
|
||||
BEAST_EXPECT(lm.missingFromCompleteLedgerRange(minSeq, maxSeq) == deleteSeqs.size());
|
||||
BEAST_EXPECT(
|
||||
lm.missingFromCompleteLedgerRange(minSeq + 1, maxSeq - 1) == deleteSeqs.size());
|
||||
BEAST_EXPECT(
|
||||
lm.missingFromCompleteLedgerRange(minSeq - 1, maxSeq + 1) == deleteSeqs.size() + 2);
|
||||
BEAST_EXPECT(
|
||||
lm.missingFromCompleteLedgerRange(minSeq - 2, maxSeq - 2) == deleteSeqs.size() + 2);
|
||||
BEAST_EXPECT(
|
||||
lm.missingFromCompleteLedgerRange(minSeq + 2, maxSeq + 2) == deleteSeqs.size() + 2);
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
run() override
|
||||
{
|
||||
@@ -610,6 +912,7 @@ public:
|
||||
testAutomatic();
|
||||
testCanDelete();
|
||||
testRotate();
|
||||
testLedgerGaps();
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include <xrpld/core/Config.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
@@ -62,6 +63,19 @@ envconfig(F&& modfunc, Args&&... args)
|
||||
return modfunc(envconfig(), std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief adjust config to enable online_delete
|
||||
*
|
||||
* @param cfg config instance to be modified
|
||||
*
|
||||
* @param deleteInterval how many new ledgers should be available before
|
||||
* rotating. Defaults to 8, because the standalone minimum is 8.
|
||||
*
|
||||
* @return unique_ptr to Config instance
|
||||
*/
|
||||
std::unique_ptr<Config>
|
||||
onlineDelete(std::unique_ptr<Config> cfg, std::uint32_t deleteInterval = 8);
|
||||
|
||||
/**
|
||||
* @brief adjust config so no admin ports are enabled
|
||||
*
|
||||
|
||||
@@ -7,8 +7,10 @@
|
||||
#include <xrpl/config/Constants.h>
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace xrpl::test {
|
||||
@@ -60,6 +62,15 @@ setupConfigForUnitTests(Config& cfg)
|
||||
|
||||
namespace jtx {
|
||||
|
||||
std::unique_ptr<Config>
|
||||
onlineDelete(std::unique_ptr<Config> cfg, std::uint32_t deleteInterval)
|
||||
{
|
||||
cfg->ledgerHistory = deleteInterval;
|
||||
auto& section = cfg->section(Sections::kNodeDatabase);
|
||||
section.set(Keys::kOnlineDelete, std::to_string(deleteInterval));
|
||||
return cfg;
|
||||
}
|
||||
|
||||
std::unique_ptr<Config>
|
||||
noAdmin(std::unique_ptr<Config> cfg)
|
||||
{
|
||||
|
||||
@@ -123,7 +123,10 @@ public:
|
||||
failedSave(std::uint32_t seq, uint256 const& hash);
|
||||
|
||||
std::string
|
||||
getCompleteLedgers();
|
||||
getCompleteLedgers() const;
|
||||
|
||||
std::size_t
|
||||
missingFromCompleteLedgerRange(LedgerIndex first, LedgerIndex last) const;
|
||||
|
||||
/**
|
||||
* Apply held transactions to the open ledger
|
||||
@@ -190,7 +193,7 @@ public:
|
||||
fixMismatch(ReadView const& ledger);
|
||||
|
||||
bool
|
||||
haveLedger(std::uint32_t seq);
|
||||
haveLedger(std::uint32_t seq) const;
|
||||
void
|
||||
clearLedger(std::uint32_t seq);
|
||||
bool
|
||||
@@ -348,7 +351,7 @@ private:
|
||||
// A set of transactions to replay during the next close
|
||||
std::unique_ptr<LedgerReplay> replayData_;
|
||||
|
||||
std::recursive_mutex completeLock_;
|
||||
std::recursive_mutex mutable completeLock_;
|
||||
RangeSet<std::uint32_t> completeLedgers_;
|
||||
|
||||
// Publish thread is running.
|
||||
|
||||
@@ -57,6 +57,7 @@
|
||||
#include <xrpl/shamap/SHAMapMissingNode.h>
|
||||
#include <xrpl/shamap/SHAMapTreeNode.h>
|
||||
|
||||
#include <boost/icl/concept/interval_associator.hpp>
|
||||
#include <boost/icl/concept/interval_set.hpp>
|
||||
|
||||
#include <xrpl.pb.h>
|
||||
@@ -492,7 +493,7 @@ LedgerMaster::setBuildingLedger(LedgerIndex i)
|
||||
}
|
||||
|
||||
bool
|
||||
LedgerMaster::haveLedger(std::uint32_t seq)
|
||||
LedgerMaster::haveLedger(std::uint32_t seq) const
|
||||
{
|
||||
std::scoped_lock const sl(completeLock_);
|
||||
return boost::icl::contains(completeLedgers_, seq);
|
||||
@@ -1576,12 +1577,36 @@ LedgerMaster::getPublishedLedger()
|
||||
}
|
||||
|
||||
std::string
|
||||
LedgerMaster::getCompleteLedgers()
|
||||
LedgerMaster::getCompleteLedgers() const
|
||||
{
|
||||
std::scoped_lock const sl(completeLock_);
|
||||
return to_string(completeLedgers_);
|
||||
}
|
||||
|
||||
std::size_t
|
||||
LedgerMaster::missingFromCompleteLedgerRange(LedgerIndex first, LedgerIndex last) const
|
||||
{
|
||||
if (first > last)
|
||||
{
|
||||
// In expected usage, this will never happen because "first" is generally initialized to
|
||||
// "last", "last" is guaranteed to grow monotonically, and "first" either doesn't change
|
||||
// or grows more slowly.
|
||||
// LCOV_EXCL_START
|
||||
UNREACHABLE("xrpl::LedgerMaster::missingFromCompleteLedgerRange : invalid parameters");
|
||||
return 0;
|
||||
// LCOV_EXCL_STOP
|
||||
}
|
||||
|
||||
RangeSet<LedgerIndex> const target{range(first, last)};
|
||||
|
||||
auto const missing = [&target, this] {
|
||||
std::scoped_lock const sl(completeLock_);
|
||||
return target - completeLedgers_;
|
||||
}();
|
||||
|
||||
return boost::icl::size(missing);
|
||||
}
|
||||
|
||||
std::optional<NetClock::time_point>
|
||||
LedgerMaster::getCloseTimeBySeq(LedgerIndex ledgerIndex)
|
||||
{
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <xrpl/nodestore/Scheduler.h>
|
||||
#include <xrpl/protocol/Protocol.h>
|
||||
|
||||
#include <chrono>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
@@ -34,8 +35,8 @@ public:
|
||||
virtual void
|
||||
start() = 0;
|
||||
|
||||
virtual void
|
||||
rendezvous() const = 0;
|
||||
[[nodiscard]] virtual bool
|
||||
rendezvous(std::optional<std::chrono::milliseconds> const& timeout = {}) const = 0;
|
||||
|
||||
virtual void
|
||||
stop() = 0;
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include <xrpl/basics/FileUtilities.h>
|
||||
#include <xrpl/basics/Log.h>
|
||||
#include <xrpl/basics/contract.h>
|
||||
#include <xrpl/basics/scope.h>
|
||||
#include <xrpl/beast/core/CurrentThreadName.h>
|
||||
#include <xrpl/beast/utility/Journal.h>
|
||||
#include <xrpl/beast/utility/instrumentation.h>
|
||||
@@ -30,6 +31,8 @@
|
||||
#include <boost/algorithm/string/predicate.hpp>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <filesystem>
|
||||
#include <functional>
|
||||
@@ -127,22 +130,6 @@ SHAMapStoreImp::SHAMapStoreImp(
|
||||
|
||||
if (deleteInterval_ != 0u)
|
||||
{
|
||||
// Configuration that affects the behavior of online delete
|
||||
getIfExists(section, Keys::kDeleteBatch, deleteBatch_);
|
||||
std::uint32_t temp = 0;
|
||||
if (getIfExists(section, Keys::kBackOffMilliseconds, temp) ||
|
||||
// Included for backward compatibility with an undocumented setting
|
||||
getIfExists(section, Keys::kBackOff, temp))
|
||||
{
|
||||
backOff_ = std::chrono::milliseconds{temp};
|
||||
}
|
||||
if (getIfExists(section, Keys::kAgeThresholdSeconds, temp))
|
||||
ageThreshold_ = std::chrono::seconds{temp};
|
||||
if (getIfExists(section, Keys::kRecoveryWaitSeconds, temp))
|
||||
recoveryWaitTime_ = std::chrono::seconds{temp};
|
||||
|
||||
getIfExists(section, Keys::kAdvisoryDelete, advisoryDelete_);
|
||||
|
||||
auto const minInterval =
|
||||
config.standalone() ? kMinimumDeletionIntervalSa : kMinimumDeletionInterval;
|
||||
if (deleteInterval_ < minInterval)
|
||||
@@ -159,6 +146,40 @@ SHAMapStoreImp::SHAMapStoreImp(
|
||||
std::to_string(config.ledgerHistory) + ")");
|
||||
}
|
||||
|
||||
// Configuration that affects the behavior of online delete
|
||||
getIfExists(section, Keys::kDeleteBatch, deleteBatch_);
|
||||
std::uint32_t temp = 0;
|
||||
if (getIfExists(section, Keys::kBackOffMilliseconds, temp) ||
|
||||
// Included for backward compatibility with an undocumented setting
|
||||
getIfExists(section, Keys::kBackOff, temp))
|
||||
{
|
||||
backOff_ = std::chrono::milliseconds{temp};
|
||||
}
|
||||
if (getIfExists(section, Keys::kAgeThresholdSeconds, temp))
|
||||
ageThreshold_ = std::chrono::seconds{temp};
|
||||
if (getIfExists(section, Keys::kRecoveryWaitSeconds, temp))
|
||||
recoveryWaitTime_ = std::chrono::seconds{temp};
|
||||
if (recoveryWaitTime_ < std::chrono::seconds{1})
|
||||
Throw<std::runtime_error>("recovery_wait_seconds must be at least 1 second");
|
||||
|
||||
getIfExists(section, Keys::kAdvisoryDelete, advisoryDelete_);
|
||||
|
||||
if (getIfExists(section, Keys::kMaxWaitingLedgers, temp))
|
||||
{
|
||||
maxWaitingLedgers_ = temp;
|
||||
}
|
||||
else
|
||||
{
|
||||
maxWaitingLedgers_ = deleteInterval_;
|
||||
}
|
||||
|
||||
auto const minWaiting = minInterval / 4;
|
||||
if (maxWaitingLedgers_ < minWaiting)
|
||||
{
|
||||
Throw<std::runtime_error>(
|
||||
"max_waiting_ledgers must be at least " + std::to_string(minWaiting));
|
||||
}
|
||||
|
||||
stateDb_.init(config, dbName_);
|
||||
dbPaths();
|
||||
}
|
||||
@@ -235,14 +256,22 @@ SHAMapStoreImp::onLedgerClosed(std::shared_ptr<Ledger const> const& ledger)
|
||||
cond_.notify_one();
|
||||
}
|
||||
|
||||
void
|
||||
SHAMapStoreImp::rendezvous() const
|
||||
[[nodiscard]]
|
||||
bool
|
||||
SHAMapStoreImp::rendezvous(std::optional<std::chrono::milliseconds> const& timeout) const
|
||||
{
|
||||
if (!working_)
|
||||
return;
|
||||
return true;
|
||||
|
||||
auto notWorking = [&] { return !working_; };
|
||||
|
||||
std::unique_lock<std::mutex> lock(mutex_);
|
||||
rendezvous_.wait(lock, [&] { return !working_; });
|
||||
if (timeout)
|
||||
{
|
||||
return rendezvous_.wait_for(lock, *timeout, notWorking);
|
||||
}
|
||||
rendezvous_.wait(lock, notWorking);
|
||||
return true;
|
||||
}
|
||||
|
||||
int
|
||||
@@ -251,31 +280,80 @@ SHAMapStoreImp::fdRequired() const
|
||||
return fdRequired_;
|
||||
}
|
||||
|
||||
void
|
||||
SHAMapStoreImp::rescueNode(SHAMapTreeNode const& node, std::optional<NodeObjectType> expectedType)
|
||||
{
|
||||
XRPL_ASSERT(node.cowid() == 0, "SHAMapStoreImp::rescueNode : rescued node must be clean");
|
||||
// Reachable from the validated state map in memory, but present in
|
||||
// neither backend: its only on-disk copy lived in a backend removed by
|
||||
// an earlier rotation, and it was never rewritten because it is clean
|
||||
// (cowid == 0, so flushDirty skips it). Persist the in-memory body
|
||||
// directly into the writable backend so it survives this rotation
|
||||
// instead of later surfacing as an unresolvable SHAMapMissingNode.
|
||||
|
||||
auto const nodeType = node.getType();
|
||||
auto const objectType = std::invoke([nodeType, expectedType] {
|
||||
switch (nodeType)
|
||||
{
|
||||
case SHAMapNodeType::TnAccountState:
|
||||
return NodeObjectType::AccountNode;
|
||||
// We don't expect to see transaction nodes. The check below will prevent writing them.
|
||||
case SHAMapNodeType::TnTransactionNm:
|
||||
case SHAMapNodeType::TnTransactionMd:
|
||||
return NodeObjectType::TransactionNode;
|
||||
case SHAMapNodeType::TnInner:
|
||||
return expectedType.value_or(NodeObjectType::Unknown);
|
||||
default:
|
||||
return NodeObjectType::Unknown;
|
||||
}
|
||||
});
|
||||
|
||||
auto const hash = node.getHash().asUInt256();
|
||||
XRPL_ASSERT_IF(
|
||||
expectedType,
|
||||
*expectedType == objectType,
|
||||
"SHAMapStoreImp::rescueNode : expected node type");
|
||||
|
||||
if (objectType != NodeObjectType::AccountNode || (expectedType && *expectedType != objectType))
|
||||
{
|
||||
// LCOV_EXCL_START
|
||||
JLOG(journal_.warn())
|
||||
<< "rescueNode: unable to re-store node with unsupported/unknown type, hash=" << hash
|
||||
<< " type=" << static_cast<int>(nodeType);
|
||||
// We do not expect to see Inner nodes rescued without an expected type. Analysis and
|
||||
// experimentation so far indicate that it just doesn't happen, specifically in
|
||||
// freshenCaches. This UNREACHABLE is as much a developer alert as it is a safety check. If
|
||||
// it does happen, we want to know about it. It won't affect production deployments.
|
||||
UNREACHABLE("SHAMapStoreImp::rescueNode : unsupported node type");
|
||||
return;
|
||||
// LCOV_EXCL_STOP
|
||||
}
|
||||
|
||||
Serializer s;
|
||||
node.serializeWithPrefix(s);
|
||||
dbRotating_->store(objectType, std::move(s.modData()), hash, 0);
|
||||
|
||||
JLOG(journal_.info()) << "rescueNode: re-stored node missing from both backends, hash=" << hash
|
||||
<< " type=" << static_cast<int>(nodeType);
|
||||
}
|
||||
|
||||
bool
|
||||
SHAMapStoreImp::copyNode(std::uint64_t& nodeCount, SHAMapTreeNode const& node)
|
||||
SHAMapStoreImp::copyNode(
|
||||
std::uint64_t& nodeCount,
|
||||
std::uint64_t& rescuedCount,
|
||||
SHAMapTreeNode const& node)
|
||||
{
|
||||
// Copy a single record from node to dbRotating_
|
||||
auto obj = dbRotating_->fetchNodeObject(
|
||||
node.getHash().asUInt256(), 0, node_store::FetchType::Synchronous, true);
|
||||
if (!obj)
|
||||
{
|
||||
XRPL_ASSERT(node.cowid() == 0, "SHAMapStoreImp::copyNode : rescued node must be clean");
|
||||
// Reachable from the validated state map in memory, but present in
|
||||
// neither backend: its only on-disk copy lived in a backend removed by
|
||||
// an earlier rotation, and it was never rewritten because it is clean
|
||||
// (cowid == 0, so flushDirty skips it). Persist the in-memory body
|
||||
// directly into the writable backend so it survives this rotation
|
||||
// instead of later surfacing as an unresolvable SHAMapMissingNode.
|
||||
auto const hash = node.getHash().asUInt256();
|
||||
Serializer s;
|
||||
node.serializeWithPrefix(s);
|
||||
dbRotating_->store(NodeObjectType::AccountNode, std::move(s.modData()), hash, 0);
|
||||
JLOG(journal_.warn()) << "copyNode: re-stored node missing from both backends, hash="
|
||||
<< hash << " type=" << static_cast<int>(node.getType());
|
||||
rescueNode(node, NodeObjectType::AccountNode);
|
||||
++rescuedCount;
|
||||
}
|
||||
if ((++nodeCount % checkHealthInterval_) == 0u)
|
||||
{
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -297,6 +375,11 @@ SHAMapStoreImp::run()
|
||||
|
||||
while (true)
|
||||
{
|
||||
XRPL_ASSERT(
|
||||
!dbRotating_->isRotationInFlight(),
|
||||
"SHAMapStoreImp::run : rotationInFlight_ must be false "
|
||||
"outside rotation window");
|
||||
|
||||
healthy_ = true;
|
||||
std::shared_ptr<Ledger const> validatedLedger;
|
||||
|
||||
@@ -326,30 +409,76 @@ SHAMapStoreImp::run()
|
||||
stateDb_.setLastRotated(lastRotated);
|
||||
}
|
||||
|
||||
// We're starting a new cycle, so reset back to the default.
|
||||
lastSuccessfulHealthCheck_ = 0;
|
||||
|
||||
bool const readyToRotate = validatedSeq >= lastRotated + deleteInterval_ &&
|
||||
canDelete_ >= lastRotated - 1 && healthWait() == HealthResult::KeepGoing;
|
||||
|
||||
{
|
||||
// Note that this is set after the healthWait() check, so that we
|
||||
// don't start the rotation until the validated ledger is fully
|
||||
// processed. It is not guaranteed to be done at this point. It also
|
||||
// allows the testLedgerGaps unit test to work.
|
||||
std::unique_lock<std::mutex> lock(mutex_);
|
||||
if (newLedger_)
|
||||
{
|
||||
// It is possible, though very unlikely outside of tests which manipulate internals,
|
||||
// that healthWait() took so long that the validated ledger (newLedger_) has moved
|
||||
// on from where we started. If that's the case, update lastGoodValidatedLedger_
|
||||
// to that ledger's sequence number.
|
||||
lastGoodValidatedLedger_ = newLedger_->header().seq;
|
||||
}
|
||||
else
|
||||
{
|
||||
lastGoodValidatedLedger_ = validatedSeq;
|
||||
}
|
||||
auto const l = lastGoodValidatedLedger_;
|
||||
lock.unlock();
|
||||
JLOG(journal_.trace()) << "run: Set lastGoodValidatedLedger_ to " << l;
|
||||
}
|
||||
|
||||
// will delete up to (not including) lastRotated
|
||||
if (readyToRotate)
|
||||
{
|
||||
JLOG(journal_.warn()) << "rotating validatedSeq " << validatedSeq << " lastRotated "
|
||||
<< lastRotated << " deleteInterval " << deleteInterval_
|
||||
<< " canDelete_ " << canDelete_ << " state "
|
||||
auto const diff = validatedSeq - lastRotated;
|
||||
JLOG(journal_.warn()) << "ROTATING: validatedSeq " << validatedSeq << " lastRotated "
|
||||
<< lastRotated << " diff " << diff << " deleteInterval "
|
||||
<< deleteInterval_ << " canDelete_ " << canDelete_ << " state "
|
||||
<< app_.getOPs().strOperatingMode(false) << " age "
|
||||
<< ledgerMaster_->getValidatedLedgerAge().count() << 's';
|
||||
<< ledgerMaster_->getValidatedLedgerAge().count()
|
||||
<< "s. Complete ledgers: " << ledgerMaster_->getCompleteLedgers();
|
||||
|
||||
// Close the getKeys()->swap exposure window: from here until
|
||||
// rotate() completes, an ordinary read for new ledgers served by the archive is
|
||||
// copied forward into the writable backend, so a node fetched
|
||||
// from the doomed archive cannot be left RAM-only when the
|
||||
// archive is deleted. Use ScopeExit so the early returns and continues below (and any
|
||||
// exceptions) also clear the flag.
|
||||
ScopeExit const clearRotationInFlight{
|
||||
[this] { dbRotating_->setRotationInFlight(false); }};
|
||||
dbRotating_->setRotationInFlight(true);
|
||||
|
||||
clearPrior(lastRotated);
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
switch (healthWait())
|
||||
{
|
||||
case HealthResult::Stopping:
|
||||
return;
|
||||
case HealthResult::Expired:
|
||||
continue;
|
||||
case HealthResult::KeepGoing:
|
||||
break;
|
||||
}
|
||||
|
||||
JLOG(journal_.debug()) << "copying ledger " << validatedSeq;
|
||||
std::uint64_t nodeCount = 0;
|
||||
std::uint64_t rescuedCount = 0;
|
||||
|
||||
try
|
||||
{
|
||||
validatedLedger->stateMap().snapShot(false)->visitNodes(
|
||||
[this, &nodeCount](SHAMapTreeNode const& node) {
|
||||
return copyNode(nodeCount, node);
|
||||
[this, &nodeCount, &rescuedCount](SHAMapTreeNode const& node) {
|
||||
return copyNode(nodeCount, rescuedCount, node);
|
||||
});
|
||||
}
|
||||
catch (SHAMapMissingNode const& e)
|
||||
@@ -359,43 +488,53 @@ SHAMapStoreImp::run()
|
||||
continue;
|
||||
}
|
||||
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
// Only log if we completed without a "health" abort
|
||||
JLOG(journal_.debug())
|
||||
<< "copied ledger " << validatedSeq << " nodecount " << nodeCount;
|
||||
|
||||
// Close the getKeys()->swap exposure window: from here until
|
||||
// rotate() completes, an ordinary read served by the archive is
|
||||
// copied forward into the writable backend, so a node fetched
|
||||
// from the doomed archive cannot be left RAM-only when the
|
||||
// archive is deleted. RAII so the early returns below (and any
|
||||
// exception) also clear the flag.
|
||||
struct RotationExposureGuard
|
||||
switch (healthWait())
|
||||
{
|
||||
node_store::DatabaseRotating& db;
|
||||
~RotationExposureGuard()
|
||||
{
|
||||
db.setRotationInFlight(false);
|
||||
}
|
||||
};
|
||||
RotationExposureGuard const rotationExposureGuard{*dbRotating_};
|
||||
dbRotating_->setRotationInFlight(true);
|
||||
case HealthResult::Stopping:
|
||||
return;
|
||||
case HealthResult::Expired:
|
||||
continue;
|
||||
case HealthResult::KeepGoing:
|
||||
break;
|
||||
}
|
||||
{
|
||||
// Only log if we completed without a "health" abort
|
||||
auto const copyDuplications = dbRotating_->getAndResetDuplicationCount();
|
||||
JLOG(journal_.debug())
|
||||
<< "copied ledger " << validatedSeq << " duplicated " << copyDuplications
|
||||
<< " / " << nodeCount << " nodes. Rescued " << rescuedCount << " nodes";
|
||||
}
|
||||
|
||||
JLOG(journal_.debug()) << "freshening caches";
|
||||
freshenCaches();
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
rescuedCount = 0;
|
||||
freshenCaches(rescuedCount);
|
||||
switch (healthWait())
|
||||
{
|
||||
case HealthResult::Stopping:
|
||||
return;
|
||||
case HealthResult::Expired:
|
||||
continue;
|
||||
case HealthResult::KeepGoing:
|
||||
break;
|
||||
}
|
||||
// Only log if we completed without a "health" abort
|
||||
JLOG(journal_.debug()) << validatedSeq << " freshened caches";
|
||||
JLOG(journal_.debug())
|
||||
<< validatedSeq << " freshened caches. Rescued " << rescuedCount << " nodes.";
|
||||
|
||||
JLOG(journal_.trace()) << "Making a new backend";
|
||||
auto newBackend = makeBackendRotating();
|
||||
JLOG(journal_.debug()) << validatedSeq << " new backend " << newBackend->getName();
|
||||
|
||||
clearCaches(validatedSeq);
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
switch (healthWait())
|
||||
{
|
||||
case HealthResult::Stopping:
|
||||
return;
|
||||
case HealthResult::Expired:
|
||||
continue;
|
||||
case HealthResult::KeepGoing:
|
||||
break;
|
||||
}
|
||||
|
||||
lastRotated = validatedSeq;
|
||||
|
||||
@@ -411,7 +550,14 @@ SHAMapStoreImp::run()
|
||||
clearCaches(validatedSeq);
|
||||
});
|
||||
|
||||
JLOG(journal_.warn()) << "finished rotation " << validatedSeq;
|
||||
auto const currentValidatedSeq = ledgerMaster_->getValidLedgerIndex();
|
||||
auto const processingDiff = currentValidatedSeq - validatedSeq;
|
||||
JLOG(journal_.warn())
|
||||
<< "FINISHED ROTATION: validatedSeq: " << validatedSeq
|
||||
<< ", lastRotated: " << lastRotated << " diff " << diff
|
||||
<< ". Updated validated seq is " << currentValidatedSeq << ", " << processingDiff
|
||||
<< " ledgers were validated during the rotation processs. Complete ledgers: "
|
||||
<< ledgerMaster_->getCompleteLedgers();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -559,7 +705,7 @@ SHAMapStoreImp::clearSql(
|
||||
min = *m;
|
||||
}
|
||||
|
||||
if (min > lastRotated || healthWait() == HealthResult::Stopping)
|
||||
if (min > lastRotated || healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
if (min == lastRotated)
|
||||
{
|
||||
@@ -572,18 +718,19 @@ SHAMapStoreImp::clearSql(
|
||||
<< lastRotated;
|
||||
while (min < lastRotated)
|
||||
{
|
||||
// The very first sleep is, arguably wasted, but clearSql is called multiple times for
|
||||
// different tables, so the time is amortized among all the operations. This results in
|
||||
// a backoff in between each set of tables, too.
|
||||
std::this_thread::sleep_for(backOff_);
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
|
||||
min = std::min(lastRotated, min + deleteBatch_);
|
||||
JLOG(journal_.trace()) << "Begin: Delete up to " << deleteBatch_
|
||||
<< " rows with LedgerSeq < " << min << " from: " << tableName;
|
||||
deleteBeforeSeq(min);
|
||||
JLOG(journal_.trace()) << "End: Delete up to " << deleteBatch_ << " rows with LedgerSeq < "
|
||||
<< min << " from: " << tableName;
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
if (min < lastRotated)
|
||||
std::this_thread::sleep_for(backOff_);
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
return;
|
||||
}
|
||||
JLOG(journal_.debug()) << "finished deleting from: " << tableName;
|
||||
}
|
||||
@@ -599,12 +746,11 @@ SHAMapStoreImp::clearCaches(LedgerIndex validatedSeq)
|
||||
}
|
||||
|
||||
void
|
||||
SHAMapStoreImp::freshenCaches()
|
||||
SHAMapStoreImp::freshenCaches(std::uint64_t& rescuedCount)
|
||||
{
|
||||
if (freshenCache(*treeNodeCache_))
|
||||
return;
|
||||
if (freshenCache(app_.getMasterTransaction().getCache()))
|
||||
if (freshenCache(*treeNodeCache_, rescuedCount))
|
||||
return;
|
||||
freshenCache(app_.getMasterTransaction().getCache(), rescuedCount);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -616,7 +762,7 @@ SHAMapStoreImp::clearPrior(LedgerIndex lastRotated)
|
||||
JLOG(journal_.trace()) << "Begin: Clear internal ledgers up to " << lastRotated;
|
||||
ledgerMaster_->clearPriorLedgers(lastRotated);
|
||||
JLOG(journal_.trace()) << "End: Clear internal ledgers up to " << lastRotated;
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
|
||||
auto& db = app_.getRelationalDatabase();
|
||||
@@ -626,7 +772,7 @@ SHAMapStoreImp::clearPrior(LedgerIndex lastRotated)
|
||||
"Ledgers",
|
||||
[&db]() -> std::optional<LedgerIndex> { return db.getMinLedgerSeq(); },
|
||||
[&db](LedgerIndex min) -> void { db.deleteBeforeLedgerSeq(min); });
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
|
||||
if (!app_.config().useTxTables())
|
||||
@@ -637,7 +783,7 @@ SHAMapStoreImp::clearPrior(LedgerIndex lastRotated)
|
||||
"Transactions",
|
||||
[&db]() -> std::optional<LedgerIndex> { return db.getTransactionsMinLedgerSeq(); },
|
||||
[&db](LedgerIndex min) -> void { db.deleteTransactionsBeforeLedgerSeq(min); });
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
|
||||
clearSql(
|
||||
@@ -645,30 +791,135 @@ SHAMapStoreImp::clearPrior(LedgerIndex lastRotated)
|
||||
"AccountTransactions",
|
||||
[&db]() -> std::optional<LedgerIndex> { return db.getAccountTransactionsMinLedgerSeq(); },
|
||||
[&db](LedgerIndex min) -> void { db.deleteAccountTransactionsBeforeLedgerSeq(min); });
|
||||
if (healthWait() == HealthResult::Stopping)
|
||||
if (healthWait() != HealthResult::KeepGoing)
|
||||
return;
|
||||
}
|
||||
|
||||
SHAMapStoreImp::HealthResult
|
||||
SHAMapStoreImp::healthWait()
|
||||
{
|
||||
auto age = ledgerMaster_->getValidatedLedgerAge();
|
||||
OperatingMode mode = netOPs_->getOperatingMode();
|
||||
std::unique_lock lock(mutex_);
|
||||
while (!stop_ && (mode != OperatingMode::FULL || age > ageThreshold_))
|
||||
{
|
||||
lock.unlock();
|
||||
JLOG(journal_.warn()) << "Waiting " << recoveryWaitTime_.count()
|
||||
<< "s for node to stabilize. state: "
|
||||
<< app_.getOPs().strOperatingMode(mode, false) << ". age "
|
||||
<< age.count() << 's';
|
||||
std::this_thread::sleep_for(recoveryWaitTime_);
|
||||
// Gets the current status of the server from ledgerMaster_ and netOPs_. Must be called
|
||||
// while mutex_ is unlocked to avoid unlikely, but possible, deadlock with ledgerMaster_'s
|
||||
// completeLock_.
|
||||
// Releasing the lock may mean that status will be slightly out of date when the lock is
|
||||
// reacquired, but it's close enough. In a normal rotation, healthWait() is called frequently,
|
||||
// so a false positive will be detected on the next call, and a false negative will be detected
|
||||
// in the next loop iteration. Database rotation is important, but not timely, so an extra
|
||||
// delay is fine.
|
||||
auto readServerStatus = [this](
|
||||
LedgerIndex& index,
|
||||
bool& buildingIndex,
|
||||
std::chrono::seconds& age,
|
||||
OperatingMode& mode,
|
||||
std::size_t& numMissing,
|
||||
LedgerIndex const lowerBound,
|
||||
ScopeUnlock<decltype(mutex_)> const&) {
|
||||
index = ledgerMaster_->getValidLedgerIndex();
|
||||
bool const haveIndex = ledgerMaster_->haveLedger(index);
|
||||
age = ledgerMaster_->getValidatedLedgerAge();
|
||||
mode = netOPs_->getOperatingMode();
|
||||
lock.lock();
|
||||
|
||||
numMissing =
|
||||
lowerBound == 0 ? 0 : ledgerMaster_->missingFromCompleteLedgerRange(lowerBound, index);
|
||||
|
||||
buildingIndex = (numMissing == 1 && !haveIndex);
|
||||
};
|
||||
// Tracked server status properties
|
||||
LedgerIndex index = 0;
|
||||
bool buildingIndex = false;
|
||||
std::chrono::seconds age;
|
||||
OperatingMode mode = OperatingMode::DISCONNECTED;
|
||||
std::size_t numMissing = 0;
|
||||
|
||||
std::unique_lock lock(mutex_);
|
||||
|
||||
auto const waitTime = recoveryWaitTime_;
|
||||
auto const ageThreshold = ageThreshold_;
|
||||
{
|
||||
auto const lowerBound = lastGoodValidatedLedger_;
|
||||
|
||||
ScopeUnlock const unlock(lock);
|
||||
|
||||
readServerStatus(index, buildingIndex, age, mode, numMissing, lowerBound, unlock);
|
||||
}
|
||||
// If index gets past this point without the health check succeeding, return
|
||||
// HealthWait::Expired. This depends on index being initialized, so it must be after
|
||||
// readServerStatus().
|
||||
auto const lastSuccess = lastSuccessfulHealthCheck_ == 0 ? index : lastSuccessfulHealthCheck_;
|
||||
auto const circuitBreaker = lastSuccess + maxWaitingLedgers_;
|
||||
|
||||
auto healthy = [&] {
|
||||
// Special case: If the server is disconnected, it's not doing any ledger I/O, because
|
||||
// it's focused on trying to get peers. A disconnected state is should never be caused by
|
||||
// the activity of the server. It's usually limited to hardware or connectivity issues. Take
|
||||
// advantage of that to run as much rotation I/O as possible before it comes back online.
|
||||
if (mode == OperatingMode::DISCONNECTED)
|
||||
return true;
|
||||
if (age > ageThreshold)
|
||||
return false;
|
||||
if (numMissing > 0)
|
||||
return false;
|
||||
if (mode != OperatingMode::FULL)
|
||||
return false;
|
||||
return true;
|
||||
};
|
||||
|
||||
while (!stop_ && !healthy() && index < circuitBreaker)
|
||||
{
|
||||
// Future-proofing: this value shouldn't change while we are sleeping, but grab it while we
|
||||
// have the lock in case it does.
|
||||
auto const lowerBound = lastGoodValidatedLedger_;
|
||||
|
||||
ScopeUnlock const unlock(lock);
|
||||
|
||||
auto const [stream, waitMs] = std::invoke(
|
||||
[mode, age, ageThreshold, buildingIndex, waitTime, index, lastSuccess, this]
|
||||
-> std::pair<beast::Journal::Stream, std::chrono::milliseconds> {
|
||||
if (mode != OperatingMode::FULL || age > ageThreshold ||
|
||||
(index - lastSuccess > maxWaitingLedgers_ / 4))
|
||||
return {journal_.warn(), waitTime};
|
||||
if (buildingIndex)
|
||||
{
|
||||
// We expect this ledger to be built soon, so log at a lower level, and don't
|
||||
// wait as long.
|
||||
return {
|
||||
journal_.trace(),
|
||||
std::chrono::duration_cast<std::chrono::milliseconds>(waitTime) / 10};
|
||||
}
|
||||
return {journal_.info(), waitTime};
|
||||
});
|
||||
JLOG(stream) << "Waiting " << waitMs.count() << "ms for node to stabilize. state: "
|
||||
<< app_.getOPs().strOperatingMode(mode, false) << ". age " << age.count()
|
||||
<< "s. Missing ledgers: " << numMissing << ". Expect: " << lowerBound << "-"
|
||||
<< index << ". Complete ledgers: " << ledgerMaster_->getCompleteLedgers();
|
||||
std::this_thread::sleep_for(waitMs);
|
||||
|
||||
[[maybe_unused]]
|
||||
LedgerIndex const lastLedger = index;
|
||||
readServerStatus(index, buildingIndex, age, mode, numMissing, lowerBound, unlock);
|
||||
SOMETIMES(
|
||||
index > lastLedger, "SHAMapStoreImp::healthWait : validated ledger index changed");
|
||||
}
|
||||
|
||||
return stop_ ? HealthResult::Stopping : HealthResult::KeepGoing;
|
||||
auto const result = std::invoke([index, circuitBreaker, this]() -> HealthResult {
|
||||
if (stop_)
|
||||
return HealthResult::Stopping;
|
||||
if (index < circuitBreaker)
|
||||
return HealthResult::KeepGoing;
|
||||
JLOG(journal_.error()) << "online_delete rotation has been unable to make progress for "
|
||||
<< maxWaitingLedgers_ << " ledgers. "
|
||||
<< "validated ledger index: " << index
|
||||
<< ", last successful health check index: "
|
||||
<< lastSuccessfulHealthCheck_
|
||||
<< ", circuit breaker index: " << circuitBreaker;
|
||||
return HealthResult::Expired;
|
||||
});
|
||||
|
||||
XRPL_ASSERT(lock.owns_lock(), "SHAMapStoreImp::healthWait : lock held");
|
||||
if (result == HealthResult::KeepGoing)
|
||||
lastSuccessfulHealthCheck_ = index;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
void
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#include <xrpl/nodestore/Backend.h>
|
||||
#include <xrpl/nodestore/Database.h>
|
||||
#include <xrpl/nodestore/DatabaseRotating.h>
|
||||
#include <xrpl/nodestore/NodeObject.h>
|
||||
#include <xrpl/nodestore/Scheduler.h>
|
||||
#include <xrpl/protocol/Protocol.h>
|
||||
#include <xrpl/rdb/DatabaseCon.h>
|
||||
@@ -21,6 +22,7 @@
|
||||
#include <algorithm>
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <concepts>
|
||||
#include <condition_variable>
|
||||
#include <cstdint>
|
||||
#include <functional>
|
||||
@@ -88,6 +90,13 @@ private:
|
||||
std::thread thread_;
|
||||
bool stop_ = false;
|
||||
bool healthy_ = true;
|
||||
// Used to prevent ledger gaps from forming during online deletion. Keeps
|
||||
// track of the last validated ledger that was processed without gaps. There
|
||||
// are no guarantees about gaps while online delete is not running. For
|
||||
// that, use advisory_delete and check for gaps externally.
|
||||
LedgerIndex lastGoodValidatedLedger_ = 0;
|
||||
// Used to prevent the circuit breaker from tripping too quickly.
|
||||
LedgerIndex lastSuccessfulHealthCheck_ = 0;
|
||||
mutable std::condition_variable cond_;
|
||||
mutable std::condition_variable rendezvous_;
|
||||
mutable std::mutex mutex_;
|
||||
@@ -102,12 +111,18 @@ private:
|
||||
std::chrono::milliseconds backOff_{100};
|
||||
std::chrono::seconds ageThreshold_{60};
|
||||
/**
|
||||
* If the node is out of sync during an online_delete healthWait()
|
||||
* call, sleep the thread for this time, and continue checking until
|
||||
* recovery.
|
||||
* If the node is out of sync, or any recent ledgers are not
|
||||
* available during an online_delete healthWait() call, sleep
|
||||
* the thread for this time, and continue checking until recovery.
|
||||
* See also: "recovery_wait_seconds" in xrpld-example.cfg
|
||||
*/
|
||||
std::chrono::seconds recoveryWaitTime_{5};
|
||||
std::chrono::seconds recoveryWaitTime_{2};
|
||||
/**
|
||||
* If the rotation stays "unhealthy" for a very long time, the process is aborted, and tried
|
||||
* again later. This value represents the number of ledgers that must be validated without
|
||||
* making rotation progress before the process is aborted.
|
||||
*/
|
||||
std::uint32_t maxWaitingLedgers_ = deleteBatch_;
|
||||
|
||||
// these do not exist upon SHAMapStore creation, but do exist
|
||||
// as of run() or before
|
||||
@@ -163,8 +178,9 @@ public:
|
||||
void
|
||||
onLedgerClosed(std::shared_ptr<Ledger const> const& ledger) override;
|
||||
|
||||
void
|
||||
rendezvous() const override;
|
||||
[[nodiscard]]
|
||||
bool
|
||||
rendezvous(std::optional<std::chrono::milliseconds> const& timeout = {}) const override;
|
||||
int
|
||||
fdRequired() const override;
|
||||
|
||||
@@ -172,9 +188,14 @@ public:
|
||||
minimumOnline() const override;
|
||||
|
||||
private:
|
||||
// Force write a node to the writable backend during rotation so it doesn't get lost
|
||||
void
|
||||
rescueNode(
|
||||
SHAMapTreeNode const& node,
|
||||
std::optional<NodeObjectType> expectedType = std::nullopt);
|
||||
// callback for visitNodes
|
||||
bool
|
||||
copyNode(std::uint64_t& nodeCount, SHAMapTreeNode const& node);
|
||||
copyNode(std::uint64_t& nodeCount, std::uint64_t& rescuedCount, SHAMapTreeNode const& node);
|
||||
void
|
||||
run();
|
||||
void
|
||||
@@ -185,14 +206,28 @@ private:
|
||||
|
||||
template <class CacheInstance>
|
||||
bool
|
||||
freshenCache(CacheInstance& cache)
|
||||
freshenCache(CacheInstance& cache, std::uint64_t& rescuedCount)
|
||||
{
|
||||
std::uint64_t check = 0;
|
||||
|
||||
for (auto const& key : cache.getKeys())
|
||||
{
|
||||
dbRotating_->fetchNodeObject(key, 0, node_store::FetchType::Synchronous, true);
|
||||
if (!(++check % checkHealthInterval_) && healthWait() == HealthResult::Stopping)
|
||||
[[maybe_unused]]
|
||||
auto const obj =
|
||||
dbRotating_->fetchNodeObject(key, 0, node_store::FetchType::Synchronous, true);
|
||||
if constexpr (std::derived_from<typename CacheInstance::mapped_type, SHAMapTreeNode>)
|
||||
{
|
||||
if (!obj)
|
||||
{
|
||||
auto const node = cache.fetch(key);
|
||||
if (node)
|
||||
{
|
||||
rescueNode(*node);
|
||||
++rescuedCount;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!(++check % checkHealthInterval_) && healthWait() != HealthResult::KeepGoing)
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -213,18 +248,18 @@ private:
|
||||
void
|
||||
clearCaches(LedgerIndex validatedSeq);
|
||||
void
|
||||
freshenCaches();
|
||||
freshenCaches(std::uint64_t& rescuedCount);
|
||||
void
|
||||
clearPrior(LedgerIndex lastRotated);
|
||||
|
||||
/**
|
||||
* This is a health check for online deletion that waits until xrpld is
|
||||
* stable before returning. It returns an indication of whether the server
|
||||
* is stopping.
|
||||
* is stopping, or if this attempt should be abandoned.
|
||||
*
|
||||
* @return Whether the server is stopping.
|
||||
*/
|
||||
enum class HealthResult { Stopping, KeepGoing };
|
||||
enum class HealthResult { Stopping, Expired, KeepGoing };
|
||||
[[nodiscard]] HealthResult
|
||||
healthWait();
|
||||
|
||||
|
||||
@@ -267,9 +267,9 @@ saveValidatedLedger(
|
||||
app.getAcceptedLedgerCache().canonicalizeReplaceClient(ledger->header().hash, aLedger);
|
||||
}
|
||||
}
|
||||
catch (std::exception const&)
|
||||
catch (std::exception const& e)
|
||||
{
|
||||
JLOG(j.warn()) << "An accepted ledger was missing nodes";
|
||||
JLOG(j.warn()) << "An accepted ledger was missing nodes " << e.what();
|
||||
app.getLedgerMaster().failedSave(seq, ledger->header().hash);
|
||||
// Clients can now trust the database for information about this
|
||||
// ledger sequence.
|
||||
|
||||
Reference in New Issue
Block a user