This is an automated email from the ASF dual-hosted git repository.
smengcl pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/ozone.git
The following commit(s) were added to refs/heads/master by this push:
new 5e99e226bb0 HDDS-16081. TestOMRatisSnapshots fails due to leader
applied index taken before flush (#10959)
5e99e226bb0 is described below
commit 5e99e226bb0ee1343ddc61b0f730de584efd5d79
Author: Sergey Soldatov <[email protected]>
AuthorDate: Tue Aug 11 11:49:11 2026 -0700
HDDS-16081. TestOMRatisSnapshots fails due to leader applied index taken
before flush (#10959)
Co-authored-by: Claude Opus <[email protected]>
---
.../hadoop/ozone/om/TestOMRatisSnapshots.java | 31 ++++++++++++++--------
1 file changed, 20 insertions(+), 11 deletions(-)
diff --git
a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
index 27cfa6d10e9..1bfab3c3c6b 100644
---
a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
+++
b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
@@ -301,14 +301,16 @@ public void testInstallSnapshotWithClientWrite() throws
Exception {
assertLogCapture(logCapture, msg);
assertLogCapture(logCapture, "Install Checkpoint is finished");
- // Wait for the follower to apply everything the leader has applied; all
+ // Wait for the follower to apply everything the leader has committed; all
// writes have completed on the leader, so after this no further snapshot
// install (and DB reload) can occur and the follower DB reads below are
// safe from "Rocks Database is closed" races.
- long leaderApplied = leaderOM.getOmRatisServer()
- .getLastAppliedTermIndex().getIndex();
+ // The applied index only advances on double buffer flush, so it can lag
+ // acked writes; the commit index covers them all.
+ long leaderCommitted = leaderRatisServer.getServerDivision().getRaftLog()
+ .getLastCommittedIndex();
GenericTestUtils.waitFor(() -> followerOM.getOmRatisServer()
- .getLastAppliedTermIndex().getIndex() >= leaderApplied, 100, 30_000);
+ .getLastAppliedTermIndex().getIndex() >= leaderCommitted, 100, 30_000);
long followerOMLastAppliedIndex =
followerOM.getOmRatisServer().getLastAppliedTermIndex().getIndex();
@@ -373,6 +375,12 @@ public void testInstallSnapshotWithClientRead() throws
Exception {
long leaderOMSnapshotIndex = leaderOMTermIndex.getIndex();
long leaderOMSnapshotTermIndex = leaderOMTermIndex.getTerm();
+ // The snapshot index above only advances on double buffer flush, so it can
+ // lag the keys written above; the commit index covers them all. Read after
+ // the snapshot index, so it is never behind it.
+ long leaderCommitted = leaderRatisServer.getServerDivision().getRaftLog()
+ .getLastCommittedIndex();
+
// Start the inactive OM. Checkpoint installation will happen
spontaneously.
OzoneManager.setTestInstallSnapshot(true);
cluster.startInactiveOM(followerNodeId);
@@ -395,12 +403,11 @@ public void testInstallSnapshotWithClientRead() throws
Exception {
});
readFuture.get();
- // The recently started OM should be lagging behind the leader OM.
- // Wait & for follower to update transactions to leader snapshot index.
- // Timeout error if follower does not load update within 3s
+ // The recently started OM should be lagging behind the leader OM. Wait for
+ // it to catch up, which covers the snapshot index and all the keys.
GenericTestUtils.waitFor(() -> {
return followerOM.getOmRatisServer().getLastAppliedTermIndex().getIndex()
- >= leaderOMSnapshotIndex - 1;
+ >= leaderCommitted;
}, 100, 30_000);
long followerOMLastAppliedIndex =
@@ -469,10 +476,12 @@ public void testInstallOldCheckpointFailure() throws
Exception {
// Wait for the follower to finish applying in-flight transactions, so
// that the TermIndex read below matches what installCheckpoint observes.
- long leaderAppliedIndex = leaderOM.getOmRatisServer()
- .getLastAppliedTermIndex().getIndex();
+ // The applied index only advances on double buffer flush, so it can lag
+ // acked writes; the commit index covers them all.
+ long leaderCommitted = leaderOM.getOmRatisServer().getServerDivision()
+ .getRaftLog().getLastCommittedIndex();
GenericTestUtils.waitFor(() -> followerRatisServer
- .getLastAppliedTermIndex().getIndex() >= leaderAppliedIndex, 100,
10_000);
+ .getLastAppliedTermIndex().getIndex() >= leaderCommitted, 100, 10_000);
// Install the old checkpoint on the follower OM. This should fail as the
// followerOM is already ahead of that transactionLogIndex and the OM
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]