This is an automated email from the ASF dual-hosted git repository.

smengcl pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/ozone.git


The following commit(s) were added to refs/heads/master by this push:
     new 5e99e226bb0 HDDS-16081. TestOMRatisSnapshots fails due to leader 
applied index taken before flush (#10959)
5e99e226bb0 is described below

commit 5e99e226bb0ee1343ddc61b0f730de584efd5d79
Author: Sergey Soldatov <[email protected]>
AuthorDate: Tue Aug 11 11:49:11 2026 -0700

    HDDS-16081. TestOMRatisSnapshots fails due to leader applied index taken 
before flush (#10959)
    
    Co-authored-by: Claude Opus <[email protected]>
---
 .../hadoop/ozone/om/TestOMRatisSnapshots.java      | 31 ++++++++++++++--------
 1 file changed, 20 insertions(+), 11 deletions(-)

diff --git 
a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
 
b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
index 27cfa6d10e9..1bfab3c3c6b 100644
--- 
a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
+++ 
b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/om/TestOMRatisSnapshots.java
@@ -301,14 +301,16 @@ public void testInstallSnapshotWithClientWrite() throws 
Exception {
     assertLogCapture(logCapture, msg);
     assertLogCapture(logCapture, "Install Checkpoint is finished");
 
-    // Wait for the follower to apply everything the leader has applied; all
+    // Wait for the follower to apply everything the leader has committed; all
     // writes have completed on the leader, so after this no further snapshot
     // install (and DB reload) can occur and the follower DB reads below are
     // safe from "Rocks Database is closed" races.
-    long leaderApplied = leaderOM.getOmRatisServer()
-        .getLastAppliedTermIndex().getIndex();
+    // The applied index only advances on double buffer flush, so it can lag
+    // acked writes; the commit index covers them all.
+    long leaderCommitted = leaderRatisServer.getServerDivision().getRaftLog()
+        .getLastCommittedIndex();
     GenericTestUtils.waitFor(() -> followerOM.getOmRatisServer()
-        .getLastAppliedTermIndex().getIndex() >= leaderApplied, 100, 30_000);
+        .getLastAppliedTermIndex().getIndex() >= leaderCommitted, 100, 30_000);
 
     long followerOMLastAppliedIndex =
         followerOM.getOmRatisServer().getLastAppliedTermIndex().getIndex();
@@ -373,6 +375,12 @@ public void testInstallSnapshotWithClientRead() throws 
Exception {
     long leaderOMSnapshotIndex = leaderOMTermIndex.getIndex();
     long leaderOMSnapshotTermIndex = leaderOMTermIndex.getTerm();
 
+    // The snapshot index above only advances on double buffer flush, so it can
+    // lag the keys written above; the commit index covers them all. Read after
+    // the snapshot index, so it is never behind it.
+    long leaderCommitted = leaderRatisServer.getServerDivision().getRaftLog()
+        .getLastCommittedIndex();
+
     // Start the inactive OM. Checkpoint installation will happen 
spontaneously.
     OzoneManager.setTestInstallSnapshot(true);
     cluster.startInactiveOM(followerNodeId);
@@ -395,12 +403,11 @@ public void testInstallSnapshotWithClientRead() throws 
Exception {
     });
     readFuture.get();
 
-    // The recently started OM should be lagging behind the leader OM.
-    // Wait & for follower to update transactions to leader snapshot index.
-    // Timeout error if follower does not load update within 3s
+    // The recently started OM should be lagging behind the leader OM. Wait for
+    // it to catch up, which covers the snapshot index and all the keys.
     GenericTestUtils.waitFor(() -> {
       return followerOM.getOmRatisServer().getLastAppliedTermIndex().getIndex()
-          >= leaderOMSnapshotIndex - 1;
+          >= leaderCommitted;
     }, 100, 30_000);
 
     long followerOMLastAppliedIndex =
@@ -469,10 +476,12 @@ public void testInstallOldCheckpointFailure() throws 
Exception {
 
     // Wait for the follower to finish applying in-flight transactions, so
     // that the TermIndex read below matches what installCheckpoint observes.
-    long leaderAppliedIndex = leaderOM.getOmRatisServer()
-        .getLastAppliedTermIndex().getIndex();
+    // The applied index only advances on double buffer flush, so it can lag
+    // acked writes; the commit index covers them all.
+    long leaderCommitted = leaderOM.getOmRatisServer().getServerDivision()
+        .getRaftLog().getLastCommittedIndex();
     GenericTestUtils.waitFor(() -> followerRatisServer
-        .getLastAppliedTermIndex().getIndex() >= leaderAppliedIndex, 100, 
10_000);
+        .getLastAppliedTermIndex().getIndex() >= leaderCommitted, 100, 10_000);
 
     // Install the old checkpoint on the follower OM. This should fail as the
     // followerOM is already ahead of that transactionLogIndex and the OM


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to