This is an automated email from the ASF dual-hosted git repository.

numinnex pushed a commit to branch torn_wal_on_kernel_version
in repository https://gitbox.apache.org/repos/asf/iggy.git

commit c06dccf1fe130b8c6d503491b6902d64f64fcf3d
Author: Grzegorz Koszyk <[email protected]>
AuthorDate: Thu Aug 27 13:11:16 2026 +0200

    fix(journal): repair a torn WAL tail instead of panicking the shard
---
 core/journal/src/file_storage.rs    | 82 ++++++++++++++++++++++++++++++++-----
 core/journal/src/prepare_journal.rs | 10 +----
 helm/charts/iggy/README.md          |  8 +++-
 3 files changed, 80 insertions(+), 20 deletions(-)

diff --git a/core/journal/src/file_storage.rs b/core/journal/src/file_storage.rs
index 1273d27b9..8ac36b2ba 100644
--- a/core/journal/src/file_storage.rs
+++ b/core/journal/src/file_storage.rs
@@ -19,6 +19,7 @@ use crate::Storage;
 use compio::buf::IoBuf;
 use compio::io::{AsyncReadAtExt, AsyncWriteAtExt};
 use std::cell::{Cell, UnsafeCell};
+use std::fs;
 use std::io;
 use std::path::{Path, PathBuf};
 
@@ -57,18 +58,27 @@ impl FileStorage {
         self.write_offset.get()
     }
 
-    /// Truncate the file to `len` bytes.
+    /// Truncate the file to `len` bytes and make the new length durable.
+    ///
+    /// Synchronous `std::fs` on a separate descriptor, not compio: compio's
+    /// `set_len` submits `IORING_OP_FTRUNCATE`, which landed in kernel 6.9.
+    /// Below that the driver falls back to its blocking pool, and shard
+    /// proactors run with `thread_pool_limit(0)`, so the fallback panics the
+    /// shard instead of repairing the WAL. `std::fs` needs neither the opcode
+    /// nor the pool. The sole caller is boot-time torn-tail repair, so
+    /// blocking the shard thread here costs nothing.
+    ///
+    /// `sync_all`, not `sync_data`: the file length is metadata, and without
+    /// it a power cut right after the repair re-presents the torn tail on the
+    /// next boot. Mirrors the write-then-fsync the `write_append` path pairs
+    /// with, and the truncate-then-`sync_all` segment recovery performs.
     ///
     /// # Errors
-    /// Returns an I/O error if truncation fails.
-    // TODO(hubcio): compio `set_len` submits IORING_OP_FTRUNCATE, which 
kernels
-    // below 6.9 do not support; the driver then falls back to its blocking
-    // pool, and shard proactors run with `thread_pool_limit(0)`, so the torn
-    // WAL repair panics the shard on such kernels instead of repairing. Use a
-    // synchronous `std::fs` truncate here (boot-time path) or gate on a probe.
-    pub async fn truncate(&self, len: u64) -> io::Result<()> {
-        let file = unsafe { &*self.file.get() };
-        file.set_len(len).await?;
+    /// Returns an I/O error if the file cannot be opened, truncated, or 
synced.
+    pub fn truncate(&self, len: u64) -> io::Result<()> {
+        let file = fs::OpenOptions::new().write(true).open(&self.path)?;
+        file.set_len(len)?;
+        file.sync_all()?;
         self.write_offset.set(len);
         Ok(())
     }
@@ -178,3 +188,55 @@ impl Storage for FileStorage {
         Ok(buffer)
     }
 }
+
+#[cfg(test)]
+mod tests {
+    use super::FileStorage;
+    use server_common::executor::create_shard_executor;
+    use tempfile::tempdir;
+
+    /// Torn-tail repair runs on a shard executor, which builds its proactor
+    /// with `thread_pool_limit(0)`. Any truncate that reaches compio's
+    /// blocking pool panics that shard rather than repairing the WAL, which
+    /// is what every kernel below 6.9 did while `set_len` was an `io_uring`
+    /// submission. Driving the repair through a real shard executor is the
+    /// only way to keep the no-blocking-pool constraint pinned.
+    #[test]
+    fn 
given_a_shard_executor_with_no_blocking_pool_when_truncating_should_repair_the_file()
 {
+        let runtime = create_shard_executor().unwrap();
+        runtime.block_on(async {
+            let dir = tempdir().unwrap();
+            let path = dir.path().join("journal.wal");
+            let storage = FileStorage::open(&path).await.unwrap();
+            storage.write_append(vec![0xAB_u8; 128]).await.unwrap();
+
+            storage.truncate(64).unwrap();
+
+            assert_eq!(storage.file_len(), 64);
+            assert_eq!(std::fs::metadata(&path).unwrap().len(), 64);
+        });
+    }
+
+    /// The repair has to survive the crash it is repairing from: a reopen
+    /// that still saw the torn tail would walk and truncate it again on every
+    /// boot, and a reopen that saw a longer file would resurrect the bytes
+    /// recovery just proved dead.
+    #[test]
+    fn given_a_truncated_file_when_reopened_should_see_the_shortened_length() {
+        let runtime = create_shard_executor().unwrap();
+        runtime.block_on(async {
+            let dir = tempdir().unwrap();
+            let path = dir.path().join("journal.wal");
+            {
+                let storage = FileStorage::open(&path).await.unwrap();
+                storage.write_append(vec![0xAB_u8; 128]).await.unwrap();
+                storage.fsync().await.unwrap();
+                storage.truncate(64).unwrap();
+            }
+
+            let reopened = FileStorage::open(&path).await.unwrap();
+
+            assert_eq!(reopened.file_len(), 64);
+        });
+    }
+}
diff --git a/core/journal/src/prepare_journal.rs 
b/core/journal/src/prepare_journal.rs
index 4c0f3dc54..9251dfe2a 100644
--- a/core/journal/src/prepare_journal.rs
+++ b/core/journal/src/prepare_journal.rs
@@ -243,12 +243,7 @@ async fn truncate_or_fail(
         reason,
         "truncating torn WAL tail; no complete entry follows the damage"
     );
-    storage.truncate(pos).await?;
-    // The repair must be crash-durable. `FileStorage::truncate` is a
-    // bare `set_len`; without this fsync a power loss right after the
-    // repair re-presents the torn tail on the next boot. Mirrors the
-    // write-then-fsync the `append` path already does.
-    storage.fsync().await?;
+    storage.truncate(pos)?;
     Ok(())
 }
 
@@ -1745,8 +1740,7 @@ mod tests {
             let storage = FileStorage::open(&path).await.unwrap();
             let full_len = storage.file_len();
             // Remove the last 10 bytes (partial second entry)
-            storage.truncate(full_len - 10).await.unwrap();
-            storage.fsync().await.unwrap();
+            storage.truncate(full_len - 10).unwrap();
         }
 
         // Reopen, should recover only the first entry
diff --git a/helm/charts/iggy/README.md b/helm/charts/iggy/README.md
index 4e4cd4211..9e7780fd6 100644
--- a/helm/charts/iggy/README.md
+++ b/helm/charts/iggy/README.md
@@ -15,8 +15,12 @@ A Helm chart for Apache Iggy server and web-ui
 
 Iggy server uses `io_uring` for high-performance async I/O. This requires:
 
-1. **IPC_LOCK capability** - For locking memory required by io_uring
-2. **Unconfined seccomp profile** - To allow io_uring syscalls
+1. **Linux kernel 5.19 or newer on the node** - The shard executor sets up its
+   rings with `IORING_SETUP_COOP_TASKRUN` and `IORING_SETUP_TASKRUN_FLAG`, 
which
+   the kernel rejects below 5.19. The server fails during startup on older
+   kernels. The node kernel is what matters, not the container image.
+2. **IPC_LOCK capability** - For locking memory required by io_uring
+3. **Unconfined seccomp profile** - To allow io_uring syscalls
 
 These are configured by default for the Iggy server via the chart's root-level
 `securityContext` and `podSecurityContext`. The web UI uses 
`ui.securityContext`

Reply via email to