This is an automated email from the ASF dual-hosted git repository. numinnex pushed a commit to branch torn_wal_on_kernel_version in repository https://gitbox.apache.org/repos/asf/iggy.git
commit c06dccf1fe130b8c6d503491b6902d64f64fcf3d Author: Grzegorz Koszyk <[email protected]> AuthorDate: Thu Aug 27 13:11:16 2026 +0200 fix(journal): repair a torn WAL tail instead of panicking the shard --- core/journal/src/file_storage.rs | 82 ++++++++++++++++++++++++++++++++----- core/journal/src/prepare_journal.rs | 10 +---- helm/charts/iggy/README.md | 8 +++- 3 files changed, 80 insertions(+), 20 deletions(-) diff --git a/core/journal/src/file_storage.rs b/core/journal/src/file_storage.rs index 1273d27b9..8ac36b2ba 100644 --- a/core/journal/src/file_storage.rs +++ b/core/journal/src/file_storage.rs @@ -19,6 +19,7 @@ use crate::Storage; use compio::buf::IoBuf; use compio::io::{AsyncReadAtExt, AsyncWriteAtExt}; use std::cell::{Cell, UnsafeCell}; +use std::fs; use std::io; use std::path::{Path, PathBuf}; @@ -57,18 +58,27 @@ impl FileStorage { self.write_offset.get() } - /// Truncate the file to `len` bytes. + /// Truncate the file to `len` bytes and make the new length durable. + /// + /// Synchronous `std::fs` on a separate descriptor, not compio: compio's + /// `set_len` submits `IORING_OP_FTRUNCATE`, which landed in kernel 6.9. + /// Below that the driver falls back to its blocking pool, and shard + /// proactors run with `thread_pool_limit(0)`, so the fallback panics the + /// shard instead of repairing the WAL. `std::fs` needs neither the opcode + /// nor the pool. The sole caller is boot-time torn-tail repair, so + /// blocking the shard thread here costs nothing. + /// + /// `sync_all`, not `sync_data`: the file length is metadata, and without + /// it a power cut right after the repair re-presents the torn tail on the + /// next boot. Mirrors the write-then-fsync the `write_append` path pairs + /// with, and the truncate-then-`sync_all` segment recovery performs. /// /// # Errors - /// Returns an I/O error if truncation fails. - // TODO(hubcio): compio `set_len` submits IORING_OP_FTRUNCATE, which kernels - // below 6.9 do not support; the driver then falls back to its blocking - // pool, and shard proactors run with `thread_pool_limit(0)`, so the torn - // WAL repair panics the shard on such kernels instead of repairing. Use a - // synchronous `std::fs` truncate here (boot-time path) or gate on a probe. - pub async fn truncate(&self, len: u64) -> io::Result<()> { - let file = unsafe { &*self.file.get() }; - file.set_len(len).await?; + /// Returns an I/O error if the file cannot be opened, truncated, or synced. + pub fn truncate(&self, len: u64) -> io::Result<()> { + let file = fs::OpenOptions::new().write(true).open(&self.path)?; + file.set_len(len)?; + file.sync_all()?; self.write_offset.set(len); Ok(()) } @@ -178,3 +188,55 @@ impl Storage for FileStorage { Ok(buffer) } } + +#[cfg(test)] +mod tests { + use super::FileStorage; + use server_common::executor::create_shard_executor; + use tempfile::tempdir; + + /// Torn-tail repair runs on a shard executor, which builds its proactor + /// with `thread_pool_limit(0)`. Any truncate that reaches compio's + /// blocking pool panics that shard rather than repairing the WAL, which + /// is what every kernel below 6.9 did while `set_len` was an `io_uring` + /// submission. Driving the repair through a real shard executor is the + /// only way to keep the no-blocking-pool constraint pinned. + #[test] + fn given_a_shard_executor_with_no_blocking_pool_when_truncating_should_repair_the_file() { + let runtime = create_shard_executor().unwrap(); + runtime.block_on(async { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let storage = FileStorage::open(&path).await.unwrap(); + storage.write_append(vec![0xAB_u8; 128]).await.unwrap(); + + storage.truncate(64).unwrap(); + + assert_eq!(storage.file_len(), 64); + assert_eq!(std::fs::metadata(&path).unwrap().len(), 64); + }); + } + + /// The repair has to survive the crash it is repairing from: a reopen + /// that still saw the torn tail would walk and truncate it again on every + /// boot, and a reopen that saw a longer file would resurrect the bytes + /// recovery just proved dead. + #[test] + fn given_a_truncated_file_when_reopened_should_see_the_shortened_length() { + let runtime = create_shard_executor().unwrap(); + runtime.block_on(async { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + { + let storage = FileStorage::open(&path).await.unwrap(); + storage.write_append(vec![0xAB_u8; 128]).await.unwrap(); + storage.fsync().await.unwrap(); + storage.truncate(64).unwrap(); + } + + let reopened = FileStorage::open(&path).await.unwrap(); + + assert_eq!(reopened.file_len(), 64); + }); + } +} diff --git a/core/journal/src/prepare_journal.rs b/core/journal/src/prepare_journal.rs index 4c0f3dc54..9251dfe2a 100644 --- a/core/journal/src/prepare_journal.rs +++ b/core/journal/src/prepare_journal.rs @@ -243,12 +243,7 @@ async fn truncate_or_fail( reason, "truncating torn WAL tail; no complete entry follows the damage" ); - storage.truncate(pos).await?; - // The repair must be crash-durable. `FileStorage::truncate` is a - // bare `set_len`; without this fsync a power loss right after the - // repair re-presents the torn tail on the next boot. Mirrors the - // write-then-fsync the `append` path already does. - storage.fsync().await?; + storage.truncate(pos)?; Ok(()) } @@ -1745,8 +1740,7 @@ mod tests { let storage = FileStorage::open(&path).await.unwrap(); let full_len = storage.file_len(); // Remove the last 10 bytes (partial second entry) - storage.truncate(full_len - 10).await.unwrap(); - storage.fsync().await.unwrap(); + storage.truncate(full_len - 10).unwrap(); } // Reopen, should recover only the first entry diff --git a/helm/charts/iggy/README.md b/helm/charts/iggy/README.md index 4e4cd4211..9e7780fd6 100644 --- a/helm/charts/iggy/README.md +++ b/helm/charts/iggy/README.md @@ -15,8 +15,12 @@ A Helm chart for Apache Iggy server and web-ui Iggy server uses `io_uring` for high-performance async I/O. This requires: -1. **IPC_LOCK capability** - For locking memory required by io_uring -2. **Unconfined seccomp profile** - To allow io_uring syscalls +1. **Linux kernel 5.19 or newer on the node** - The shard executor sets up its + rings with `IORING_SETUP_COOP_TASKRUN` and `IORING_SETUP_TASKRUN_FLAG`, which + the kernel rejects below 5.19. The server fails during startup on older + kernels. The node kernel is what matters, not the container image. +2. **IPC_LOCK capability** - For locking memory required by io_uring +3. **Unconfined seccomp profile** - To allow io_uring syscalls These are configured by default for the Iggy server via the chart's root-level `securityContext` and `podSecurityContext`. The web UI uses `ui.securityContext`
