Hey Samuel.

Oh wow the timing. Please apply the v2 instead...or if you want i can send
the delta!

On Mon, Oct 5, 2026 at 10:40 AM Samuel Thibault <[email protected]>
wrote:

> Applied, thanks!
>
> Milos Nikic, le dim. 04 oct. 2026 13:17:43 -0700, a ecrit:
> > Previously journal was notified after the fact (block has already been
> > changed).
> > This patch changes this, journal is now notified ahead of time that a
> > block is about to be altered, and journal is also notified when we are
> > done editing.
> >
> > This helps keep the main filesystem free of torn writes and makes
> > our journal behave much more closely to the journal in ext4.
> >
> > Function names have been changed to reflect ext4 journal functions
> > (journal_get_write_access() to pre-notify the journal and
> > journal_mark_dirty() to tell the journal we are done modifying.)
> >
> > journal_record_freed_blocks () is now taking transaction handle instead
> > of trying to find out which txn is running inside. This aligns it more
> > to the other journaling functions.
> >
> > A new function, journal_thread_transaction (), returns the transaction of
> > the calling thread's open handle.  The edit sites use it to get the
> > transaction they pass to journal_get_write_access () and
> > journal_mark_dirty ().  While the journal is live every metadata edit
> > must run inside a handle, so the result is never NULL there, and the
> > handle keeps the transaction T_RUNNING or T_LOCKED until it is released;
> > both are asserted.  It returns NULL only when there is no journal, or the
> > journal is shutting down and no handle was opened, and callers then skip
> > journaling.  Unlike the old lookup in journal_record_freed_blocks (), it
> > never falls back to j_running_transaction for a thread without a handle.
> >
> > journal_get_write_access is a silent no-op when the system is
> > unjournaled.
> > ---
> >  ext2fs/balloc.c   |  10 +++-
> >  ext2fs/ext2fs.h   |  72 ++++++++++++++++++----------
> >  ext2fs/getblk.c   |   8 ++++
> >  ext2fs/hyper.c    |   2 +
> >  ext2fs/ialloc.c   |  11 ++++-
> >  ext2fs/inode.c    |  20 ++++++--
> >  ext2fs/journal.c  | 117 +++++++++++++++++++++++++++++++++++-----------
> >  ext2fs/journal.h  |   2 +-
> >  ext2fs/orphan.c   |  66 ++++++++++++--------------
> >  ext2fs/truncate.c |   7 +++
> >  ext2fs/xattr.c    |   9 ++++
> >  11 files changed, 228 insertions(+), 96 deletions(-)
> >
> > diff --git a/ext2fs/balloc.c b/ext2fs/balloc.c
> > index cc00fe4cc..6b06a3688 100644
> > --- a/ext2fs/balloc.c
> > +++ b/ext2fs/balloc.c
> > @@ -63,11 +63,14 @@ ext2_free_blocks (block_t block, unsigned long count)
> >    unsigned long bit;
> >    unsigned long i;
> >    struct ext2_group_desc *gdp;
> > +  diskfs_transaction_t *txn;
> >
> >    /* Trap trying to free superblock, block group descriptor table, or
> beyond the end */
> >    assert_backtrace (block >= group_desc_block_end
> >                && block + count <= store->size >> log2_block_size);
> >
> > +  txn = journal_thread_transaction ();
> > +
> >    pthread_spin_lock (&global_lock);
> >
> >    if (block < le32toh (sblock->s_first_data_block) ||
> > @@ -103,6 +106,8 @@ ext2_free_blocks (block_t block, unsigned long count)
> >                     block, count);
> >       }
> >        gdp = group_desc (block_group);
> > +      journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
> > +      journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
> >        bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap));
> >
> >        if (in_range (le32toh (gdp->bg_block_bitmap), block, gcount) ||
> > @@ -113,7 +118,7 @@ ext2_free_blocks (block_t block, unsigned long count)
> >                   "block = %u, count = %lu",
> >                   block, count);
> >
> > -      journal_record_freed_blocks (block, gcount);
> > +      journal_record_freed_blocks (txn, block, gcount);
> >        for (i = 0; i < gcount; i++)
> >       {
> >         if (!clear_bit (bit + i, bh))
> > @@ -160,6 +165,7 @@ ext2_new_block (block_t goal,
> >    uint32_t lmap;
> >    struct ext2_group_desc *gdp;
> >
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> >  #ifdef EXT2FS_DEBUG
> >    static int goal_hits = 0, goal_attempts = 0;
> >  #endif
> > @@ -317,6 +323,8 @@ search_back:
> >
> >  got_block:
> >    assert_backtrace (bh != NULL);
> > +  journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
> > +  journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
> >
> >    ext2_debug ("using block group %d (%d)", i, le16toh
> (gdp->bg_free_blocks_count));
> >
> > diff --git a/ext2fs/ext2fs.h b/ext2fs/ext2fs.h
> > index 9282e5417..458446984 100644
> > --- a/ext2fs/ext2fs.h
> > +++ b/ext2fs/ext2fs.h
> > @@ -346,12 +346,39 @@ extern struct journal *ext2_journal;
> >  #define JRNL_LOG_WARN(fmt, ...) ext2_warning ("[JOURNAL] " fmt,
> ##__VA_ARGS__)
> >
> >  /**
> > - * Mark dirty: Add a modified filesystem block to the given transaction.
> > - * Performs a shadow copy of 'data' into the journal memory.
> > + * Notify the journal that the content of the block fs_blocknr is about
> to get
> > + * modified by a filesystem operation.
> > + * Journal will arm necessary buffers to track the block and it will
> > + * automatically copy the memory contained at that block when the
> > + * transaction txn is committing and the block's memory has settled.
> > + *
> > + * If the memory of the block is modified before calling this function
> such
> > + * modifications won't be part of the journal transaction.
> > + *
> > + * In between the invocation of this function and the commit, journal
> will be
> > + * intercepting pager's writes to this block and instead of to disk
> they will
> > + * be temporarily stored in journal's caches.
> >   */
> >  error_t
> > -journal_dirty_block (diskfs_transaction_t * txn, block_t fs_blocknr);
> > +journal_get_write_access (diskfs_transaction_t * txn, block_t
> fs_blocknr);
> > +
> > +/**
> > + * Notifies the journal that modifications for this block have been
> complete;
> > + * the block is copied into the transaction again at the next sweep.
> > + *
> > + * Shouldn't be called if no journal_get_write_access() was called
> previously
> > + * for the same transaction and the same block.
> > + */
> > +void
> > +journal_mark_dirty (diskfs_transaction_t * txn, block_t fs_blocknr);
> >
> > +/* Return the transaction of the calling thread's open handle.  While
> the
> > +   journal is live the caller must hold a handle and the result is a
> > +   T_RUNNING or T_LOCKED transaction that stays usable until the handle
> is
> > +   released.  NULL means there is nothing to journal into: no journal,
> or
> > +   the journal is shutting down and no handle was opened.  */
> > +diskfs_transaction_t *
> > +journal_thread_transaction (void);
> >
> >  void ext2_orphan_drop_ram_link (struct node *np);
> >
> > @@ -531,17 +558,9 @@ extern void alloc_sync (struct node *np);
> >
> >  #if defined(__USE_EXTERN_INLINES) || defined(EXT2FS_DEFINE_EI)
> >  EXT2FS_EI void
> > -journal_notify_block_changed (block_t block)
> > +journal_mark_dirty_current (block_t block)
> >  {
> > -  if (!ext2_journal)
> > -    return;
> > -
> > -  diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> > -  error_t err = journal_dirty_block (txn, block);
> > -  if (err)
> > -    JRNL_LOG_WARN ("Didn't manage to add a dirty block %u to the
> journal. (%s).",
> > -                block, strerror (err));
> > -  diskfs_journal_stop_transaction (txn);
> > +  journal_mark_dirty (journal_thread_transaction (), block);
> >  }
> >
> >  /* Marks the global block BLOCK as being modified, and returns true if
> we
> > @@ -569,7 +588,7 @@ record_global_poke (void *ptr)
> >  {
> >    block_t block = boffs_block (bptr_offs (ptr));
> >    void *block_ptr = bptr (block);
> > -  journal_notify_block_changed (block);
> > +  journal_mark_dirty_current (block);
> >    ext2_debug ("(%p = %p)", ptr, block_ptr);
> >  #ifdef EXT2FS_DEBUG
> >    assert_backtrace (disk_cache_block_is_ref (block));
> > @@ -584,7 +603,7 @@ sync_global_ptr (void *ptr, int wait)
> >  {
> >    block_t block = boffs_block (bptr_offs (ptr));
> >    void *block_ptr = bptr (block);
> > -  journal_notify_block_changed (block);
> > +  journal_mark_dirty_current (block);
> >    ext2_debug ("(%p -> %u)", ptr, block);
> >    global_block_modified (block);
> >    _disk_cache_block_deref (block_ptr);
> > @@ -611,7 +630,7 @@ record_indir_poke (struct node *node, void *ptr)
> >  {
> >    block_t block = boffs_block (bptr_offs (ptr));
> >    void *block_ptr = bptr (block);
> > -  journal_notify_block_changed (block);
> > +  journal_mark_dirty_current (block);
> >    ext2_debug ("(%llu, %p)", node->cache_id, ptr);
> >  #ifdef EXT2FS_DEBUG
> >    assert_backtrace (disk_cache_block_is_ref (block));
> > @@ -630,8 +649,8 @@ sync_global (int wait)
> >  }
> >
> >  /* Sync all allocation information and node NP if diskfs_synchronous.
> > -   If journaling is active, we just update memory (wait=0) and let the
> > -   transaction commit handle durability. */
> > +   If journaling is active, push the superblock into this transaction;
> > +   the commit provides durability.  */
> >  EXT2FS_EI void
> >  alloc_sync (struct node *np)
> >  {
> > @@ -640,15 +659,16 @@ alloc_sync (struct node *np)
> >      {
> >        diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> >
> > +      /* If the superblock was modified in memory, push it to the disk
> cache
> > +         now so it gets bundled into this transaction's WAL commit.
> > +         diskfs_set_hypermetadata natively handles the get_write_access
> ->
> > +         memcpy -> mark_dirty pipeline! */
> > +      if (sblock_dirty && ext2_journal)
> > +        diskfs_set_hypermetadata (0, 0);
> > +
> >        if (np)
> >          diskfs_node_update (np, diskfs_synchronous);
> >
> > -      if (sblock_dirty && ext2_journal)
> > -        {
> > -          block_t sb_blocknr = boffs_block (SBLOCK_OFFS);
> > -          journal_dirty_block (txn, sb_blocknr);
> > -        }
> > -
> >        diskfs_journal_stop_transaction (txn);
> >      }
> >
> > @@ -658,7 +678,9 @@ alloc_sync (struct node *np)
> >        if (np)
> >          pokel_sync (&diskfs_node_disknode (np)->indir_pokel, 1);
> >
> > -      diskfs_set_hypermetadata (1, 0);
> > +      /* Only flush if journaling didn't already update the cache above
> */
> > +      if (!ext2_journal)
> > +        diskfs_set_hypermetadata (1, 0);
> >      }
> >  }
> >  #endif /* Use extern inlines.  */
> > diff --git a/ext2fs/getblk.c b/ext2fs/getblk.c
> > index f2a219437..7fe219f2b 100644
> > --- a/ext2fs/getblk.c
> > +++ b/ext2fs/getblk.c
> > @@ -70,6 +70,7 @@ ext2_alloc_block (struct node *node, block_t goal, int
> zero)
> >    static unsigned long alloc_hits = 0, alloc_attempts = 0;
> >  #endif
> >    block_t result;
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> >
> >  #ifdef EXT2_PREALLOCATE
> >    if (diskfs_node_disknode (node)->info.i_prealloc_count &&
> > @@ -110,7 +111,9 @@ ext2_alloc_block (struct node *node, block_t goal,
> int zero)
> >    if (result && zero)
> >      {
> >        char *bh = disk_cache_block_ref (result);
> > +      journal_get_write_access (txn, result);
> >        memset (bh, 0, block_size);
> > +      /* record_indir_poke handles journal_mark_dirty internally. */
> >        record_indir_poke (node, bh);
> >      }
> >
> > @@ -196,6 +199,7 @@ block_getblk (struct node *node, block_t block, int
> nr, int create, int zero,
> >    int i;
> >    block_t goal = 0;
> >    block_t *bh = (block_t *)disk_cache_block_ref (block);
> > +  diskfs_transaction_t *txn;
> >
> >    *result = bh[nr];
> >    if (*result)
> > @@ -209,6 +213,7 @@ block_getblk (struct node *node, block_t block, int
> nr, int create, int zero,
> >        disk_cache_block_deref (bh);
> >        return EINVAL;
> >      }
> > +  txn = journal_thread_transaction ();
> >
> >    if (diskfs_node_disknode (node)->info.i_next_alloc_block == new_block)
> >      goal = diskfs_node_disknode (node)->info.i_next_alloc_goal;
> > @@ -233,10 +238,12 @@ block_getblk (struct node *node, block_t block,
> int nr, int create, int zero,
> >        return ENOSPC;
> >      }
> >
> > +  journal_get_write_access (txn, block);
> >    bh[nr] = *result;
> >
> >    if (diskfs_synchronous || diskfs_node_disknode (node)->info.i_osync)
> >      {
> > +      /* calls journal_mark_dirty internally */
> >        sync_global_ptr (bh, 1);
> >        /* We just wrote a new indirect block pointer.
> >           If this doesn't hit the platter, the file is corrupt. */
> > @@ -245,6 +252,7 @@ block_getblk (struct node *node, block_t block, int
> nr, int create, int zero,
> >       ext2_warning ("indirect block flush failed: %s", strerror (err));
> >      }
> >    else
> > +    /* record_indir_poke handles journal_mark_dirty internally */
> >      record_indir_poke (node, bh);
> >
> >    diskfs_node_disknode (node)->info.i_next_alloc_block = new_block;
> > diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c
> > index 35cdb7ed0..60be3e22f 100644
> > --- a/ext2fs/hyper.c
> > +++ b/ext2fs/hyper.c
> > @@ -240,6 +240,8 @@ diskfs_set_hypermetadata (int wait, int clean)
> >        /* Before writing, set the time of write */
> >        sblock->s_wtime = htole32 (diskfs_mtime->seconds);
> >        sblock_dirty = 0;
> > +      block_t blk = boffs_block (bptr_offs (mapped_sblock));
> > +      journal_get_write_access (txn, blk);
> >        memcpy (mapped_sblock, sblock, SBLOCK_SIZE);
> >        disk_cache_block_ref_ptr (mapped_sblock);
> >        record_global_poke (mapped_sblock);
> > diff --git a/ext2fs/ialloc.c b/ext2fs/ialloc.c
> > index 438de8089..cc9a1edf3 100644
> > --- a/ext2fs/ialloc.c
> > +++ b/ext2fs/ialloc.c
> > @@ -59,6 +59,8 @@ diskfs_free_node (struct node *np, mode_t old_mode)
> >    unsigned long bit;
> >    struct ext2_group_desc *gdp;
> >    ino_t inum = np->cache_id;
> > +  block_t gdp_block, gdp_bitmap_blk;
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> >
> >    assert_backtrace (!diskfs_readonly);
> >
> > @@ -79,8 +81,12 @@ diskfs_free_node (struct node *np, mode_t old_mode)
> >    bit = (inum - 1) % le32toh (sblock->s_inodes_per_group);
> >
> >    gdp = group_desc (block_group);
> > -  bh = disk_cache_block_ref (le32toh (gdp->bg_inode_bitmap));
> > +  gdp_block = boffs_block (bptr_offs (gdp));
> > +  gdp_bitmap_blk = le32toh (gdp->bg_inode_bitmap);
> > +  bh = disk_cache_block_ref (gdp_bitmap_blk);
> >
> > +  journal_get_write_access (txn, gdp_bitmap_blk);
> > +  journal_get_write_access (txn, gdp_block);
> >    if (!clear_bit (bit, bh))
> >      ext2_warning ("bit already cleared for inode %" PRIu64, inum);
> >    else
> > @@ -123,6 +129,7 @@ ext2_alloc_inode (ino_t dir_inum, mode_t mode)
> >    ino_t inum;
> >    struct ext2_group_desc *gdp;
> >    struct ext2_group_desc *tmp;
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> >
> >    pthread_spin_lock (&global_lock);
> >
> > @@ -228,6 +235,7 @@ repeat:
> >         find_first_zero_bit ((uint32_t *) bh, le32toh
> (sblock->s_inodes_per_group)))
> >        < le32toh (sblock->s_inodes_per_group))
> >      {
> > +      journal_get_write_access (txn, le32toh (gdp->bg_inode_bitmap));
> >        if (set_bit (inum, bh))
> >       {
> >         ext2_warning ("bit already set for inode %" PRIu64, inum);
> > @@ -258,6 +266,7 @@ repeat:
> >        goto sync_out;
> >      }
> >
> > +  journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
> >    gdp->bg_free_inodes_count = htole16 (le16toh
> (gdp->bg_free_inodes_count) - 1);
> >    if (S_ISDIR (mode))
> >      gdp->bg_used_dirs_count = htole16 (le16toh
> (gdp->bg_used_dirs_count) + 1);
> > diff --git a/ext2fs/inode.c b/ext2fs/inode.c
> > index 00c4bebb4..21256f843 100644
> > --- a/ext2fs/inode.c
> > +++ b/ext2fs/inode.c
> > @@ -408,6 +408,7 @@ write_node (struct node *np)
> >    error_t err;
> >    struct stat *st = &np->dn_stat;
> >    struct ext2_inode *di;
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> >
> >    ext2_debug ("(%llu)", np->cache_id);
> >
> > @@ -428,6 +429,7 @@ write_node (struct node *np)
> >
> >        di = dino_ref (np->cache_id);
> >
> > +      journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> >        di->i_generation = htole32 (st->st_gen);
> >
> >        /* We happen to know that the stat mode bits are the same
> > @@ -578,13 +580,17 @@ void
> >  diskfs_write_disknode (struct node *np, int wait)
> >  {
> >    error_t err;
> > +  diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> > +
> >    struct ext2_inode *di = write_node (np);
> >    if (!di)
> > -    return;
> > +    {
> > +      diskfs_journal_stop_transaction (txn);
> > +      return;
> > +    }
> >
> >    if (ext2_journal)
> >      {
> > -      diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> >        record_global_poke (di);
> >        if (wait)
> >          diskfs_journal_set_sync (txn);
> > @@ -601,9 +607,7 @@ diskfs_write_disknode (struct node *np, int wait)
> >          ext2_warning ("device flush failed: %s", strerror (err));
> >      }
> >    else
> > -    {
> > -      record_global_poke (di);
> > -    }
> > +    record_global_poke (di);
> >  }
> >
> >  /* Set *ST with appropriate values to reflect the current state of the
> > @@ -634,6 +638,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >                      struct protid *cred)
> >  {
> >    error_t err;
> > +  diskfs_transaction_t *txn;
> >
> >    assert_backtrace (!diskfs_readonly);
> >
> > @@ -641,6 +646,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >    if (err)
> >      return err;
> >
> > +  txn = journal_thread_transaction ();
> >    /* If xattr is supported for this filesystem, use xattr to store
> translator
> >       record, otherwise, use legacy translator record */
> >    if (EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR)
> > @@ -662,6 +668,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >         ext2_debug ("Old translator record found, clear it");
> >
> >         /* Clear block for translator going away. */
> > +       journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> >         di->i_translator = htole32 (0);
> >         diskfs_node_disknode (np)->info_i_translator = 0;
> >         record_global_poke (di);
> > @@ -761,6 +768,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >             np->dn_stat.st_mode = newmode;
> >           }
> >
> > +       journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> >         di->i_translator = htole32 (blkno);
> >         diskfs_node_disknode (np)->info_i_translator = blkno;
> >         record_global_poke (di);
> > @@ -771,6 +779,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >        else if (!namelen && blkno)
> >       {
> >         /* Clear block for translator going away. */
> > +       journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> >         di->i_translator = htole32 (0);
> >         diskfs_node_disknode (np)->info_i_translator = 0;
> >         record_global_poke (di);
> > @@ -796,6 +805,7 @@ diskfs_set_translator (struct node *np, const char
> *name, mach_msg_type_number_t
> >         memcpy (buf + 2, name, namelen);
> >
> >         blkptr = disk_cache_block_ref (blkno);
> > +       journal_get_write_access (txn, blkno);
> >         memcpy (blkptr, buf, block_size);
> >         record_global_poke (blkptr);
> >
> > diff --git a/ext2fs/journal.c b/ext2fs/journal.c
> > index 6828aa4d8..85dde2ad5 100644
> > --- a/ext2fs/journal.c
> > +++ b/ext2fs/journal.c
> > @@ -89,7 +89,7 @@
> >   * Slab Allocator Pool Size.
> >   * Pre-allocates a contiguous chunk of memory for journal buffers
> >   * (512 * 4KB = 2MB).
> > - * This allows journal_dirty_block to be a zero-allocation operation
> for the
> > + * This allows journal_get_write_access to be a zero-allocation
> operation for the
> >   * vast majority of workloads, falling back to dynamic allocation only
> under
> >   * extreme metadata pressure.
> >   */
> > @@ -287,7 +287,7 @@ typedef struct journal
> >    /* Pre-allocated buffers for zero-allocation commits */
> >    void *j_descriptor_buf;
> >    void *j_commit_buf;
> > -  /* Pre-allocated buffers for (near) zero-allocation
> journal_dirty_block */
> > +  /* Pre-allocated buffers for (near) zero-allocation
> journal_get_write_access */
> >    journal_buffer_t *j_pool_memory;   /* The raw contiguous block */
> >    journal_buffer_t *j_free_buffers;  /* The linked list head */
> >
> > @@ -877,9 +877,10 @@ journal_get_oldest_transaction_locked (journal_t
> *journal)
> >   * checkpoint lists AFTER this transaction safely commits.
> >   */
> >  void
> > -journal_record_freed_blocks (block_t start, unsigned long count)
> > +journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
> > +                          unsigned long count)
> >  {
> > -  if (!ext2_journal)
> > +  if (!ext2_journal || !txn)
> >      return;
> >
> >    journal_freed_extent_t *ext = malloc (sizeof
> (journal_freed_extent_t));
> > @@ -894,11 +895,6 @@ journal_record_freed_blocks (block_t start,
> unsigned long count)
> >    ext->fe_count = count;
> >
> >    JOURNAL_LOCK (ext2_journal);
> > -  /* Record against this thread's transaction when it holds one: the
> free
> > -     belongs to the same RPC, and that transaction may be T_LOCKED while
> > -     j_running_transaction is NULL.  Otherwise use the running one. */
> > -  diskfs_transaction_t *txn = journal_thread_depth > 0
> > -    ? journal_thread_txn : ext2_journal->j_running_transaction;
> >    if (!txn || (txn->t_state != T_RUNNING && txn->t_state != T_LOCKED))
> >      {
> >        JRNL_LOG_DEBUG ("Cannot record freed blocks, no running
> transaction.");
> > @@ -1077,7 +1073,7 @@ journal_stop_transaction_locked (journal_t
> *journal,
> >            */
> >         memcpy (curr->jb_shadow_data, live_cache_ptr, block_size);
> >         /* needs_copy was cleared under the lock when this block was
> > -        * listed.  If the block is modified again, journal_dirty_block
> > +        * listed.  If the block is modified again,
> journal_get_write_access
> >          * sets it back to 1 and a later sweep recopies. */
> >
> >         journal_buffer_t *next = curr->jb_next;
> > @@ -1611,23 +1607,53 @@ journal_wait_on_tid_locked (journal_t *journal,
> uint32_t target_tid)
> >      JOURNAL_WAIT (&journal->j_commit_done, journal);
> >  }
> >
> > +void
> > +journal_mark_dirty (diskfs_transaction_t *txn, block_t fs_blocknr)
> > +{
> > +  if (!ext2_journal || !txn)
> > +    return;
> > +
> > +  JOURNAL_LOCK (ext2_journal);
> > +  journal_buffer_t *jb = journal_map_lookup (&txn->t_buffer_map,
> fs_blocknr);
> > +  if (!jb)
> > +    {
> > +      /* STRICT JBD2 BEHAVIOR:
> > +         If we hit this, the filesystem modified a block without calling
> > +         journal_get_write_access() first! */
> > +      JRNL_LOG_WARN ("mark_dirty called on unreserved block %u! Missing
> get_write_access?",
> > +                     fs_blocknr);
> > +      goto out;
> > +    }
> > +
> > +  /* We don't delete the intercepted chunk here, its the only copy of
> the data
> > +     we have. It won't be used for hydration, or for post commit
> flushing but
> > +     we still need it for now for the journal_store_read(). We will
> instead
> > +     instruct the hydration that this needs a fresh copy. */
> > +  jb->needs_copy = 1;
> > +
> > +  /* Reset the physical write flag so the Checkpoint thread knows to
> > +     overwrite the disk with our pristine shadow buffer later. */
> > +  if (jb->jb_is_written)
> > +    {
> > +      jb->jb_is_written = 0;
> > +      txn->t_outstanding_io++;
> > +    }
> > +
> > +out:
> > +  JOURNAL_UNLOCK (ext2_journal);
> > +}
> > +
> >  /**
> >   * Adds a modified filesystem block to the SPECIFIC transaction handle.
> >   * Defers the actual memory copy until the transaction stops.
> >   */
> >  static error_t
> > -journal_dirty_block_locked (diskfs_transaction_t *txn, block_t
> fs_blocknr)
> > +journal_get_write_access_locked (diskfs_transaction_t *txn, block_t
> fs_blocknr)
> >  {
> >    journal_buffer_t *jb;
> >    journal_buffer_t *new_jb;
> >    error_t err = 0;
> >
> > -  if (!txn)
> > -    {
> > -      JRNL_LOG_DEBUG ("[TRX] Transaction null but block dirty.");
> > -      goto out;
> > -    }
> > -
> >    assert_backtrace (txn->t_state == T_RUNNING || txn->t_state ==
> T_LOCKED);
> >    jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr);
> >
> > @@ -1644,11 +1670,6 @@ journal_dirty_block_locked (diskfs_transaction_t
> *txn, block_t fs_blocknr)
> >         jb->jb_is_written = 0;
> >         txn->t_outstanding_io++;
> >       }
> > -      if (jb->jb_intercepted_data)
> > -     {
> > -       journal_free_intercept_chunk (ext2_journal,
> jb->jb_intercepted_data);
> > -       jb->jb_intercepted_data = NULL;
> > -     }
> >        goto out;
> >      }
> >
> > @@ -1786,17 +1807,61 @@ journal_thread_release (diskfs_transaction_t
> *txn)
> >   * Defers the actual memory copy until the transaction stops.
> >   */
> >  error_t
> > -journal_dirty_block (diskfs_transaction_t *txn, block_t fs_blocknr)
> > +journal_get_write_access (diskfs_transaction_t *txn, block_t fs_blocknr)
> >  {
> > -  error_t err;
> > +  error_t err = 0;
> >    if (!ext2_journal)
> > -    return EINVAL;
> > +    goto out;
> > +
> > +  if (!txn)
> > +    {
> > +      JRNL_LOG_WARN
> > +     ("Transaction null so no write access granted for block %u.",
> > +      fs_blocknr);
> > +      goto out;
> > +    }
> > +
> >    JOURNAL_LOCK (ext2_journal);
> > -  err = journal_dirty_block_locked (txn, fs_blocknr);
> > +  err = journal_get_write_access_locked (txn, fs_blocknr);
> > +  if (err)
> > +    JRNL_LOG_WARN
> > +      ("Didn't manage to get journal write access for block %u. (%s).",
> > +       fs_blocknr, strerror (err));
> >    JOURNAL_UNLOCK (ext2_journal);
> > +
> > +out:
> >    return err;
> >  }
> >
> > +/* Return the transaction of the calling thread's open handle.
> > +
> > +   While the journal is live the caller must hold a handle, taken with
> > +   diskfs_journal_start_transaction; calling without one is a bug and
> > +   asserts.  The result is then never NULL, and the handle's
> > +   t_active_threads count keeps it T_RUNNING or T_LOCKED until the
> handle
> > +   is released, so blocks may be added to it.
> > +
> > +   NULL is returned only when the thread holds no handle because there
> is
> > +   no journal, or because the journal is shutting down (j_must_exit) and
> > +   diskfs_journal_start_transaction declined to open one.  Callers must
> > +   accept NULL and treat it as "do not journal".
> > +
> > +   Under NDEBUG the missing-handle check is compiled out: a caller that
> > +   forgot its handle gets NULL and its edits silently go unjournaled.
> */
> > +diskfs_transaction_t *
> > +journal_thread_transaction (void)
> > +{
> > +  diskfs_transaction_t *txn = journal_thread_txn;
> > +
> > +  assert_backtrace (!ext2_journal || ext2_journal->j_must_exit
> > +                 || journal_thread_depth > 0);
> > +  /* The handle's t_active_threads count keeps commit from moving the
> > +     transaction past T_LOCKED.  */
> > +  assert_backtrace (!txn || txn->t_state == T_RUNNING
> > +                 || txn->t_state == T_LOCKED);
> > +  return txn;
> > +}
> > +
> >  /* Two flags.  journal_thread_sync tells this thread's RPC tail to
> commit
> >     rather than stop.  txn->sync_needed tells the outermost stop to wake
> >     kjournald when t_active_threads reaches 0, for callers that never
> commit
> > diff --git a/ext2fs/journal.h b/ext2fs/journal.h
> > index ab7ab1e6b..3cc8212b1 100644
> > --- a/ext2fs/journal.h
> > +++ b/ext2fs/journal.h
> > @@ -84,7 +84,7 @@ journal_store_read (block_t start_block, size_t
> length, void **buf,
> >   * Records a range of deleted blocks so they can be unpinned from older
> >   * checkpoint lists AFTER this transaction safely commits.
> >   */
> > -void journal_record_freed_blocks (block_t start, unsigned long count);
> > +void journal_record_freed_blocks (diskfs_transaction_t *txn, block_t
> start, unsigned long count);
> >
> >  /**
> >   * Marks the calling thread as running a pager callback (ON = 1) or done
> > diff --git a/ext2fs/orphan.c b/ext2fs/orphan.c
> > index 56782d39b..24f10e7d9 100644
> > --- a/ext2fs/orphan.c
> > +++ b/ext2fs/orphan.c
> > @@ -43,11 +43,12 @@ diskfs_orphan_add (struct node *np)
> >  {
> >    ino_t inum = np->cache_id;
> >    struct ext2_inode *di;
> > -  diskfs_transaction_t *txn = NULL;
> > +  diskfs_transaction_t *txn;
> >
> >    assert_backtrace (!diskfs_readonly);
> >    assert_backtrace (np->dn_stat.st_nlink == 0);
> >
> > +  /* The orphan list is exclusively an ext3/journaling feature. */
> >    if (!ext2_journal)
> >      return;
> >
> > @@ -59,15 +60,13 @@ diskfs_orphan_add (struct node *np)
> >       2. global_lock: Protects the in-memory superblock modifications.
> >       3. Journal Transaction (txn): Guarantees that the superblock
> pointer and the
> >          inode pointer hit the physical disk as a single, atomic
> operation. */
> > -  txn = diskfs_journal_start_transaction ();
> > +  txn = journal_thread_transaction ();
> >
> >    pthread_mutex_lock (&orphan_lock);
> >
> >    if (diskfs_node_disknode (np)->on_orphan_list)
> >      {
> >        pthread_mutex_unlock (&orphan_lock);
> > -      if (txn)
> > -     diskfs_journal_stop_transaction (txn);
> >        return;
> >      }
> >
> > @@ -80,7 +79,9 @@ diskfs_orphan_add (struct node *np)
> >
> >    di = dino_ref (inum);
> >
> > -  /* Atomically link the new orphan to the head of the on-disk list. */
> > +  /* Reserve the inode block in the journal */
> > +  journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> > +
> >    pthread_spin_lock (&global_lock);
> >    di->i_dtime = sblock->s_last_orphan;
> >    sblock->s_last_orphan = htole32 (inum);
> > @@ -97,16 +98,8 @@ diskfs_orphan_add (struct node *np)
> >      di->i_size_high = 0;
> >    memset (di->i_block, 0, EXT2_N_BLOCKS * sizeof di->i_block[0]);
> >
> > -  if (txn)
> > -    {
> > -      /* Atomically bundle the superblock and the placeholder inode.
> > -         By dirtying both blocks in the same transaction, we guarantee
> that a
> > -         crash cannot leave a severed list chain. */
> > -      memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
> > -      journal_dirty_block (txn, boffs_block (bptr_offs (di)));
> > -      journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
> > -    }
> > -
> > +  /* Let the journal know we are done editing. */
> > +  journal_mark_dirty (txn, boffs_block (bptr_offs (di)));
> >    dino_deref (di);
> >
> >    /* Maintain the in-memory doubly linked list for O(1) removals. */
> > @@ -118,10 +111,8 @@ diskfs_orphan_add (struct node *np)
> >
> >    pthread_mutex_unlock (&orphan_lock);
> >
> > -  if (txn)
> > -    diskfs_journal_stop_transaction (txn);
> > -  else
> > -    diskfs_set_hypermetadata (0, 0);
> > +  /* hyper.c handles the superblock's get_write_access -> memcpy ->
> mark_dirty! */
> > +  diskfs_set_hypermetadata (0, 0);
> >  }
> >
> >  /* Remove inode NP from the orphan list.  NP is locked by the caller. */
> > @@ -129,7 +120,7 @@ void
> >  diskfs_orphan_del (struct node *np)
> >  {
> >    ino_t inum = np->cache_id;
> > -  diskfs_transaction_t *txn = NULL;
> > +  diskfs_transaction_t *txn;
> >    int update_super = 0;
> >
> >    if (!ext2_journal)
> > @@ -138,15 +129,13 @@ diskfs_orphan_del (struct node *np)
> >    if (!diskfs_node_disknode (np)->on_orphan_list)
> >      return;
> >
> > -  txn = diskfs_journal_start_transaction ();
> > +  txn = journal_thread_transaction ();
> >
> >    pthread_mutex_lock (&orphan_lock);
> >
> >    if (!diskfs_node_disknode (np)->on_orphan_list)
> >      {
> >        pthread_mutex_unlock (&orphan_lock);
> > -      if (txn)
> > -     diskfs_journal_stop_transaction (txn);
> >        return;
> >      }
> >
> > @@ -154,12 +143,16 @@ diskfs_orphan_del (struct node *np)
> >
> >    struct ext2_inode *my_di = dino_ref (inum);
> >    __u32 my_next = le32toh (my_di->i_dtime);
> > +  block_t blocknr = boffs_block (bptr_offs (my_di));
> >
> >    /* This inode is leaving the list.  i_dtime becomes a normal deletion
> >       stamp in the caller's following write_node (mode is already 0).
> > -     We do not journal_dirty it here: that copy can run before
> write_node
> > -     stores the cleared block map. */
> > +     We let the journal know we are about to modify the associated
> block. */
> > +  journal_get_write_access (txn, blocknr);
> >    my_di->i_dtime = 0;
> > +  /* Done editing. */
> > +  journal_mark_dirty (txn, blocknr);
> > +
> >    dino_deref (my_di);
> >
> >    struct node *prev = diskfs_node_disknode (np)->orphan_prev;
> > @@ -172,14 +165,7 @@ diskfs_orphan_del (struct node *np)
> >        sblock_dirty = 1;
> >        pthread_spin_unlock (&global_lock);
> >
> > -      if (txn)
> > -     {
> > -       memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
> > -       journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
> > -     }
> > -      else
> > -     update_super = 1;
> > -
> > +      update_super = 1;
> >        ram_orphan_head = next;
> >      }
> >    else
> > @@ -188,9 +174,9 @@ diskfs_orphan_del (struct node *np)
> >
> >        /* prev stays on the list, so its cached i_block[] is already the
> >           placeholder (zeros).  Only i_dtime changes. */
> > +      journal_get_write_access (txn, boffs_block (bptr_offs (prev_di)));
> >        prev_di->i_dtime = htole32 (my_next);
> > -      if (txn)
> > -     journal_dirty_block (txn, boffs_block (bptr_offs (prev_di)));
> > +      journal_mark_dirty (txn, boffs_block (bptr_offs (prev_di)));
> >        dino_deref (prev_di);
> >
> >        diskfs_node_disknode (prev)->orphan_next = next;
> > @@ -205,10 +191,9 @@ diskfs_orphan_del (struct node *np)
> >
> >    pthread_mutex_unlock (&orphan_lock);
> >
> > +  /* Bundle the superblock modification into the transaction if needed
> */
> >    if (update_super)
> >      diskfs_set_hypermetadata (0, 0);
> > -  else
> > -    diskfs_journal_stop_transaction (txn);
> >  }
> >
> >  /* Recover (clean up) the orphan list at mount time.
> > @@ -229,6 +214,7 @@ ext2_recover_orphan_list (void)
> >    int count = 0;
> >    struct ext2_inode *di;
> >    struct node *np = NULL;
> > +  diskfs_transaction_t *txn;
> >    error_t err;
> >    __u32 max_inodes = le32toh (sblock->s_inodes_count);
> >
> > @@ -270,9 +256,13 @@ ext2_recover_orphan_list (void)
> >        next_orphan = le32toh (di->i_dtime);
> >        dino_deref (di);
> >
> > +      /* One handle per orphan, taken before the node lock like an
> RPC's:
> > +      orphan_del and the drop in diskfs_nput edit metadata in it.  */
> > +      txn = diskfs_journal_start_transaction ();
> >        err = diskfs_cached_lookup (inum, &np);
> >        if (err || !np)
> >       {
> > +       diskfs_journal_stop_transaction (txn);
> >         ext2_warning ("cannot look up orphan inode %lu: %s",
> >                       (unsigned long) inum,
> >                       err ? strerror (err) : "not found");
> > @@ -298,6 +288,7 @@ ext2_recover_orphan_list (void)
> >                       (unsigned long) inum);
> >         diskfs_orphan_del (np);
> >         diskfs_nput (np);
> > +       diskfs_journal_stop_transaction (txn);
> >         continue;
> >       }
> >
> > @@ -305,6 +296,7 @@ ext2_recover_orphan_list (void)
> >           truncate the file and call diskfs_orphan_del (np), which
> >           advances s_last_orphan and clears the head.  */
> >        diskfs_nput (np);
> > +      diskfs_journal_stop_transaction (txn);
> >        count++;
> >      }
> >
> > diff --git a/ext2fs/truncate.c b/ext2fs/truncate.c
> > index 3dd9f8dfb..9c458f46e 100644
> > --- a/ext2fs/truncate.c
> > +++ b/ext2fs/truncate.c
> > @@ -120,6 +120,8 @@ trunc_indirect (struct node *node, block_t end,
> >               void (*free_block)(block_t *p, unsigned index),
> >               struct free_block_run *fbr)
> >  {
> > +  diskfs_transaction_t *txn = journal_thread_transaction ();
> > +
> >    if (*p)
> >      {
> >        unsigned index;
> > @@ -127,6 +129,7 @@ trunc_indirect (struct node *node, block_t end,
> >        block_t *ind_bh = (block_t *) disk_cache_block_ref (*p);
> >        unsigned first = end < offset ? 0 : end - offset;
> >
> > +      journal_get_write_access (txn, *p);
> >        for (index = first; index < addr_per_block; index++)
> >       if (ind_bh[index])
> >         {
> > @@ -139,6 +142,10 @@ trunc_indirect (struct node *node, block_t end,
> >
> >        if (first == 0 && all_freed)
> >       {
> > +       /* We modified this block before killing it.
> > +          Use *p (the block number), not ind_bh (the RAM pointer). */
> > +       if (modified && ext2_journal)
> > +         journal_mark_dirty (txn, *p);
> >         pager_flush_some (diskfs_disk_pager,
> >                           bptr_index (ind_bh) << log2_block_size,
> >                           block_size, 1);
> > diff --git a/ext2fs/xattr.c b/ext2fs/xattr.c
> > index a1e02368f..1d007ef11 100644
> > --- a/ext2fs/xattr.c
> > +++ b/ext2fs/xattr.c
> > @@ -436,6 +436,7 @@ ext2_free_xattr_block (struct node *np)
> >    void *block;
> >    struct ext2_inode *ei;
> >    struct ext2_xattr_header *header;
> > +  diskfs_transaction_t *txn;
> >
> >    if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
> >      {
> > @@ -443,6 +444,7 @@ ext2_free_xattr_block (struct node *np)
> >        return EOPNOTSUPP;
> >      }
> >
> > +  txn = journal_thread_transaction ();
> >    err = 0;
> >    block = NULL;
> >
> > @@ -484,11 +486,13 @@ ext2_free_xattr_block (struct node *np)
> >      {
> >         ext2_debug("h_refcount: %d", le32toh (header->h_refcount));
> >
> > +       journal_get_write_access (txn, blkno);
> >         header->h_refcount = htole32 (le32toh (header->h_refcount) - 1);
> >         record_global_poke (block);
> >      }
> >
> >
> > +  journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
> >    ei->i_file_acl = 0;
> >    record_global_poke (ei);
> >
> > @@ -680,6 +684,7 @@ ext2_set_xattr (struct node *np, const char *name,
> const char *value,
> >    struct ext2_xattr_header *header;
> >    struct ext2_xattr_entry *entry;
> >    struct ext2_xattr_entry *location;
> > +  diskfs_transaction_t *txn;
> >
> >    if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
> >      {
> > @@ -692,6 +697,7 @@ ext2_set_xattr (struct node *np, const char *name,
> const char *value,
> >
> >    if (strlen(name) > 255 || len > block_size)
> >      return ERANGE;
> > +  txn = journal_thread_transaction ();
> >
> >    ei = dino_ref (np->cache_id);
> >    blkno = ei->i_file_acl;
> > @@ -722,6 +728,7 @@ ext2_set_xattr (struct node *np, const char *name,
> const char *value,
> >       }
> >
> >        block = disk_cache_block_ref (blkno);
> > +      journal_get_write_access (txn, blkno);
> >        memset (block, 0, block_size);
> >
> >        header = EXT2_XATTR_HEADER (block);
> > @@ -739,6 +746,7 @@ ext2_set_xattr (struct node *np, const char *name,
> const char *value,
> >         err = EIO;
> >         goto cleanup;
> >       }
> > +      journal_get_write_access (txn, blkno);
> >      }
> >
> >    entry = EXT2_XATTR_ENTRY_FIRST (header);
> > @@ -864,6 +872,7 @@ ext2_set_xattr (struct node *np, const char *name,
> const char *value,
> >             np->dn_stat.st_blocks += 1 << log2_stat_blocks_per_fs_block;
> >             np->dn_set_ctime = 1;
> >
> > +           journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
> >             ei->i_file_acl = blkno;
> >             record_global_poke (ei);
> >           }
> > --
> > 2.56.0
> >
> >
>
> --
> Samuel
> +#if defined(__alpha__) && defined(CONFIG_PCI)
> +       /*
> +        * The meaning of life, the universe, and everything. Plus
> +        * this makes the year come out right.
> +        */
> +       year -= 42;
> +#endif
> (From the patch for 1.3.2: (kernel/time.c), submitted by Marcus Meissner)
>

Reply via email to