Previously journal was notified after the fact (block has already been
changed).
This patch changes this, journal is now notified ahead of time that a
block is about to be altered, and journal is also notified when we are
done editing.
This helps keep the main filesystem free of torn writes and makes
our journal behave much more closely to the journal in ext4.
Function names have been changed to reflect ext4 journal functions
(journal_get_write_access() to pre-notify the journal and
journal_mark_dirty() to tell the journal we are done modifying.)
journal_record_freed_blocks () is now taking transaction handle instead
of trying to find out which txn is running inside. This aligns it more
to the other journaling functions.
A new function, journal_thread_transaction (), returns the transaction of
the calling thread's open handle. The edit sites use it to get the
transaction they pass to journal_get_write_access () and
journal_mark_dirty (). While the journal is live every metadata edit
must run inside a handle, so the result is never NULL there, and the
handle keeps the transaction T_RUNNING or T_LOCKED until it is released;
both are asserted. It returns NULL only when there is no journal, or the
journal is shutting down and no handle was opened, and callers then skip
journaling. Unlike the old lookup in journal_record_freed_blocks (), it
never falls back to j_running_transaction for a thread without a handle.
journal_get_write_access is a silent no-op when the system is
unjournaled.
---
ext2fs/balloc.c | 10 +++-
ext2fs/ext2fs.h | 72 ++++++++++++++++++----------
ext2fs/getblk.c | 8 ++++
ext2fs/hyper.c | 2 +
ext2fs/ialloc.c | 11 ++++-
ext2fs/inode.c | 20 ++++++--
ext2fs/journal.c | 117 +++++++++++++++++++++++++++++++++++-----------
ext2fs/journal.h | 2 +-
ext2fs/orphan.c | 66 ++++++++++++--------------
ext2fs/truncate.c | 7 +++
ext2fs/xattr.c | 9 ++++
11 files changed, 228 insertions(+), 96 deletions(-)
diff --git a/ext2fs/balloc.c b/ext2fs/balloc.c
index cc00fe4cc..6b06a3688 100644
--- a/ext2fs/balloc.c
+++ b/ext2fs/balloc.c
@@ -63,11 +63,14 @@ ext2_free_blocks (block_t block, unsigned long count)
unsigned long bit;
unsigned long i;
struct ext2_group_desc *gdp;
+ diskfs_transaction_t *txn;
/* Trap trying to free superblock, block group descriptor table, or beyond
the end */
assert_backtrace (block >= group_desc_block_end
&& block + count <= store->size >> log2_block_size);
+ txn = journal_thread_transaction ();
+
pthread_spin_lock (&global_lock);
if (block < le32toh (sblock->s_first_data_block) ||
@@ -103,6 +106,8 @@ ext2_free_blocks (block_t block, unsigned long count)
block, count);
}
gdp = group_desc (block_group);
+ journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
+ journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap));
if (in_range (le32toh (gdp->bg_block_bitmap), block, gcount) ||
@@ -113,7 +118,7 @@ ext2_free_blocks (block_t block, unsigned long count)
"block = %u, count = %lu",
block, count);
- journal_record_freed_blocks (block, gcount);
+ journal_record_freed_blocks (txn, block, gcount);
for (i = 0; i < gcount; i++)
{
if (!clear_bit (bit + i, bh))
@@ -160,6 +165,7 @@ ext2_new_block (block_t goal,
uint32_t lmap;
struct ext2_group_desc *gdp;
+ diskfs_transaction_t *txn = journal_thread_transaction ();
#ifdef EXT2FS_DEBUG
static int goal_hits = 0, goal_attempts = 0;
#endif
@@ -317,6 +323,8 @@ search_back:
got_block:
assert_backtrace (bh != NULL);
+ journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
+ journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
ext2_debug ("using block group %d (%d)", i, le16toh
(gdp->bg_free_blocks_count));
diff --git a/ext2fs/ext2fs.h b/ext2fs/ext2fs.h
index 9282e5417..458446984 100644
--- a/ext2fs/ext2fs.h
+++ b/ext2fs/ext2fs.h
@@ -346,12 +346,39 @@ extern struct journal *ext2_journal;
#define JRNL_LOG_WARN(fmt, ...) ext2_warning ("[JOURNAL] " fmt, ##__VA_ARGS__)
/**
- * Mark dirty: Add a modified filesystem block to the given transaction.
- * Performs a shadow copy of 'data' into the journal memory.
+ * Notify the journal that the content of the block fs_blocknr is about to get
+ * modified by a filesystem operation.
+ * Journal will arm necessary buffers to track the block and it will
+ * automatically copy the memory contained at that block when the
+ * transaction txn is committing and the block's memory has settled.
+ *
+ * If the memory of the block is modified before calling this function such
+ * modifications won't be part of the journal transaction.
+ *
+ * In between the invocation of this function and the commit, journal will be
+ * intercepting pager's writes to this block and instead of to disk they will
+ * be temporarily stored in journal's caches.
*/
error_t
-journal_dirty_block (diskfs_transaction_t * txn, block_t fs_blocknr);
+journal_get_write_access (diskfs_transaction_t * txn, block_t fs_blocknr);
+
+/**
+ * Notifies the journal that modifications for this block have been complete;
+ * the block is copied into the transaction again at the next sweep.
+ *
+ * Shouldn't be called if no journal_get_write_access() was called previously
+ * for the same transaction and the same block.
+ */
+void
+journal_mark_dirty (diskfs_transaction_t * txn, block_t fs_blocknr);
+/* Return the transaction of the calling thread's open handle. While the
+ journal is live the caller must hold a handle and the result is a
+ T_RUNNING or T_LOCKED transaction that stays usable until the handle is
+ released. NULL means there is nothing to journal into: no journal, or
+ the journal is shutting down and no handle was opened. */
+diskfs_transaction_t *
+journal_thread_transaction (void);
void ext2_orphan_drop_ram_link (struct node *np);
@@ -531,17 +558,9 @@ extern void alloc_sync (struct node *np);
#if defined(__USE_EXTERN_INLINES) || defined(EXT2FS_DEFINE_EI)
EXT2FS_EI void
-journal_notify_block_changed (block_t block)
+journal_mark_dirty_current (block_t block)
{
- if (!ext2_journal)
- return;
-
- diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
- error_t err = journal_dirty_block (txn, block);
- if (err)
- JRNL_LOG_WARN ("Didn't manage to add a dirty block %u to the journal.
(%s).",
- block, strerror (err));
- diskfs_journal_stop_transaction (txn);
+ journal_mark_dirty (journal_thread_transaction (), block);
}
/* Marks the global block BLOCK as being modified, and returns true if we
@@ -569,7 +588,7 @@ record_global_poke (void *ptr)
{
block_t block = boffs_block (bptr_offs (ptr));
void *block_ptr = bptr (block);
- journal_notify_block_changed (block);
+ journal_mark_dirty_current (block);
ext2_debug ("(%p = %p)", ptr, block_ptr);
#ifdef EXT2FS_DEBUG
assert_backtrace (disk_cache_block_is_ref (block));
@@ -584,7 +603,7 @@ sync_global_ptr (void *ptr, int wait)
{
block_t block = boffs_block (bptr_offs (ptr));
void *block_ptr = bptr (block);
- journal_notify_block_changed (block);
+ journal_mark_dirty_current (block);
ext2_debug ("(%p -> %u)", ptr, block);
global_block_modified (block);
_disk_cache_block_deref (block_ptr);
@@ -611,7 +630,7 @@ record_indir_poke (struct node *node, void *ptr)
{
block_t block = boffs_block (bptr_offs (ptr));
void *block_ptr = bptr (block);
- journal_notify_block_changed (block);
+ journal_mark_dirty_current (block);
ext2_debug ("(%llu, %p)", node->cache_id, ptr);
#ifdef EXT2FS_DEBUG
assert_backtrace (disk_cache_block_is_ref (block));
@@ -630,8 +649,8 @@ sync_global (int wait)
}
/* Sync all allocation information and node NP if diskfs_synchronous.
- If journaling is active, we just update memory (wait=0) and let the
- transaction commit handle durability. */
+ If journaling is active, push the superblock into this transaction;
+ the commit provides durability. */
EXT2FS_EI void
alloc_sync (struct node *np)
{
@@ -640,15 +659,16 @@ alloc_sync (struct node *np)
{
diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
+ /* If the superblock was modified in memory, push it to the disk cache
+ now so it gets bundled into this transaction's WAL commit.
+ diskfs_set_hypermetadata natively handles the get_write_access ->
+ memcpy -> mark_dirty pipeline! */
+ if (sblock_dirty && ext2_journal)
+ diskfs_set_hypermetadata (0, 0);
+
if (np)
diskfs_node_update (np, diskfs_synchronous);
- if (sblock_dirty && ext2_journal)
- {
- block_t sb_blocknr = boffs_block (SBLOCK_OFFS);
- journal_dirty_block (txn, sb_blocknr);
- }
-
diskfs_journal_stop_transaction (txn);
}
@@ -658,7 +678,9 @@ alloc_sync (struct node *np)
if (np)
pokel_sync (&diskfs_node_disknode (np)->indir_pokel, 1);
- diskfs_set_hypermetadata (1, 0);
+ /* Only flush if journaling didn't already update the cache above */
+ if (!ext2_journal)
+ diskfs_set_hypermetadata (1, 0);
}
}
#endif /* Use extern inlines. */
diff --git a/ext2fs/getblk.c b/ext2fs/getblk.c
index f2a219437..7fe219f2b 100644
--- a/ext2fs/getblk.c
+++ b/ext2fs/getblk.c
@@ -70,6 +70,7 @@ ext2_alloc_block (struct node *node, block_t goal, int zero)
static unsigned long alloc_hits = 0, alloc_attempts = 0;
#endif
block_t result;
+ diskfs_transaction_t *txn = journal_thread_transaction ();
#ifdef EXT2_PREALLOCATE
if (diskfs_node_disknode (node)->info.i_prealloc_count &&
@@ -110,7 +111,9 @@ ext2_alloc_block (struct node *node, block_t goal, int zero)
if (result && zero)
{
char *bh = disk_cache_block_ref (result);
+ journal_get_write_access (txn, result);
memset (bh, 0, block_size);
+ /* record_indir_poke handles journal_mark_dirty internally. */
record_indir_poke (node, bh);
}
@@ -196,6 +199,7 @@ block_getblk (struct node *node, block_t block, int nr, int
create, int zero,
int i;
block_t goal = 0;
block_t *bh = (block_t *)disk_cache_block_ref (block);
+ diskfs_transaction_t *txn;
*result = bh[nr];
if (*result)
@@ -209,6 +213,7 @@ block_getblk (struct node *node, block_t block, int nr, int
create, int zero,
disk_cache_block_deref (bh);
return EINVAL;
}
+ txn = journal_thread_transaction ();
if (diskfs_node_disknode (node)->info.i_next_alloc_block == new_block)
goal = diskfs_node_disknode (node)->info.i_next_alloc_goal;
@@ -233,10 +238,12 @@ block_getblk (struct node *node, block_t block, int nr,
int create, int zero,
return ENOSPC;
}
+ journal_get_write_access (txn, block);
bh[nr] = *result;
if (diskfs_synchronous || diskfs_node_disknode (node)->info.i_osync)
{
+ /* calls journal_mark_dirty internally */
sync_global_ptr (bh, 1);
/* We just wrote a new indirect block pointer.
If this doesn't hit the platter, the file is corrupt. */
@@ -245,6 +252,7 @@ block_getblk (struct node *node, block_t block, int nr, int
create, int zero,
ext2_warning ("indirect block flush failed: %s", strerror (err));
}
else
+ /* record_indir_poke handles journal_mark_dirty internally */
record_indir_poke (node, bh);
diskfs_node_disknode (node)->info.i_next_alloc_block = new_block;
diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c
index 35cdb7ed0..60be3e22f 100644
--- a/ext2fs/hyper.c
+++ b/ext2fs/hyper.c
@@ -240,6 +240,8 @@ diskfs_set_hypermetadata (int wait, int clean)
/* Before writing, set the time of write */
sblock->s_wtime = htole32 (diskfs_mtime->seconds);
sblock_dirty = 0;
+ block_t blk = boffs_block (bptr_offs (mapped_sblock));
+ journal_get_write_access (txn, blk);
memcpy (mapped_sblock, sblock, SBLOCK_SIZE);
disk_cache_block_ref_ptr (mapped_sblock);
record_global_poke (mapped_sblock);
diff --git a/ext2fs/ialloc.c b/ext2fs/ialloc.c
index 438de8089..cc9a1edf3 100644
--- a/ext2fs/ialloc.c
+++ b/ext2fs/ialloc.c
@@ -59,6 +59,8 @@ diskfs_free_node (struct node *np, mode_t old_mode)
unsigned long bit;
struct ext2_group_desc *gdp;
ino_t inum = np->cache_id;
+ block_t gdp_block, gdp_bitmap_blk;
+ diskfs_transaction_t *txn = journal_thread_transaction ();
assert_backtrace (!diskfs_readonly);
@@ -79,8 +81,12 @@ diskfs_free_node (struct node *np, mode_t old_mode)
bit = (inum - 1) % le32toh (sblock->s_inodes_per_group);
gdp = group_desc (block_group);
- bh = disk_cache_block_ref (le32toh (gdp->bg_inode_bitmap));
+ gdp_block = boffs_block (bptr_offs (gdp));
+ gdp_bitmap_blk = le32toh (gdp->bg_inode_bitmap);
+ bh = disk_cache_block_ref (gdp_bitmap_blk);
+ journal_get_write_access (txn, gdp_bitmap_blk);
+ journal_get_write_access (txn, gdp_block);
if (!clear_bit (bit, bh))
ext2_warning ("bit already cleared for inode %" PRIu64, inum);
else
@@ -123,6 +129,7 @@ ext2_alloc_inode (ino_t dir_inum, mode_t mode)
ino_t inum;
struct ext2_group_desc *gdp;
struct ext2_group_desc *tmp;
+ diskfs_transaction_t *txn = journal_thread_transaction ();
pthread_spin_lock (&global_lock);
@@ -228,6 +235,7 @@ repeat:
find_first_zero_bit ((uint32_t *) bh, le32toh
(sblock->s_inodes_per_group)))
< le32toh (sblock->s_inodes_per_group))
{
+ journal_get_write_access (txn, le32toh (gdp->bg_inode_bitmap));
if (set_bit (inum, bh))
{
ext2_warning ("bit already set for inode %" PRIu64, inum);
@@ -258,6 +266,7 @@ repeat:
goto sync_out;
}
+ journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
gdp->bg_free_inodes_count = htole16 (le16toh (gdp->bg_free_inodes_count) -
1);
if (S_ISDIR (mode))
gdp->bg_used_dirs_count = htole16 (le16toh (gdp->bg_used_dirs_count) + 1);
diff --git a/ext2fs/inode.c b/ext2fs/inode.c
index 00c4bebb4..21256f843 100644
--- a/ext2fs/inode.c
+++ b/ext2fs/inode.c
@@ -408,6 +408,7 @@ write_node (struct node *np)
error_t err;
struct stat *st = &np->dn_stat;
struct ext2_inode *di;
+ diskfs_transaction_t *txn = journal_thread_transaction ();
ext2_debug ("(%llu)", np->cache_id);
@@ -428,6 +429,7 @@ write_node (struct node *np)
di = dino_ref (np->cache_id);
+ journal_get_write_access (txn, boffs_block (bptr_offs (di)));
di->i_generation = htole32 (st->st_gen);
/* We happen to know that the stat mode bits are the same
@@ -578,13 +580,17 @@ void
diskfs_write_disknode (struct node *np, int wait)
{
error_t err;
+ diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
+
struct ext2_inode *di = write_node (np);
if (!di)
- return;
+ {
+ diskfs_journal_stop_transaction (txn);
+ return;
+ }
if (ext2_journal)
{
- diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
record_global_poke (di);
if (wait)
diskfs_journal_set_sync (txn);
@@ -601,9 +607,7 @@ diskfs_write_disknode (struct node *np, int wait)
ext2_warning ("device flush failed: %s", strerror (err));
}
else
- {
- record_global_poke (di);
- }
+ record_global_poke (di);
}
/* Set *ST with appropriate values to reflect the current state of the
@@ -634,6 +638,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
struct protid *cred)
{
error_t err;
+ diskfs_transaction_t *txn;
assert_backtrace (!diskfs_readonly);
@@ -641,6 +646,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
if (err)
return err;
+ txn = journal_thread_transaction ();
/* If xattr is supported for this filesystem, use xattr to store translator
record, otherwise, use legacy translator record */
if (EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR)
@@ -662,6 +668,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
ext2_debug ("Old translator record found, clear it");
/* Clear block for translator going away. */
+ journal_get_write_access (txn, boffs_block (bptr_offs (di)));
di->i_translator = htole32 (0);
diskfs_node_disknode (np)->info_i_translator = 0;
record_global_poke (di);
@@ -761,6 +768,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
np->dn_stat.st_mode = newmode;
}
+ journal_get_write_access (txn, boffs_block (bptr_offs (di)));
di->i_translator = htole32 (blkno);
diskfs_node_disknode (np)->info_i_translator = blkno;
record_global_poke (di);
@@ -771,6 +779,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
else if (!namelen && blkno)
{
/* Clear block for translator going away. */
+ journal_get_write_access (txn, boffs_block (bptr_offs (di)));
di->i_translator = htole32 (0);
diskfs_node_disknode (np)->info_i_translator = 0;
record_global_poke (di);
@@ -796,6 +805,7 @@ diskfs_set_translator (struct node *np, const char *name,
mach_msg_type_number_t
memcpy (buf + 2, name, namelen);
blkptr = disk_cache_block_ref (blkno);
+ journal_get_write_access (txn, blkno);
memcpy (blkptr, buf, block_size);
record_global_poke (blkptr);
diff --git a/ext2fs/journal.c b/ext2fs/journal.c
index 6828aa4d8..85dde2ad5 100644
--- a/ext2fs/journal.c
+++ b/ext2fs/journal.c
@@ -89,7 +89,7 @@
* Slab Allocator Pool Size.
* Pre-allocates a contiguous chunk of memory for journal buffers
* (512 * 4KB = 2MB).
- * This allows journal_dirty_block to be a zero-allocation operation for the
+ * This allows journal_get_write_access to be a zero-allocation operation for
the
* vast majority of workloads, falling back to dynamic allocation only under
* extreme metadata pressure.
*/
@@ -287,7 +287,7 @@ typedef struct journal
/* Pre-allocated buffers for zero-allocation commits */
void *j_descriptor_buf;
void *j_commit_buf;
- /* Pre-allocated buffers for (near) zero-allocation journal_dirty_block */
+ /* Pre-allocated buffers for (near) zero-allocation journal_get_write_access
*/
journal_buffer_t *j_pool_memory; /* The raw contiguous block */
journal_buffer_t *j_free_buffers; /* The linked list head */
@@ -877,9 +877,10 @@ journal_get_oldest_transaction_locked (journal_t *journal)
* checkpoint lists AFTER this transaction safely commits.
*/
void
-journal_record_freed_blocks (block_t start, unsigned long count)
+journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
+ unsigned long count)
{
- if (!ext2_journal)
+ if (!ext2_journal || !txn)
return;
journal_freed_extent_t *ext = malloc (sizeof (journal_freed_extent_t));
@@ -894,11 +895,6 @@ journal_record_freed_blocks (block_t start, unsigned long
count)
ext->fe_count = count;
JOURNAL_LOCK (ext2_journal);
- /* Record against this thread's transaction when it holds one: the free
- belongs to the same RPC, and that transaction may be T_LOCKED while
- j_running_transaction is NULL. Otherwise use the running one. */
- diskfs_transaction_t *txn = journal_thread_depth > 0
- ? journal_thread_txn : ext2_journal->j_running_transaction;
if (!txn || (txn->t_state != T_RUNNING && txn->t_state != T_LOCKED))
{
JRNL_LOG_DEBUG ("Cannot record freed blocks, no running transaction.");
@@ -1077,7 +1073,7 @@ journal_stop_transaction_locked (journal_t *journal,
*/
memcpy (curr->jb_shadow_data, live_cache_ptr, block_size);
/* needs_copy was cleared under the lock when this block was
- * listed. If the block is modified again, journal_dirty_block
+ * listed. If the block is modified again, journal_get_write_access
* sets it back to 1 and a later sweep recopies. */
journal_buffer_t *next = curr->jb_next;
@@ -1611,23 +1607,53 @@ journal_wait_on_tid_locked (journal_t *journal,
uint32_t target_tid)
JOURNAL_WAIT (&journal->j_commit_done, journal);
}
+void
+journal_mark_dirty (diskfs_transaction_t *txn, block_t fs_blocknr)
+{
+ if (!ext2_journal || !txn)
+ return;
+
+ JOURNAL_LOCK (ext2_journal);
+ journal_buffer_t *jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr);
+ if (!jb)
+ {
+ /* STRICT JBD2 BEHAVIOR:
+ If we hit this, the filesystem modified a block without calling
+ journal_get_write_access() first! */
+ JRNL_LOG_WARN ("mark_dirty called on unreserved block %u! Missing
get_write_access?",
+ fs_blocknr);
+ goto out;
+ }
+
+ /* We don't delete the intercepted chunk here, its the only copy of the data
+ we have. It won't be used for hydration, or for post commit flushing but
+ we still need it for now for the journal_store_read(). We will instead
+ instruct the hydration that this needs a fresh copy. */
+ jb->needs_copy = 1;
+
+ /* Reset the physical write flag so the Checkpoint thread knows to
+ overwrite the disk with our pristine shadow buffer later. */
+ if (jb->jb_is_written)
+ {
+ jb->jb_is_written = 0;
+ txn->t_outstanding_io++;
+ }
+
+out:
+ JOURNAL_UNLOCK (ext2_journal);
+}
+
/**
* Adds a modified filesystem block to the SPECIFIC transaction handle.
* Defers the actual memory copy until the transaction stops.
*/
static error_t
-journal_dirty_block_locked (diskfs_transaction_t *txn, block_t fs_blocknr)
+journal_get_write_access_locked (diskfs_transaction_t *txn, block_t fs_blocknr)
{
journal_buffer_t *jb;
journal_buffer_t *new_jb;
error_t err = 0;
- if (!txn)
- {
- JRNL_LOG_DEBUG ("[TRX] Transaction null but block dirty.");
- goto out;
- }
-
assert_backtrace (txn->t_state == T_RUNNING || txn->t_state == T_LOCKED);
jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr);
@@ -1644,11 +1670,6 @@ journal_dirty_block_locked (diskfs_transaction_t *txn,
block_t fs_blocknr)
jb->jb_is_written = 0;
txn->t_outstanding_io++;
}
- if (jb->jb_intercepted_data)
- {
- journal_free_intercept_chunk (ext2_journal, jb->jb_intercepted_data);
- jb->jb_intercepted_data = NULL;
- }
goto out;
}
@@ -1786,17 +1807,61 @@ journal_thread_release (diskfs_transaction_t *txn)
* Defers the actual memory copy until the transaction stops.
*/
error_t
-journal_dirty_block (diskfs_transaction_t *txn, block_t fs_blocknr)
+journal_get_write_access (diskfs_transaction_t *txn, block_t fs_blocknr)
{
- error_t err;
+ error_t err = 0;
if (!ext2_journal)
- return EINVAL;
+ goto out;
+
+ if (!txn)
+ {
+ JRNL_LOG_WARN
+ ("Transaction null so no write access granted for block %u.",
+ fs_blocknr);
+ goto out;
+ }
+
JOURNAL_LOCK (ext2_journal);
- err = journal_dirty_block_locked (txn, fs_blocknr);
+ err = journal_get_write_access_locked (txn, fs_blocknr);
+ if (err)
+ JRNL_LOG_WARN
+ ("Didn't manage to get journal write access for block %u. (%s).",
+ fs_blocknr, strerror (err));
JOURNAL_UNLOCK (ext2_journal);
+
+out:
return err;
}
+/* Return the transaction of the calling thread's open handle.
+
+ While the journal is live the caller must hold a handle, taken with
+ diskfs_journal_start_transaction; calling without one is a bug and
+ asserts. The result is then never NULL, and the handle's
+ t_active_threads count keeps it T_RUNNING or T_LOCKED until the handle
+ is released, so blocks may be added to it.
+
+ NULL is returned only when the thread holds no handle because there is
+ no journal, or because the journal is shutting down (j_must_exit) and
+ diskfs_journal_start_transaction declined to open one. Callers must
+ accept NULL and treat it as "do not journal".
+
+ Under NDEBUG the missing-handle check is compiled out: a caller that
+ forgot its handle gets NULL and its edits silently go unjournaled. */
+diskfs_transaction_t *
+journal_thread_transaction (void)
+{
+ diskfs_transaction_t *txn = journal_thread_txn;
+
+ assert_backtrace (!ext2_journal || ext2_journal->j_must_exit
+ || journal_thread_depth > 0);
+ /* The handle's t_active_threads count keeps commit from moving the
+ transaction past T_LOCKED. */
+ assert_backtrace (!txn || txn->t_state == T_RUNNING
+ || txn->t_state == T_LOCKED);
+ return txn;
+}
+
/* Two flags. journal_thread_sync tells this thread's RPC tail to commit
rather than stop. txn->sync_needed tells the outermost stop to wake
kjournald when t_active_threads reaches 0, for callers that never commit
diff --git a/ext2fs/journal.h b/ext2fs/journal.h
index ab7ab1e6b..3cc8212b1 100644
--- a/ext2fs/journal.h
+++ b/ext2fs/journal.h
@@ -84,7 +84,7 @@ journal_store_read (block_t start_block, size_t length, void
**buf,
* Records a range of deleted blocks so they can be unpinned from older
* checkpoint lists AFTER this transaction safely commits.
*/
-void journal_record_freed_blocks (block_t start, unsigned long count);
+void journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
unsigned long count);
/**
* Marks the calling thread as running a pager callback (ON = 1) or done
diff --git a/ext2fs/orphan.c b/ext2fs/orphan.c
index 56782d39b..24f10e7d9 100644
--- a/ext2fs/orphan.c
+++ b/ext2fs/orphan.c
@@ -43,11 +43,12 @@ diskfs_orphan_add (struct node *np)
{
ino_t inum = np->cache_id;
struct ext2_inode *di;
- diskfs_transaction_t *txn = NULL;
+ diskfs_transaction_t *txn;
assert_backtrace (!diskfs_readonly);
assert_backtrace (np->dn_stat.st_nlink == 0);
+ /* The orphan list is exclusively an ext3/journaling feature. */
if (!ext2_journal)
return;
@@ -59,15 +60,13 @@ diskfs_orphan_add (struct node *np)
2. global_lock: Protects the in-memory superblock modifications.
3. Journal Transaction (txn): Guarantees that the superblock pointer and
the
inode pointer hit the physical disk as a single, atomic operation. */
- txn = diskfs_journal_start_transaction ();
+ txn = journal_thread_transaction ();
pthread_mutex_lock (&orphan_lock);
if (diskfs_node_disknode (np)->on_orphan_list)
{
pthread_mutex_unlock (&orphan_lock);
- if (txn)
- diskfs_journal_stop_transaction (txn);
return;
}
@@ -80,7 +79,9 @@ diskfs_orphan_add (struct node *np)
di = dino_ref (inum);
- /* Atomically link the new orphan to the head of the on-disk list. */
+ /* Reserve the inode block in the journal */
+ journal_get_write_access (txn, boffs_block (bptr_offs (di)));
+
pthread_spin_lock (&global_lock);
di->i_dtime = sblock->s_last_orphan;
sblock->s_last_orphan = htole32 (inum);
@@ -97,16 +98,8 @@ diskfs_orphan_add (struct node *np)
di->i_size_high = 0;
memset (di->i_block, 0, EXT2_N_BLOCKS * sizeof di->i_block[0]);
- if (txn)
- {
- /* Atomically bundle the superblock and the placeholder inode.
- By dirtying both blocks in the same transaction, we guarantee that a
- crash cannot leave a severed list chain. */
- memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
- journal_dirty_block (txn, boffs_block (bptr_offs (di)));
- journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
- }
-
+ /* Let the journal know we are done editing. */
+ journal_mark_dirty (txn, boffs_block (bptr_offs (di)));
dino_deref (di);
/* Maintain the in-memory doubly linked list for O(1) removals. */
@@ -118,10 +111,8 @@ diskfs_orphan_add (struct node *np)
pthread_mutex_unlock (&orphan_lock);
- if (txn)
- diskfs_journal_stop_transaction (txn);
- else
- diskfs_set_hypermetadata (0, 0);
+ /* hyper.c handles the superblock's get_write_access -> memcpy ->
mark_dirty! */
+ diskfs_set_hypermetadata (0, 0);
}
/* Remove inode NP from the orphan list. NP is locked by the caller. */
@@ -129,7 +120,7 @@ void
diskfs_orphan_del (struct node *np)
{
ino_t inum = np->cache_id;
- diskfs_transaction_t *txn = NULL;
+ diskfs_transaction_t *txn;
int update_super = 0;
if (!ext2_journal)
@@ -138,15 +129,13 @@ diskfs_orphan_del (struct node *np)
if (!diskfs_node_disknode (np)->on_orphan_list)
return;
- txn = diskfs_journal_start_transaction ();
+ txn = journal_thread_transaction ();
pthread_mutex_lock (&orphan_lock);
if (!diskfs_node_disknode (np)->on_orphan_list)
{
pthread_mutex_unlock (&orphan_lock);
- if (txn)
- diskfs_journal_stop_transaction (txn);
return;
}
@@ -154,12 +143,16 @@ diskfs_orphan_del (struct node *np)
struct ext2_inode *my_di = dino_ref (inum);
__u32 my_next = le32toh (my_di->i_dtime);
+ block_t blocknr = boffs_block (bptr_offs (my_di));
/* This inode is leaving the list. i_dtime becomes a normal deletion
stamp in the caller's following write_node (mode is already 0).
- We do not journal_dirty it here: that copy can run before write_node
- stores the cleared block map. */
+ We let the journal know we are about to modify the associated block. */
+ journal_get_write_access (txn, blocknr);
my_di->i_dtime = 0;
+ /* Done editing. */
+ journal_mark_dirty (txn, blocknr);
+
dino_deref (my_di);
struct node *prev = diskfs_node_disknode (np)->orphan_prev;
@@ -172,14 +165,7 @@ diskfs_orphan_del (struct node *np)
sblock_dirty = 1;
pthread_spin_unlock (&global_lock);
- if (txn)
- {
- memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
- journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
- }
- else
- update_super = 1;
-
+ update_super = 1;
ram_orphan_head = next;
}
else
@@ -188,9 +174,9 @@ diskfs_orphan_del (struct node *np)
/* prev stays on the list, so its cached i_block[] is already the
placeholder (zeros). Only i_dtime changes. */
+ journal_get_write_access (txn, boffs_block (bptr_offs (prev_di)));
prev_di->i_dtime = htole32 (my_next);
- if (txn)
- journal_dirty_block (txn, boffs_block (bptr_offs (prev_di)));
+ journal_mark_dirty (txn, boffs_block (bptr_offs (prev_di)));
dino_deref (prev_di);
diskfs_node_disknode (prev)->orphan_next = next;
@@ -205,10 +191,9 @@ diskfs_orphan_del (struct node *np)
pthread_mutex_unlock (&orphan_lock);
+ /* Bundle the superblock modification into the transaction if needed */
if (update_super)
diskfs_set_hypermetadata (0, 0);
- else
- diskfs_journal_stop_transaction (txn);
}
/* Recover (clean up) the orphan list at mount time.
@@ -229,6 +214,7 @@ ext2_recover_orphan_list (void)
int count = 0;
struct ext2_inode *di;
struct node *np = NULL;
+ diskfs_transaction_t *txn;
error_t err;
__u32 max_inodes = le32toh (sblock->s_inodes_count);
@@ -270,9 +256,13 @@ ext2_recover_orphan_list (void)
next_orphan = le32toh (di->i_dtime);
dino_deref (di);
+ /* One handle per orphan, taken before the node lock like an RPC's:
+ orphan_del and the drop in diskfs_nput edit metadata in it. */
+ txn = diskfs_journal_start_transaction ();
err = diskfs_cached_lookup (inum, &np);
if (err || !np)
{
+ diskfs_journal_stop_transaction (txn);
ext2_warning ("cannot look up orphan inode %lu: %s",
(unsigned long) inum,
err ? strerror (err) : "not found");
@@ -298,6 +288,7 @@ ext2_recover_orphan_list (void)
(unsigned long) inum);
diskfs_orphan_del (np);
diskfs_nput (np);
+ diskfs_journal_stop_transaction (txn);
continue;
}
@@ -305,6 +296,7 @@ ext2_recover_orphan_list (void)
truncate the file and call diskfs_orphan_del (np), which
advances s_last_orphan and clears the head. */
diskfs_nput (np);
+ diskfs_journal_stop_transaction (txn);
count++;
}
diff --git a/ext2fs/truncate.c b/ext2fs/truncate.c
index 3dd9f8dfb..9c458f46e 100644
--- a/ext2fs/truncate.c
+++ b/ext2fs/truncate.c
@@ -120,6 +120,8 @@ trunc_indirect (struct node *node, block_t end,
void (*free_block)(block_t *p, unsigned index),
struct free_block_run *fbr)
{
+ diskfs_transaction_t *txn = journal_thread_transaction ();
+
if (*p)
{
unsigned index;
@@ -127,6 +129,7 @@ trunc_indirect (struct node *node, block_t end,
block_t *ind_bh = (block_t *) disk_cache_block_ref (*p);
unsigned first = end < offset ? 0 : end - offset;
+ journal_get_write_access (txn, *p);
for (index = first; index < addr_per_block; index++)
if (ind_bh[index])
{
@@ -139,6 +142,10 @@ trunc_indirect (struct node *node, block_t end,
if (first == 0 && all_freed)
{
+ /* We modified this block before killing it.
+ Use *p (the block number), not ind_bh (the RAM pointer). */
+ if (modified && ext2_journal)
+ journal_mark_dirty (txn, *p);
pager_flush_some (diskfs_disk_pager,
bptr_index (ind_bh) << log2_block_size,
block_size, 1);
diff --git a/ext2fs/xattr.c b/ext2fs/xattr.c
index a1e02368f..1d007ef11 100644
--- a/ext2fs/xattr.c
+++ b/ext2fs/xattr.c
@@ -436,6 +436,7 @@ ext2_free_xattr_block (struct node *np)
void *block;
struct ext2_inode *ei;
struct ext2_xattr_header *header;
+ diskfs_transaction_t *txn;
if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
{
@@ -443,6 +444,7 @@ ext2_free_xattr_block (struct node *np)
return EOPNOTSUPP;
}
+ txn = journal_thread_transaction ();
err = 0;
block = NULL;
@@ -484,11 +486,13 @@ ext2_free_xattr_block (struct node *np)
{
ext2_debug("h_refcount: %d", le32toh (header->h_refcount));
+ journal_get_write_access (txn, blkno);
header->h_refcount = htole32 (le32toh (header->h_refcount) - 1);
record_global_poke (block);
}
+ journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
ei->i_file_acl = 0;
record_global_poke (ei);
@@ -680,6 +684,7 @@ ext2_set_xattr (struct node *np, const char *name, const
char *value,
struct ext2_xattr_header *header;
struct ext2_xattr_entry *entry;
struct ext2_xattr_entry *location;
+ diskfs_transaction_t *txn;
if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
{
@@ -692,6 +697,7 @@ ext2_set_xattr (struct node *np, const char *name, const
char *value,
if (strlen(name) > 255 || len > block_size)
return ERANGE;
+ txn = journal_thread_transaction ();
ei = dino_ref (np->cache_id);
blkno = ei->i_file_acl;
@@ -722,6 +728,7 @@ ext2_set_xattr (struct node *np, const char *name, const
char *value,
}
block = disk_cache_block_ref (blkno);
+ journal_get_write_access (txn, blkno);
memset (block, 0, block_size);
header = EXT2_XATTR_HEADER (block);
@@ -739,6 +746,7 @@ ext2_set_xattr (struct node *np, const char *name, const
char *value,
err = EIO;
goto cleanup;
}
+ journal_get_write_access (txn, blkno);
}
entry = EXT2_XATTR_ENTRY_FIRST (header);
@@ -864,6 +872,7 @@ ext2_set_xattr (struct node *np, const char *name, const
char *value,
np->dn_stat.st_blocks += 1 << log2_stat_blocks_per_fs_block;
np->dn_set_ctime = 1;
+ journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
ei->i_file_acl = blkno;
record_global_poke (ei);
}
--
2.56.0