Applied, thanks!
Milos Nikic, le dim. 04 oct. 2026 13:17:43 -0700, a ecrit:
> Previously journal was notified after the fact (block has already been
> changed).
> This patch changes this, journal is now notified ahead of time that a
> block is about to be altered, and journal is also notified when we are
> done editing.
>
> This helps keep the main filesystem free of torn writes and makes
> our journal behave much more closely to the journal in ext4.
>
> Function names have been changed to reflect ext4 journal functions
> (journal_get_write_access() to pre-notify the journal and
> journal_mark_dirty() to tell the journal we are done modifying.)
>
> journal_record_freed_blocks () is now taking transaction handle instead
> of trying to find out which txn is running inside. This aligns it more
> to the other journaling functions.
>
> A new function, journal_thread_transaction (), returns the transaction of
> the calling thread's open handle. The edit sites use it to get the
> transaction they pass to journal_get_write_access () and
> journal_mark_dirty (). While the journal is live every metadata edit
> must run inside a handle, so the result is never NULL there, and the
> handle keeps the transaction T_RUNNING or T_LOCKED until it is released;
> both are asserted. It returns NULL only when there is no journal, or the
> journal is shutting down and no handle was opened, and callers then skip
> journaling. Unlike the old lookup in journal_record_freed_blocks (), it
> never falls back to j_running_transaction for a thread without a handle.
>
> journal_get_write_access is a silent no-op when the system is
> unjournaled.
> ---
> ext2fs/balloc.c | 10 +++-
> ext2fs/ext2fs.h | 72 ++++++++++++++++++----------
> ext2fs/getblk.c | 8 ++++
> ext2fs/hyper.c | 2 +
> ext2fs/ialloc.c | 11 ++++-
> ext2fs/inode.c | 20 ++++++--
> ext2fs/journal.c | 117 +++++++++++++++++++++++++++++++++++-----------
> ext2fs/journal.h | 2 +-
> ext2fs/orphan.c | 66 ++++++++++++--------------
> ext2fs/truncate.c | 7 +++
> ext2fs/xattr.c | 9 ++++
> 11 files changed, 228 insertions(+), 96 deletions(-)
>
> diff --git a/ext2fs/balloc.c b/ext2fs/balloc.c
> index cc00fe4cc..6b06a3688 100644
> --- a/ext2fs/balloc.c
> +++ b/ext2fs/balloc.c
> @@ -63,11 +63,14 @@ ext2_free_blocks (block_t block, unsigned long count)
> unsigned long bit;
> unsigned long i;
> struct ext2_group_desc *gdp;
> + diskfs_transaction_t *txn;
>
> /* Trap trying to free superblock, block group descriptor table, or beyond
> the end */
> assert_backtrace (block >= group_desc_block_end
> && block + count <= store->size >> log2_block_size);
>
> + txn = journal_thread_transaction ();
> +
> pthread_spin_lock (&global_lock);
>
> if (block < le32toh (sblock->s_first_data_block) ||
> @@ -103,6 +106,8 @@ ext2_free_blocks (block_t block, unsigned long count)
> block, count);
> }
> gdp = group_desc (block_group);
> + journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
> + journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
> bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap));
>
> if (in_range (le32toh (gdp->bg_block_bitmap), block, gcount) ||
> @@ -113,7 +118,7 @@ ext2_free_blocks (block_t block, unsigned long count)
> "block = %u, count = %lu",
> block, count);
>
> - journal_record_freed_blocks (block, gcount);
> + journal_record_freed_blocks (txn, block, gcount);
> for (i = 0; i < gcount; i++)
> {
> if (!clear_bit (bit + i, bh))
> @@ -160,6 +165,7 @@ ext2_new_block (block_t goal,
> uint32_t lmap;
> struct ext2_group_desc *gdp;
>
> + diskfs_transaction_t *txn = journal_thread_transaction ();
> #ifdef EXT2FS_DEBUG
> static int goal_hits = 0, goal_attempts = 0;
> #endif
> @@ -317,6 +323,8 @@ search_back:
>
> got_block:
> assert_backtrace (bh != NULL);
> + journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap));
> + journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
>
> ext2_debug ("using block group %d (%d)", i, le16toh
> (gdp->bg_free_blocks_count));
>
> diff --git a/ext2fs/ext2fs.h b/ext2fs/ext2fs.h
> index 9282e5417..458446984 100644
> --- a/ext2fs/ext2fs.h
> +++ b/ext2fs/ext2fs.h
> @@ -346,12 +346,39 @@ extern struct journal *ext2_journal;
> #define JRNL_LOG_WARN(fmt, ...) ext2_warning ("[JOURNAL] " fmt,
> ##__VA_ARGS__)
>
> /**
> - * Mark dirty: Add a modified filesystem block to the given transaction.
> - * Performs a shadow copy of 'data' into the journal memory.
> + * Notify the journal that the content of the block fs_blocknr is about to
> get
> + * modified by a filesystem operation.
> + * Journal will arm necessary buffers to track the block and it will
> + * automatically copy the memory contained at that block when the
> + * transaction txn is committing and the block's memory has settled.
> + *
> + * If the memory of the block is modified before calling this function such
> + * modifications won't be part of the journal transaction.
> + *
> + * In between the invocation of this function and the commit, journal will be
> + * intercepting pager's writes to this block and instead of to disk they will
> + * be temporarily stored in journal's caches.
> */
> error_t
> -journal_dirty_block (diskfs_transaction_t * txn, block_t fs_blocknr);
> +journal_get_write_access (diskfs_transaction_t * txn, block_t fs_blocknr);
> +
> +/**
> + * Notifies the journal that modifications for this block have been complete;
> + * the block is copied into the transaction again at the next sweep.
> + *
> + * Shouldn't be called if no journal_get_write_access() was called previously
> + * for the same transaction and the same block.
> + */
> +void
> +journal_mark_dirty (diskfs_transaction_t * txn, block_t fs_blocknr);
>
> +/* Return the transaction of the calling thread's open handle. While the
> + journal is live the caller must hold a handle and the result is a
> + T_RUNNING or T_LOCKED transaction that stays usable until the handle is
> + released. NULL means there is nothing to journal into: no journal, or
> + the journal is shutting down and no handle was opened. */
> +diskfs_transaction_t *
> +journal_thread_transaction (void);
>
> void ext2_orphan_drop_ram_link (struct node *np);
>
> @@ -531,17 +558,9 @@ extern void alloc_sync (struct node *np);
>
> #if defined(__USE_EXTERN_INLINES) || defined(EXT2FS_DEFINE_EI)
> EXT2FS_EI void
> -journal_notify_block_changed (block_t block)
> +journal_mark_dirty_current (block_t block)
> {
> - if (!ext2_journal)
> - return;
> -
> - diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> - error_t err = journal_dirty_block (txn, block);
> - if (err)
> - JRNL_LOG_WARN ("Didn't manage to add a dirty block %u to the journal.
> (%s).",
> - block, strerror (err));
> - diskfs_journal_stop_transaction (txn);
> + journal_mark_dirty (journal_thread_transaction (), block);
> }
>
> /* Marks the global block BLOCK as being modified, and returns true if we
> @@ -569,7 +588,7 @@ record_global_poke (void *ptr)
> {
> block_t block = boffs_block (bptr_offs (ptr));
> void *block_ptr = bptr (block);
> - journal_notify_block_changed (block);
> + journal_mark_dirty_current (block);
> ext2_debug ("(%p = %p)", ptr, block_ptr);
> #ifdef EXT2FS_DEBUG
> assert_backtrace (disk_cache_block_is_ref (block));
> @@ -584,7 +603,7 @@ sync_global_ptr (void *ptr, int wait)
> {
> block_t block = boffs_block (bptr_offs (ptr));
> void *block_ptr = bptr (block);
> - journal_notify_block_changed (block);
> + journal_mark_dirty_current (block);
> ext2_debug ("(%p -> %u)", ptr, block);
> global_block_modified (block);
> _disk_cache_block_deref (block_ptr);
> @@ -611,7 +630,7 @@ record_indir_poke (struct node *node, void *ptr)
> {
> block_t block = boffs_block (bptr_offs (ptr));
> void *block_ptr = bptr (block);
> - journal_notify_block_changed (block);
> + journal_mark_dirty_current (block);
> ext2_debug ("(%llu, %p)", node->cache_id, ptr);
> #ifdef EXT2FS_DEBUG
> assert_backtrace (disk_cache_block_is_ref (block));
> @@ -630,8 +649,8 @@ sync_global (int wait)
> }
>
> /* Sync all allocation information and node NP if diskfs_synchronous.
> - If journaling is active, we just update memory (wait=0) and let the
> - transaction commit handle durability. */
> + If journaling is active, push the superblock into this transaction;
> + the commit provides durability. */
> EXT2FS_EI void
> alloc_sync (struct node *np)
> {
> @@ -640,15 +659,16 @@ alloc_sync (struct node *np)
> {
> diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
>
> + /* If the superblock was modified in memory, push it to the disk cache
> + now so it gets bundled into this transaction's WAL commit.
> + diskfs_set_hypermetadata natively handles the get_write_access ->
> + memcpy -> mark_dirty pipeline! */
> + if (sblock_dirty && ext2_journal)
> + diskfs_set_hypermetadata (0, 0);
> +
> if (np)
> diskfs_node_update (np, diskfs_synchronous);
>
> - if (sblock_dirty && ext2_journal)
> - {
> - block_t sb_blocknr = boffs_block (SBLOCK_OFFS);
> - journal_dirty_block (txn, sb_blocknr);
> - }
> -
> diskfs_journal_stop_transaction (txn);
> }
>
> @@ -658,7 +678,9 @@ alloc_sync (struct node *np)
> if (np)
> pokel_sync (&diskfs_node_disknode (np)->indir_pokel, 1);
>
> - diskfs_set_hypermetadata (1, 0);
> + /* Only flush if journaling didn't already update the cache above */
> + if (!ext2_journal)
> + diskfs_set_hypermetadata (1, 0);
> }
> }
> #endif /* Use extern inlines. */
> diff --git a/ext2fs/getblk.c b/ext2fs/getblk.c
> index f2a219437..7fe219f2b 100644
> --- a/ext2fs/getblk.c
> +++ b/ext2fs/getblk.c
> @@ -70,6 +70,7 @@ ext2_alloc_block (struct node *node, block_t goal, int zero)
> static unsigned long alloc_hits = 0, alloc_attempts = 0;
> #endif
> block_t result;
> + diskfs_transaction_t *txn = journal_thread_transaction ();
>
> #ifdef EXT2_PREALLOCATE
> if (diskfs_node_disknode (node)->info.i_prealloc_count &&
> @@ -110,7 +111,9 @@ ext2_alloc_block (struct node *node, block_t goal, int
> zero)
> if (result && zero)
> {
> char *bh = disk_cache_block_ref (result);
> + journal_get_write_access (txn, result);
> memset (bh, 0, block_size);
> + /* record_indir_poke handles journal_mark_dirty internally. */
> record_indir_poke (node, bh);
> }
>
> @@ -196,6 +199,7 @@ block_getblk (struct node *node, block_t block, int nr,
> int create, int zero,
> int i;
> block_t goal = 0;
> block_t *bh = (block_t *)disk_cache_block_ref (block);
> + diskfs_transaction_t *txn;
>
> *result = bh[nr];
> if (*result)
> @@ -209,6 +213,7 @@ block_getblk (struct node *node, block_t block, int nr,
> int create, int zero,
> disk_cache_block_deref (bh);
> return EINVAL;
> }
> + txn = journal_thread_transaction ();
>
> if (diskfs_node_disknode (node)->info.i_next_alloc_block == new_block)
> goal = diskfs_node_disknode (node)->info.i_next_alloc_goal;
> @@ -233,10 +238,12 @@ block_getblk (struct node *node, block_t block, int nr,
> int create, int zero,
> return ENOSPC;
> }
>
> + journal_get_write_access (txn, block);
> bh[nr] = *result;
>
> if (diskfs_synchronous || diskfs_node_disknode (node)->info.i_osync)
> {
> + /* calls journal_mark_dirty internally */
> sync_global_ptr (bh, 1);
> /* We just wrote a new indirect block pointer.
> If this doesn't hit the platter, the file is corrupt. */
> @@ -245,6 +252,7 @@ block_getblk (struct node *node, block_t block, int nr,
> int create, int zero,
> ext2_warning ("indirect block flush failed: %s", strerror (err));
> }
> else
> + /* record_indir_poke handles journal_mark_dirty internally */
> record_indir_poke (node, bh);
>
> diskfs_node_disknode (node)->info.i_next_alloc_block = new_block;
> diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c
> index 35cdb7ed0..60be3e22f 100644
> --- a/ext2fs/hyper.c
> +++ b/ext2fs/hyper.c
> @@ -240,6 +240,8 @@ diskfs_set_hypermetadata (int wait, int clean)
> /* Before writing, set the time of write */
> sblock->s_wtime = htole32 (diskfs_mtime->seconds);
> sblock_dirty = 0;
> + block_t blk = boffs_block (bptr_offs (mapped_sblock));
> + journal_get_write_access (txn, blk);
> memcpy (mapped_sblock, sblock, SBLOCK_SIZE);
> disk_cache_block_ref_ptr (mapped_sblock);
> record_global_poke (mapped_sblock);
> diff --git a/ext2fs/ialloc.c b/ext2fs/ialloc.c
> index 438de8089..cc9a1edf3 100644
> --- a/ext2fs/ialloc.c
> +++ b/ext2fs/ialloc.c
> @@ -59,6 +59,8 @@ diskfs_free_node (struct node *np, mode_t old_mode)
> unsigned long bit;
> struct ext2_group_desc *gdp;
> ino_t inum = np->cache_id;
> + block_t gdp_block, gdp_bitmap_blk;
> + diskfs_transaction_t *txn = journal_thread_transaction ();
>
> assert_backtrace (!diskfs_readonly);
>
> @@ -79,8 +81,12 @@ diskfs_free_node (struct node *np, mode_t old_mode)
> bit = (inum - 1) % le32toh (sblock->s_inodes_per_group);
>
> gdp = group_desc (block_group);
> - bh = disk_cache_block_ref (le32toh (gdp->bg_inode_bitmap));
> + gdp_block = boffs_block (bptr_offs (gdp));
> + gdp_bitmap_blk = le32toh (gdp->bg_inode_bitmap);
> + bh = disk_cache_block_ref (gdp_bitmap_blk);
>
> + journal_get_write_access (txn, gdp_bitmap_blk);
> + journal_get_write_access (txn, gdp_block);
> if (!clear_bit (bit, bh))
> ext2_warning ("bit already cleared for inode %" PRIu64, inum);
> else
> @@ -123,6 +129,7 @@ ext2_alloc_inode (ino_t dir_inum, mode_t mode)
> ino_t inum;
> struct ext2_group_desc *gdp;
> struct ext2_group_desc *tmp;
> + diskfs_transaction_t *txn = journal_thread_transaction ();
>
> pthread_spin_lock (&global_lock);
>
> @@ -228,6 +235,7 @@ repeat:
> find_first_zero_bit ((uint32_t *) bh, le32toh
> (sblock->s_inodes_per_group)))
> < le32toh (sblock->s_inodes_per_group))
> {
> + journal_get_write_access (txn, le32toh (gdp->bg_inode_bitmap));
> if (set_bit (inum, bh))
> {
> ext2_warning ("bit already set for inode %" PRIu64, inum);
> @@ -258,6 +266,7 @@ repeat:
> goto sync_out;
> }
>
> + journal_get_write_access (txn, boffs_block (bptr_offs (gdp)));
> gdp->bg_free_inodes_count = htole16 (le16toh (gdp->bg_free_inodes_count) -
> 1);
> if (S_ISDIR (mode))
> gdp->bg_used_dirs_count = htole16 (le16toh (gdp->bg_used_dirs_count) +
> 1);
> diff --git a/ext2fs/inode.c b/ext2fs/inode.c
> index 00c4bebb4..21256f843 100644
> --- a/ext2fs/inode.c
> +++ b/ext2fs/inode.c
> @@ -408,6 +408,7 @@ write_node (struct node *np)
> error_t err;
> struct stat *st = &np->dn_stat;
> struct ext2_inode *di;
> + diskfs_transaction_t *txn = journal_thread_transaction ();
>
> ext2_debug ("(%llu)", np->cache_id);
>
> @@ -428,6 +429,7 @@ write_node (struct node *np)
>
> di = dino_ref (np->cache_id);
>
> + journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> di->i_generation = htole32 (st->st_gen);
>
> /* We happen to know that the stat mode bits are the same
> @@ -578,13 +580,17 @@ void
> diskfs_write_disknode (struct node *np, int wait)
> {
> error_t err;
> + diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> +
> struct ext2_inode *di = write_node (np);
> if (!di)
> - return;
> + {
> + diskfs_journal_stop_transaction (txn);
> + return;
> + }
>
> if (ext2_journal)
> {
> - diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
> record_global_poke (di);
> if (wait)
> diskfs_journal_set_sync (txn);
> @@ -601,9 +607,7 @@ diskfs_write_disknode (struct node *np, int wait)
> ext2_warning ("device flush failed: %s", strerror (err));
> }
> else
> - {
> - record_global_poke (di);
> - }
> + record_global_poke (di);
> }
>
> /* Set *ST with appropriate values to reflect the current state of the
> @@ -634,6 +638,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> struct protid *cred)
> {
> error_t err;
> + diskfs_transaction_t *txn;
>
> assert_backtrace (!diskfs_readonly);
>
> @@ -641,6 +646,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> if (err)
> return err;
>
> + txn = journal_thread_transaction ();
> /* If xattr is supported for this filesystem, use xattr to store translator
> record, otherwise, use legacy translator record */
> if (EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR)
> @@ -662,6 +668,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> ext2_debug ("Old translator record found, clear it");
>
> /* Clear block for translator going away. */
> + journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> di->i_translator = htole32 (0);
> diskfs_node_disknode (np)->info_i_translator = 0;
> record_global_poke (di);
> @@ -761,6 +768,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> np->dn_stat.st_mode = newmode;
> }
>
> + journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> di->i_translator = htole32 (blkno);
> diskfs_node_disknode (np)->info_i_translator = blkno;
> record_global_poke (di);
> @@ -771,6 +779,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> else if (!namelen && blkno)
> {
> /* Clear block for translator going away. */
> + journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> di->i_translator = htole32 (0);
> diskfs_node_disknode (np)->info_i_translator = 0;
> record_global_poke (di);
> @@ -796,6 +805,7 @@ diskfs_set_translator (struct node *np, const char *name,
> mach_msg_type_number_t
> memcpy (buf + 2, name, namelen);
>
> blkptr = disk_cache_block_ref (blkno);
> + journal_get_write_access (txn, blkno);
> memcpy (blkptr, buf, block_size);
> record_global_poke (blkptr);
>
> diff --git a/ext2fs/journal.c b/ext2fs/journal.c
> index 6828aa4d8..85dde2ad5 100644
> --- a/ext2fs/journal.c
> +++ b/ext2fs/journal.c
> @@ -89,7 +89,7 @@
> * Slab Allocator Pool Size.
> * Pre-allocates a contiguous chunk of memory for journal buffers
> * (512 * 4KB = 2MB).
> - * This allows journal_dirty_block to be a zero-allocation operation for the
> + * This allows journal_get_write_access to be a zero-allocation operation
> for the
> * vast majority of workloads, falling back to dynamic allocation only under
> * extreme metadata pressure.
> */
> @@ -287,7 +287,7 @@ typedef struct journal
> /* Pre-allocated buffers for zero-allocation commits */
> void *j_descriptor_buf;
> void *j_commit_buf;
> - /* Pre-allocated buffers for (near) zero-allocation journal_dirty_block */
> + /* Pre-allocated buffers for (near) zero-allocation
> journal_get_write_access */
> journal_buffer_t *j_pool_memory; /* The raw contiguous block */
> journal_buffer_t *j_free_buffers; /* The linked list head */
>
> @@ -877,9 +877,10 @@ journal_get_oldest_transaction_locked (journal_t
> *journal)
> * checkpoint lists AFTER this transaction safely commits.
> */
> void
> -journal_record_freed_blocks (block_t start, unsigned long count)
> +journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
> + unsigned long count)
> {
> - if (!ext2_journal)
> + if (!ext2_journal || !txn)
> return;
>
> journal_freed_extent_t *ext = malloc (sizeof (journal_freed_extent_t));
> @@ -894,11 +895,6 @@ journal_record_freed_blocks (block_t start, unsigned
> long count)
> ext->fe_count = count;
>
> JOURNAL_LOCK (ext2_journal);
> - /* Record against this thread's transaction when it holds one: the free
> - belongs to the same RPC, and that transaction may be T_LOCKED while
> - j_running_transaction is NULL. Otherwise use the running one. */
> - diskfs_transaction_t *txn = journal_thread_depth > 0
> - ? journal_thread_txn : ext2_journal->j_running_transaction;
> if (!txn || (txn->t_state != T_RUNNING && txn->t_state != T_LOCKED))
> {
> JRNL_LOG_DEBUG ("Cannot record freed blocks, no running transaction.");
> @@ -1077,7 +1073,7 @@ journal_stop_transaction_locked (journal_t *journal,
> */
> memcpy (curr->jb_shadow_data, live_cache_ptr, block_size);
> /* needs_copy was cleared under the lock when this block was
> - * listed. If the block is modified again, journal_dirty_block
> + * listed. If the block is modified again, journal_get_write_access
> * sets it back to 1 and a later sweep recopies. */
>
> journal_buffer_t *next = curr->jb_next;
> @@ -1611,23 +1607,53 @@ journal_wait_on_tid_locked (journal_t *journal,
> uint32_t target_tid)
> JOURNAL_WAIT (&journal->j_commit_done, journal);
> }
>
> +void
> +journal_mark_dirty (diskfs_transaction_t *txn, block_t fs_blocknr)
> +{
> + if (!ext2_journal || !txn)
> + return;
> +
> + JOURNAL_LOCK (ext2_journal);
> + journal_buffer_t *jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr);
> + if (!jb)
> + {
> + /* STRICT JBD2 BEHAVIOR:
> + If we hit this, the filesystem modified a block without calling
> + journal_get_write_access() first! */
> + JRNL_LOG_WARN ("mark_dirty called on unreserved block %u! Missing
> get_write_access?",
> + fs_blocknr);
> + goto out;
> + }
> +
> + /* We don't delete the intercepted chunk here, its the only copy of the
> data
> + we have. It won't be used for hydration, or for post commit flushing but
> + we still need it for now for the journal_store_read(). We will instead
> + instruct the hydration that this needs a fresh copy. */
> + jb->needs_copy = 1;
> +
> + /* Reset the physical write flag so the Checkpoint thread knows to
> + overwrite the disk with our pristine shadow buffer later. */
> + if (jb->jb_is_written)
> + {
> + jb->jb_is_written = 0;
> + txn->t_outstanding_io++;
> + }
> +
> +out:
> + JOURNAL_UNLOCK (ext2_journal);
> +}
> +
> /**
> * Adds a modified filesystem block to the SPECIFIC transaction handle.
> * Defers the actual memory copy until the transaction stops.
> */
> static error_t
> -journal_dirty_block_locked (diskfs_transaction_t *txn, block_t fs_blocknr)
> +journal_get_write_access_locked (diskfs_transaction_t *txn, block_t
> fs_blocknr)
> {
> journal_buffer_t *jb;
> journal_buffer_t *new_jb;
> error_t err = 0;
>
> - if (!txn)
> - {
> - JRNL_LOG_DEBUG ("[TRX] Transaction null but block dirty.");
> - goto out;
> - }
> -
> assert_backtrace (txn->t_state == T_RUNNING || txn->t_state == T_LOCKED);
> jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr);
>
> @@ -1644,11 +1670,6 @@ journal_dirty_block_locked (diskfs_transaction_t *txn,
> block_t fs_blocknr)
> jb->jb_is_written = 0;
> txn->t_outstanding_io++;
> }
> - if (jb->jb_intercepted_data)
> - {
> - journal_free_intercept_chunk (ext2_journal, jb->jb_intercepted_data);
> - jb->jb_intercepted_data = NULL;
> - }
> goto out;
> }
>
> @@ -1786,17 +1807,61 @@ journal_thread_release (diskfs_transaction_t *txn)
> * Defers the actual memory copy until the transaction stops.
> */
> error_t
> -journal_dirty_block (diskfs_transaction_t *txn, block_t fs_blocknr)
> +journal_get_write_access (diskfs_transaction_t *txn, block_t fs_blocknr)
> {
> - error_t err;
> + error_t err = 0;
> if (!ext2_journal)
> - return EINVAL;
> + goto out;
> +
> + if (!txn)
> + {
> + JRNL_LOG_WARN
> + ("Transaction null so no write access granted for block %u.",
> + fs_blocknr);
> + goto out;
> + }
> +
> JOURNAL_LOCK (ext2_journal);
> - err = journal_dirty_block_locked (txn, fs_blocknr);
> + err = journal_get_write_access_locked (txn, fs_blocknr);
> + if (err)
> + JRNL_LOG_WARN
> + ("Didn't manage to get journal write access for block %u. (%s).",
> + fs_blocknr, strerror (err));
> JOURNAL_UNLOCK (ext2_journal);
> +
> +out:
> return err;
> }
>
> +/* Return the transaction of the calling thread's open handle.
> +
> + While the journal is live the caller must hold a handle, taken with
> + diskfs_journal_start_transaction; calling without one is a bug and
> + asserts. The result is then never NULL, and the handle's
> + t_active_threads count keeps it T_RUNNING or T_LOCKED until the handle
> + is released, so blocks may be added to it.
> +
> + NULL is returned only when the thread holds no handle because there is
> + no journal, or because the journal is shutting down (j_must_exit) and
> + diskfs_journal_start_transaction declined to open one. Callers must
> + accept NULL and treat it as "do not journal".
> +
> + Under NDEBUG the missing-handle check is compiled out: a caller that
> + forgot its handle gets NULL and its edits silently go unjournaled. */
> +diskfs_transaction_t *
> +journal_thread_transaction (void)
> +{
> + diskfs_transaction_t *txn = journal_thread_txn;
> +
> + assert_backtrace (!ext2_journal || ext2_journal->j_must_exit
> + || journal_thread_depth > 0);
> + /* The handle's t_active_threads count keeps commit from moving the
> + transaction past T_LOCKED. */
> + assert_backtrace (!txn || txn->t_state == T_RUNNING
> + || txn->t_state == T_LOCKED);
> + return txn;
> +}
> +
> /* Two flags. journal_thread_sync tells this thread's RPC tail to commit
> rather than stop. txn->sync_needed tells the outermost stop to wake
> kjournald when t_active_threads reaches 0, for callers that never commit
> diff --git a/ext2fs/journal.h b/ext2fs/journal.h
> index ab7ab1e6b..3cc8212b1 100644
> --- a/ext2fs/journal.h
> +++ b/ext2fs/journal.h
> @@ -84,7 +84,7 @@ journal_store_read (block_t start_block, size_t length,
> void **buf,
> * Records a range of deleted blocks so they can be unpinned from older
> * checkpoint lists AFTER this transaction safely commits.
> */
> -void journal_record_freed_blocks (block_t start, unsigned long count);
> +void journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
> unsigned long count);
>
> /**
> * Marks the calling thread as running a pager callback (ON = 1) or done
> diff --git a/ext2fs/orphan.c b/ext2fs/orphan.c
> index 56782d39b..24f10e7d9 100644
> --- a/ext2fs/orphan.c
> +++ b/ext2fs/orphan.c
> @@ -43,11 +43,12 @@ diskfs_orphan_add (struct node *np)
> {
> ino_t inum = np->cache_id;
> struct ext2_inode *di;
> - diskfs_transaction_t *txn = NULL;
> + diskfs_transaction_t *txn;
>
> assert_backtrace (!diskfs_readonly);
> assert_backtrace (np->dn_stat.st_nlink == 0);
>
> + /* The orphan list is exclusively an ext3/journaling feature. */
> if (!ext2_journal)
> return;
>
> @@ -59,15 +60,13 @@ diskfs_orphan_add (struct node *np)
> 2. global_lock: Protects the in-memory superblock modifications.
> 3. Journal Transaction (txn): Guarantees that the superblock pointer
> and the
> inode pointer hit the physical disk as a single, atomic operation. */
> - txn = diskfs_journal_start_transaction ();
> + txn = journal_thread_transaction ();
>
> pthread_mutex_lock (&orphan_lock);
>
> if (diskfs_node_disknode (np)->on_orphan_list)
> {
> pthread_mutex_unlock (&orphan_lock);
> - if (txn)
> - diskfs_journal_stop_transaction (txn);
> return;
> }
>
> @@ -80,7 +79,9 @@ diskfs_orphan_add (struct node *np)
>
> di = dino_ref (inum);
>
> - /* Atomically link the new orphan to the head of the on-disk list. */
> + /* Reserve the inode block in the journal */
> + journal_get_write_access (txn, boffs_block (bptr_offs (di)));
> +
> pthread_spin_lock (&global_lock);
> di->i_dtime = sblock->s_last_orphan;
> sblock->s_last_orphan = htole32 (inum);
> @@ -97,16 +98,8 @@ diskfs_orphan_add (struct node *np)
> di->i_size_high = 0;
> memset (di->i_block, 0, EXT2_N_BLOCKS * sizeof di->i_block[0]);
>
> - if (txn)
> - {
> - /* Atomically bundle the superblock and the placeholder inode.
> - By dirtying both blocks in the same transaction, we guarantee that a
> - crash cannot leave a severed list chain. */
> - memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
> - journal_dirty_block (txn, boffs_block (bptr_offs (di)));
> - journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
> - }
> -
> + /* Let the journal know we are done editing. */
> + journal_mark_dirty (txn, boffs_block (bptr_offs (di)));
> dino_deref (di);
>
> /* Maintain the in-memory doubly linked list for O(1) removals. */
> @@ -118,10 +111,8 @@ diskfs_orphan_add (struct node *np)
>
> pthread_mutex_unlock (&orphan_lock);
>
> - if (txn)
> - diskfs_journal_stop_transaction (txn);
> - else
> - diskfs_set_hypermetadata (0, 0);
> + /* hyper.c handles the superblock's get_write_access -> memcpy ->
> mark_dirty! */
> + diskfs_set_hypermetadata (0, 0);
> }
>
> /* Remove inode NP from the orphan list. NP is locked by the caller. */
> @@ -129,7 +120,7 @@ void
> diskfs_orphan_del (struct node *np)
> {
> ino_t inum = np->cache_id;
> - diskfs_transaction_t *txn = NULL;
> + diskfs_transaction_t *txn;
> int update_super = 0;
>
> if (!ext2_journal)
> @@ -138,15 +129,13 @@ diskfs_orphan_del (struct node *np)
> if (!diskfs_node_disknode (np)->on_orphan_list)
> return;
>
> - txn = diskfs_journal_start_transaction ();
> + txn = journal_thread_transaction ();
>
> pthread_mutex_lock (&orphan_lock);
>
> if (!diskfs_node_disknode (np)->on_orphan_list)
> {
> pthread_mutex_unlock (&orphan_lock);
> - if (txn)
> - diskfs_journal_stop_transaction (txn);
> return;
> }
>
> @@ -154,12 +143,16 @@ diskfs_orphan_del (struct node *np)
>
> struct ext2_inode *my_di = dino_ref (inum);
> __u32 my_next = le32toh (my_di->i_dtime);
> + block_t blocknr = boffs_block (bptr_offs (my_di));
>
> /* This inode is leaving the list. i_dtime becomes a normal deletion
> stamp in the caller's following write_node (mode is already 0).
> - We do not journal_dirty it here: that copy can run before write_node
> - stores the cleared block map. */
> + We let the journal know we are about to modify the associated block. */
> + journal_get_write_access (txn, blocknr);
> my_di->i_dtime = 0;
> + /* Done editing. */
> + journal_mark_dirty (txn, blocknr);
> +
> dino_deref (my_di);
>
> struct node *prev = diskfs_node_disknode (np)->orphan_prev;
> @@ -172,14 +165,7 @@ diskfs_orphan_del (struct node *np)
> sblock_dirty = 1;
> pthread_spin_unlock (&global_lock);
>
> - if (txn)
> - {
> - memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE);
> - journal_dirty_block (txn, boffs_block (SBLOCK_OFFS));
> - }
> - else
> - update_super = 1;
> -
> + update_super = 1;
> ram_orphan_head = next;
> }
> else
> @@ -188,9 +174,9 @@ diskfs_orphan_del (struct node *np)
>
> /* prev stays on the list, so its cached i_block[] is already the
> placeholder (zeros). Only i_dtime changes. */
> + journal_get_write_access (txn, boffs_block (bptr_offs (prev_di)));
> prev_di->i_dtime = htole32 (my_next);
> - if (txn)
> - journal_dirty_block (txn, boffs_block (bptr_offs (prev_di)));
> + journal_mark_dirty (txn, boffs_block (bptr_offs (prev_di)));
> dino_deref (prev_di);
>
> diskfs_node_disknode (prev)->orphan_next = next;
> @@ -205,10 +191,9 @@ diskfs_orphan_del (struct node *np)
>
> pthread_mutex_unlock (&orphan_lock);
>
> + /* Bundle the superblock modification into the transaction if needed */
> if (update_super)
> diskfs_set_hypermetadata (0, 0);
> - else
> - diskfs_journal_stop_transaction (txn);
> }
>
> /* Recover (clean up) the orphan list at mount time.
> @@ -229,6 +214,7 @@ ext2_recover_orphan_list (void)
> int count = 0;
> struct ext2_inode *di;
> struct node *np = NULL;
> + diskfs_transaction_t *txn;
> error_t err;
> __u32 max_inodes = le32toh (sblock->s_inodes_count);
>
> @@ -270,9 +256,13 @@ ext2_recover_orphan_list (void)
> next_orphan = le32toh (di->i_dtime);
> dino_deref (di);
>
> + /* One handle per orphan, taken before the node lock like an RPC's:
> + orphan_del and the drop in diskfs_nput edit metadata in it. */
> + txn = diskfs_journal_start_transaction ();
> err = diskfs_cached_lookup (inum, &np);
> if (err || !np)
> {
> + diskfs_journal_stop_transaction (txn);
> ext2_warning ("cannot look up orphan inode %lu: %s",
> (unsigned long) inum,
> err ? strerror (err) : "not found");
> @@ -298,6 +288,7 @@ ext2_recover_orphan_list (void)
> (unsigned long) inum);
> diskfs_orphan_del (np);
> diskfs_nput (np);
> + diskfs_journal_stop_transaction (txn);
> continue;
> }
>
> @@ -305,6 +296,7 @@ ext2_recover_orphan_list (void)
> truncate the file and call diskfs_orphan_del (np), which
> advances s_last_orphan and clears the head. */
> diskfs_nput (np);
> + diskfs_journal_stop_transaction (txn);
> count++;
> }
>
> diff --git a/ext2fs/truncate.c b/ext2fs/truncate.c
> index 3dd9f8dfb..9c458f46e 100644
> --- a/ext2fs/truncate.c
> +++ b/ext2fs/truncate.c
> @@ -120,6 +120,8 @@ trunc_indirect (struct node *node, block_t end,
> void (*free_block)(block_t *p, unsigned index),
> struct free_block_run *fbr)
> {
> + diskfs_transaction_t *txn = journal_thread_transaction ();
> +
> if (*p)
> {
> unsigned index;
> @@ -127,6 +129,7 @@ trunc_indirect (struct node *node, block_t end,
> block_t *ind_bh = (block_t *) disk_cache_block_ref (*p);
> unsigned first = end < offset ? 0 : end - offset;
>
> + journal_get_write_access (txn, *p);
> for (index = first; index < addr_per_block; index++)
> if (ind_bh[index])
> {
> @@ -139,6 +142,10 @@ trunc_indirect (struct node *node, block_t end,
>
> if (first == 0 && all_freed)
> {
> + /* We modified this block before killing it.
> + Use *p (the block number), not ind_bh (the RAM pointer). */
> + if (modified && ext2_journal)
> + journal_mark_dirty (txn, *p);
> pager_flush_some (diskfs_disk_pager,
> bptr_index (ind_bh) << log2_block_size,
> block_size, 1);
> diff --git a/ext2fs/xattr.c b/ext2fs/xattr.c
> index a1e02368f..1d007ef11 100644
> --- a/ext2fs/xattr.c
> +++ b/ext2fs/xattr.c
> @@ -436,6 +436,7 @@ ext2_free_xattr_block (struct node *np)
> void *block;
> struct ext2_inode *ei;
> struct ext2_xattr_header *header;
> + diskfs_transaction_t *txn;
>
> if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
> {
> @@ -443,6 +444,7 @@ ext2_free_xattr_block (struct node *np)
> return EOPNOTSUPP;
> }
>
> + txn = journal_thread_transaction ();
> err = 0;
> block = NULL;
>
> @@ -484,11 +486,13 @@ ext2_free_xattr_block (struct node *np)
> {
> ext2_debug("h_refcount: %d", le32toh (header->h_refcount));
>
> + journal_get_write_access (txn, blkno);
> header->h_refcount = htole32 (le32toh (header->h_refcount) - 1);
> record_global_poke (block);
> }
>
>
> + journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
> ei->i_file_acl = 0;
> record_global_poke (ei);
>
> @@ -680,6 +684,7 @@ ext2_set_xattr (struct node *np, const char *name, const
> char *value,
> struct ext2_xattr_header *header;
> struct ext2_xattr_entry *entry;
> struct ext2_xattr_entry *location;
> + diskfs_transaction_t *txn;
>
> if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR))
> {
> @@ -692,6 +697,7 @@ ext2_set_xattr (struct node *np, const char *name, const
> char *value,
>
> if (strlen(name) > 255 || len > block_size)
> return ERANGE;
> + txn = journal_thread_transaction ();
>
> ei = dino_ref (np->cache_id);
> blkno = ei->i_file_acl;
> @@ -722,6 +728,7 @@ ext2_set_xattr (struct node *np, const char *name, const
> char *value,
> }
>
> block = disk_cache_block_ref (blkno);
> + journal_get_write_access (txn, blkno);
> memset (block, 0, block_size);
>
> header = EXT2_XATTR_HEADER (block);
> @@ -739,6 +746,7 @@ ext2_set_xattr (struct node *np, const char *name, const
> char *value,
> err = EIO;
> goto cleanup;
> }
> + journal_get_write_access (txn, blkno);
> }
>
> entry = EXT2_XATTR_ENTRY_FIRST (header);
> @@ -864,6 +872,7 @@ ext2_set_xattr (struct node *np, const char *name, const
> char *value,
> np->dn_stat.st_blocks += 1 << log2_stat_blocks_per_fs_block;
> np->dn_set_ctime = 1;
>
> + journal_get_write_access (txn, boffs_block (bptr_offs (ei)));
> ei->i_file_acl = blkno;
> record_global_poke (ei);
> }
> --
> 2.56.0
>
>
--
Samuel
+#if defined(__alpha__) && defined(CONFIG_PCI)
+ /*
+ * The meaning of life, the universe, and everything. Plus
+ * this makes the year come out right.
+ */
+ year -= 42;
+#endif
(From the patch for 1.3.2: (kernel/time.c), submitted by Marcus Meissner)