Hey Samuel. Oh wow the timing. Please apply the v2 instead...or if you want i can send the delta!
On Mon, Oct 5, 2026 at 10:40 AM Samuel Thibault <[email protected]> wrote: > Applied, thanks! > > Milos Nikic, le dim. 04 oct. 2026 13:17:43 -0700, a ecrit: > > Previously journal was notified after the fact (block has already been > > changed). > > This patch changes this, journal is now notified ahead of time that a > > block is about to be altered, and journal is also notified when we are > > done editing. > > > > This helps keep the main filesystem free of torn writes and makes > > our journal behave much more closely to the journal in ext4. > > > > Function names have been changed to reflect ext4 journal functions > > (journal_get_write_access() to pre-notify the journal and > > journal_mark_dirty() to tell the journal we are done modifying.) > > > > journal_record_freed_blocks () is now taking transaction handle instead > > of trying to find out which txn is running inside. This aligns it more > > to the other journaling functions. > > > > A new function, journal_thread_transaction (), returns the transaction of > > the calling thread's open handle. The edit sites use it to get the > > transaction they pass to journal_get_write_access () and > > journal_mark_dirty (). While the journal is live every metadata edit > > must run inside a handle, so the result is never NULL there, and the > > handle keeps the transaction T_RUNNING or T_LOCKED until it is released; > > both are asserted. It returns NULL only when there is no journal, or the > > journal is shutting down and no handle was opened, and callers then skip > > journaling. Unlike the old lookup in journal_record_freed_blocks (), it > > never falls back to j_running_transaction for a thread without a handle. > > > > journal_get_write_access is a silent no-op when the system is > > unjournaled. > > --- > > ext2fs/balloc.c | 10 +++- > > ext2fs/ext2fs.h | 72 ++++++++++++++++++---------- > > ext2fs/getblk.c | 8 ++++ > > ext2fs/hyper.c | 2 + > > ext2fs/ialloc.c | 11 ++++- > > ext2fs/inode.c | 20 ++++++-- > > ext2fs/journal.c | 117 +++++++++++++++++++++++++++++++++++----------- > > ext2fs/journal.h | 2 +- > > ext2fs/orphan.c | 66 ++++++++++++-------------- > > ext2fs/truncate.c | 7 +++ > > ext2fs/xattr.c | 9 ++++ > > 11 files changed, 228 insertions(+), 96 deletions(-) > > > > diff --git a/ext2fs/balloc.c b/ext2fs/balloc.c > > index cc00fe4cc..6b06a3688 100644 > > --- a/ext2fs/balloc.c > > +++ b/ext2fs/balloc.c > > @@ -63,11 +63,14 @@ ext2_free_blocks (block_t block, unsigned long count) > > unsigned long bit; > > unsigned long i; > > struct ext2_group_desc *gdp; > > + diskfs_transaction_t *txn; > > > > /* Trap trying to free superblock, block group descriptor table, or > beyond the end */ > > assert_backtrace (block >= group_desc_block_end > > && block + count <= store->size >> log2_block_size); > > > > + txn = journal_thread_transaction (); > > + > > pthread_spin_lock (&global_lock); > > > > if (block < le32toh (sblock->s_first_data_block) || > > @@ -103,6 +106,8 @@ ext2_free_blocks (block_t block, unsigned long count) > > block, count); > > } > > gdp = group_desc (block_group); > > + journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap)); > > + journal_get_write_access (txn, boffs_block (bptr_offs (gdp))); > > bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap)); > > > > if (in_range (le32toh (gdp->bg_block_bitmap), block, gcount) || > > @@ -113,7 +118,7 @@ ext2_free_blocks (block_t block, unsigned long count) > > "block = %u, count = %lu", > > block, count); > > > > - journal_record_freed_blocks (block, gcount); > > + journal_record_freed_blocks (txn, block, gcount); > > for (i = 0; i < gcount; i++) > > { > > if (!clear_bit (bit + i, bh)) > > @@ -160,6 +165,7 @@ ext2_new_block (block_t goal, > > uint32_t lmap; > > struct ext2_group_desc *gdp; > > > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > #ifdef EXT2FS_DEBUG > > static int goal_hits = 0, goal_attempts = 0; > > #endif > > @@ -317,6 +323,8 @@ search_back: > > > > got_block: > > assert_backtrace (bh != NULL); > > + journal_get_write_access (txn, le32toh (gdp->bg_block_bitmap)); > > + journal_get_write_access (txn, boffs_block (bptr_offs (gdp))); > > > > ext2_debug ("using block group %d (%d)", i, le16toh > (gdp->bg_free_blocks_count)); > > > > diff --git a/ext2fs/ext2fs.h b/ext2fs/ext2fs.h > > index 9282e5417..458446984 100644 > > --- a/ext2fs/ext2fs.h > > +++ b/ext2fs/ext2fs.h > > @@ -346,12 +346,39 @@ extern struct journal *ext2_journal; > > #define JRNL_LOG_WARN(fmt, ...) ext2_warning ("[JOURNAL] " fmt, > ##__VA_ARGS__) > > > > /** > > - * Mark dirty: Add a modified filesystem block to the given transaction. > > - * Performs a shadow copy of 'data' into the journal memory. > > + * Notify the journal that the content of the block fs_blocknr is about > to get > > + * modified by a filesystem operation. > > + * Journal will arm necessary buffers to track the block and it will > > + * automatically copy the memory contained at that block when the > > + * transaction txn is committing and the block's memory has settled. > > + * > > + * If the memory of the block is modified before calling this function > such > > + * modifications won't be part of the journal transaction. > > + * > > + * In between the invocation of this function and the commit, journal > will be > > + * intercepting pager's writes to this block and instead of to disk > they will > > + * be temporarily stored in journal's caches. > > */ > > error_t > > -journal_dirty_block (diskfs_transaction_t * txn, block_t fs_blocknr); > > +journal_get_write_access (diskfs_transaction_t * txn, block_t > fs_blocknr); > > + > > +/** > > + * Notifies the journal that modifications for this block have been > complete; > > + * the block is copied into the transaction again at the next sweep. > > + * > > + * Shouldn't be called if no journal_get_write_access() was called > previously > > + * for the same transaction and the same block. > > + */ > > +void > > +journal_mark_dirty (diskfs_transaction_t * txn, block_t fs_blocknr); > > > > +/* Return the transaction of the calling thread's open handle. While > the > > + journal is live the caller must hold a handle and the result is a > > + T_RUNNING or T_LOCKED transaction that stays usable until the handle > is > > + released. NULL means there is nothing to journal into: no journal, > or > > + the journal is shutting down and no handle was opened. */ > > +diskfs_transaction_t * > > +journal_thread_transaction (void); > > > > void ext2_orphan_drop_ram_link (struct node *np); > > > > @@ -531,17 +558,9 @@ extern void alloc_sync (struct node *np); > > > > #if defined(__USE_EXTERN_INLINES) || defined(EXT2FS_DEFINE_EI) > > EXT2FS_EI void > > -journal_notify_block_changed (block_t block) > > +journal_mark_dirty_current (block_t block) > > { > > - if (!ext2_journal) > > - return; > > - > > - diskfs_transaction_t *txn = diskfs_journal_start_transaction (); > > - error_t err = journal_dirty_block (txn, block); > > - if (err) > > - JRNL_LOG_WARN ("Didn't manage to add a dirty block %u to the > journal. (%s).", > > - block, strerror (err)); > > - diskfs_journal_stop_transaction (txn); > > + journal_mark_dirty (journal_thread_transaction (), block); > > } > > > > /* Marks the global block BLOCK as being modified, and returns true if > we > > @@ -569,7 +588,7 @@ record_global_poke (void *ptr) > > { > > block_t block = boffs_block (bptr_offs (ptr)); > > void *block_ptr = bptr (block); > > - journal_notify_block_changed (block); > > + journal_mark_dirty_current (block); > > ext2_debug ("(%p = %p)", ptr, block_ptr); > > #ifdef EXT2FS_DEBUG > > assert_backtrace (disk_cache_block_is_ref (block)); > > @@ -584,7 +603,7 @@ sync_global_ptr (void *ptr, int wait) > > { > > block_t block = boffs_block (bptr_offs (ptr)); > > void *block_ptr = bptr (block); > > - journal_notify_block_changed (block); > > + journal_mark_dirty_current (block); > > ext2_debug ("(%p -> %u)", ptr, block); > > global_block_modified (block); > > _disk_cache_block_deref (block_ptr); > > @@ -611,7 +630,7 @@ record_indir_poke (struct node *node, void *ptr) > > { > > block_t block = boffs_block (bptr_offs (ptr)); > > void *block_ptr = bptr (block); > > - journal_notify_block_changed (block); > > + journal_mark_dirty_current (block); > > ext2_debug ("(%llu, %p)", node->cache_id, ptr); > > #ifdef EXT2FS_DEBUG > > assert_backtrace (disk_cache_block_is_ref (block)); > > @@ -630,8 +649,8 @@ sync_global (int wait) > > } > > > > /* Sync all allocation information and node NP if diskfs_synchronous. > > - If journaling is active, we just update memory (wait=0) and let the > > - transaction commit handle durability. */ > > + If journaling is active, push the superblock into this transaction; > > + the commit provides durability. */ > > EXT2FS_EI void > > alloc_sync (struct node *np) > > { > > @@ -640,15 +659,16 @@ alloc_sync (struct node *np) > > { > > diskfs_transaction_t *txn = diskfs_journal_start_transaction (); > > > > + /* If the superblock was modified in memory, push it to the disk > cache > > + now so it gets bundled into this transaction's WAL commit. > > + diskfs_set_hypermetadata natively handles the get_write_access > -> > > + memcpy -> mark_dirty pipeline! */ > > + if (sblock_dirty && ext2_journal) > > + diskfs_set_hypermetadata (0, 0); > > + > > if (np) > > diskfs_node_update (np, diskfs_synchronous); > > > > - if (sblock_dirty && ext2_journal) > > - { > > - block_t sb_blocknr = boffs_block (SBLOCK_OFFS); > > - journal_dirty_block (txn, sb_blocknr); > > - } > > - > > diskfs_journal_stop_transaction (txn); > > } > > > > @@ -658,7 +678,9 @@ alloc_sync (struct node *np) > > if (np) > > pokel_sync (&diskfs_node_disknode (np)->indir_pokel, 1); > > > > - diskfs_set_hypermetadata (1, 0); > > + /* Only flush if journaling didn't already update the cache above > */ > > + if (!ext2_journal) > > + diskfs_set_hypermetadata (1, 0); > > } > > } > > #endif /* Use extern inlines. */ > > diff --git a/ext2fs/getblk.c b/ext2fs/getblk.c > > index f2a219437..7fe219f2b 100644 > > --- a/ext2fs/getblk.c > > +++ b/ext2fs/getblk.c > > @@ -70,6 +70,7 @@ ext2_alloc_block (struct node *node, block_t goal, int > zero) > > static unsigned long alloc_hits = 0, alloc_attempts = 0; > > #endif > > block_t result; > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > > > #ifdef EXT2_PREALLOCATE > > if (diskfs_node_disknode (node)->info.i_prealloc_count && > > @@ -110,7 +111,9 @@ ext2_alloc_block (struct node *node, block_t goal, > int zero) > > if (result && zero) > > { > > char *bh = disk_cache_block_ref (result); > > + journal_get_write_access (txn, result); > > memset (bh, 0, block_size); > > + /* record_indir_poke handles journal_mark_dirty internally. */ > > record_indir_poke (node, bh); > > } > > > > @@ -196,6 +199,7 @@ block_getblk (struct node *node, block_t block, int > nr, int create, int zero, > > int i; > > block_t goal = 0; > > block_t *bh = (block_t *)disk_cache_block_ref (block); > > + diskfs_transaction_t *txn; > > > > *result = bh[nr]; > > if (*result) > > @@ -209,6 +213,7 @@ block_getblk (struct node *node, block_t block, int > nr, int create, int zero, > > disk_cache_block_deref (bh); > > return EINVAL; > > } > > + txn = journal_thread_transaction (); > > > > if (diskfs_node_disknode (node)->info.i_next_alloc_block == new_block) > > goal = diskfs_node_disknode (node)->info.i_next_alloc_goal; > > @@ -233,10 +238,12 @@ block_getblk (struct node *node, block_t block, > int nr, int create, int zero, > > return ENOSPC; > > } > > > > + journal_get_write_access (txn, block); > > bh[nr] = *result; > > > > if (diskfs_synchronous || diskfs_node_disknode (node)->info.i_osync) > > { > > + /* calls journal_mark_dirty internally */ > > sync_global_ptr (bh, 1); > > /* We just wrote a new indirect block pointer. > > If this doesn't hit the platter, the file is corrupt. */ > > @@ -245,6 +252,7 @@ block_getblk (struct node *node, block_t block, int > nr, int create, int zero, > > ext2_warning ("indirect block flush failed: %s", strerror (err)); > > } > > else > > + /* record_indir_poke handles journal_mark_dirty internally */ > > record_indir_poke (node, bh); > > > > diskfs_node_disknode (node)->info.i_next_alloc_block = new_block; > > diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c > > index 35cdb7ed0..60be3e22f 100644 > > --- a/ext2fs/hyper.c > > +++ b/ext2fs/hyper.c > > @@ -240,6 +240,8 @@ diskfs_set_hypermetadata (int wait, int clean) > > /* Before writing, set the time of write */ > > sblock->s_wtime = htole32 (diskfs_mtime->seconds); > > sblock_dirty = 0; > > + block_t blk = boffs_block (bptr_offs (mapped_sblock)); > > + journal_get_write_access (txn, blk); > > memcpy (mapped_sblock, sblock, SBLOCK_SIZE); > > disk_cache_block_ref_ptr (mapped_sblock); > > record_global_poke (mapped_sblock); > > diff --git a/ext2fs/ialloc.c b/ext2fs/ialloc.c > > index 438de8089..cc9a1edf3 100644 > > --- a/ext2fs/ialloc.c > > +++ b/ext2fs/ialloc.c > > @@ -59,6 +59,8 @@ diskfs_free_node (struct node *np, mode_t old_mode) > > unsigned long bit; > > struct ext2_group_desc *gdp; > > ino_t inum = np->cache_id; > > + block_t gdp_block, gdp_bitmap_blk; > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > > > assert_backtrace (!diskfs_readonly); > > > > @@ -79,8 +81,12 @@ diskfs_free_node (struct node *np, mode_t old_mode) > > bit = (inum - 1) % le32toh (sblock->s_inodes_per_group); > > > > gdp = group_desc (block_group); > > - bh = disk_cache_block_ref (le32toh (gdp->bg_inode_bitmap)); > > + gdp_block = boffs_block (bptr_offs (gdp)); > > + gdp_bitmap_blk = le32toh (gdp->bg_inode_bitmap); > > + bh = disk_cache_block_ref (gdp_bitmap_blk); > > > > + journal_get_write_access (txn, gdp_bitmap_blk); > > + journal_get_write_access (txn, gdp_block); > > if (!clear_bit (bit, bh)) > > ext2_warning ("bit already cleared for inode %" PRIu64, inum); > > else > > @@ -123,6 +129,7 @@ ext2_alloc_inode (ino_t dir_inum, mode_t mode) > > ino_t inum; > > struct ext2_group_desc *gdp; > > struct ext2_group_desc *tmp; > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > > > pthread_spin_lock (&global_lock); > > > > @@ -228,6 +235,7 @@ repeat: > > find_first_zero_bit ((uint32_t *) bh, le32toh > (sblock->s_inodes_per_group))) > > < le32toh (sblock->s_inodes_per_group)) > > { > > + journal_get_write_access (txn, le32toh (gdp->bg_inode_bitmap)); > > if (set_bit (inum, bh)) > > { > > ext2_warning ("bit already set for inode %" PRIu64, inum); > > @@ -258,6 +266,7 @@ repeat: > > goto sync_out; > > } > > > > + journal_get_write_access (txn, boffs_block (bptr_offs (gdp))); > > gdp->bg_free_inodes_count = htole16 (le16toh > (gdp->bg_free_inodes_count) - 1); > > if (S_ISDIR (mode)) > > gdp->bg_used_dirs_count = htole16 (le16toh > (gdp->bg_used_dirs_count) + 1); > > diff --git a/ext2fs/inode.c b/ext2fs/inode.c > > index 00c4bebb4..21256f843 100644 > > --- a/ext2fs/inode.c > > +++ b/ext2fs/inode.c > > @@ -408,6 +408,7 @@ write_node (struct node *np) > > error_t err; > > struct stat *st = &np->dn_stat; > > struct ext2_inode *di; > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > > > ext2_debug ("(%llu)", np->cache_id); > > > > @@ -428,6 +429,7 @@ write_node (struct node *np) > > > > di = dino_ref (np->cache_id); > > > > + journal_get_write_access (txn, boffs_block (bptr_offs (di))); > > di->i_generation = htole32 (st->st_gen); > > > > /* We happen to know that the stat mode bits are the same > > @@ -578,13 +580,17 @@ void > > diskfs_write_disknode (struct node *np, int wait) > > { > > error_t err; > > + diskfs_transaction_t *txn = diskfs_journal_start_transaction (); > > + > > struct ext2_inode *di = write_node (np); > > if (!di) > > - return; > > + { > > + diskfs_journal_stop_transaction (txn); > > + return; > > + } > > > > if (ext2_journal) > > { > > - diskfs_transaction_t *txn = diskfs_journal_start_transaction (); > > record_global_poke (di); > > if (wait) > > diskfs_journal_set_sync (txn); > > @@ -601,9 +607,7 @@ diskfs_write_disknode (struct node *np, int wait) > > ext2_warning ("device flush failed: %s", strerror (err)); > > } > > else > > - { > > - record_global_poke (di); > > - } > > + record_global_poke (di); > > } > > > > /* Set *ST with appropriate values to reflect the current state of the > > @@ -634,6 +638,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > struct protid *cred) > > { > > error_t err; > > + diskfs_transaction_t *txn; > > > > assert_backtrace (!diskfs_readonly); > > > > @@ -641,6 +646,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > if (err) > > return err; > > > > + txn = journal_thread_transaction (); > > /* If xattr is supported for this filesystem, use xattr to store > translator > > record, otherwise, use legacy translator record */ > > if (EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR) > > @@ -662,6 +668,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > ext2_debug ("Old translator record found, clear it"); > > > > /* Clear block for translator going away. */ > > + journal_get_write_access (txn, boffs_block (bptr_offs (di))); > > di->i_translator = htole32 (0); > > diskfs_node_disknode (np)->info_i_translator = 0; > > record_global_poke (di); > > @@ -761,6 +768,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > np->dn_stat.st_mode = newmode; > > } > > > > + journal_get_write_access (txn, boffs_block (bptr_offs (di))); > > di->i_translator = htole32 (blkno); > > diskfs_node_disknode (np)->info_i_translator = blkno; > > record_global_poke (di); > > @@ -771,6 +779,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > else if (!namelen && blkno) > > { > > /* Clear block for translator going away. */ > > + journal_get_write_access (txn, boffs_block (bptr_offs (di))); > > di->i_translator = htole32 (0); > > diskfs_node_disknode (np)->info_i_translator = 0; > > record_global_poke (di); > > @@ -796,6 +805,7 @@ diskfs_set_translator (struct node *np, const char > *name, mach_msg_type_number_t > > memcpy (buf + 2, name, namelen); > > > > blkptr = disk_cache_block_ref (blkno); > > + journal_get_write_access (txn, blkno); > > memcpy (blkptr, buf, block_size); > > record_global_poke (blkptr); > > > > diff --git a/ext2fs/journal.c b/ext2fs/journal.c > > index 6828aa4d8..85dde2ad5 100644 > > --- a/ext2fs/journal.c > > +++ b/ext2fs/journal.c > > @@ -89,7 +89,7 @@ > > * Slab Allocator Pool Size. > > * Pre-allocates a contiguous chunk of memory for journal buffers > > * (512 * 4KB = 2MB). > > - * This allows journal_dirty_block to be a zero-allocation operation > for the > > + * This allows journal_get_write_access to be a zero-allocation > operation for the > > * vast majority of workloads, falling back to dynamic allocation only > under > > * extreme metadata pressure. > > */ > > @@ -287,7 +287,7 @@ typedef struct journal > > /* Pre-allocated buffers for zero-allocation commits */ > > void *j_descriptor_buf; > > void *j_commit_buf; > > - /* Pre-allocated buffers for (near) zero-allocation > journal_dirty_block */ > > + /* Pre-allocated buffers for (near) zero-allocation > journal_get_write_access */ > > journal_buffer_t *j_pool_memory; /* The raw contiguous block */ > > journal_buffer_t *j_free_buffers; /* The linked list head */ > > > > @@ -877,9 +877,10 @@ journal_get_oldest_transaction_locked (journal_t > *journal) > > * checkpoint lists AFTER this transaction safely commits. > > */ > > void > > -journal_record_freed_blocks (block_t start, unsigned long count) > > +journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start, > > + unsigned long count) > > { > > - if (!ext2_journal) > > + if (!ext2_journal || !txn) > > return; > > > > journal_freed_extent_t *ext = malloc (sizeof > (journal_freed_extent_t)); > > @@ -894,11 +895,6 @@ journal_record_freed_blocks (block_t start, > unsigned long count) > > ext->fe_count = count; > > > > JOURNAL_LOCK (ext2_journal); > > - /* Record against this thread's transaction when it holds one: the > free > > - belongs to the same RPC, and that transaction may be T_LOCKED while > > - j_running_transaction is NULL. Otherwise use the running one. */ > > - diskfs_transaction_t *txn = journal_thread_depth > 0 > > - ? journal_thread_txn : ext2_journal->j_running_transaction; > > if (!txn || (txn->t_state != T_RUNNING && txn->t_state != T_LOCKED)) > > { > > JRNL_LOG_DEBUG ("Cannot record freed blocks, no running > transaction."); > > @@ -1077,7 +1073,7 @@ journal_stop_transaction_locked (journal_t > *journal, > > */ > > memcpy (curr->jb_shadow_data, live_cache_ptr, block_size); > > /* needs_copy was cleared under the lock when this block was > > - * listed. If the block is modified again, journal_dirty_block > > + * listed. If the block is modified again, > journal_get_write_access > > * sets it back to 1 and a later sweep recopies. */ > > > > journal_buffer_t *next = curr->jb_next; > > @@ -1611,23 +1607,53 @@ journal_wait_on_tid_locked (journal_t *journal, > uint32_t target_tid) > > JOURNAL_WAIT (&journal->j_commit_done, journal); > > } > > > > +void > > +journal_mark_dirty (diskfs_transaction_t *txn, block_t fs_blocknr) > > +{ > > + if (!ext2_journal || !txn) > > + return; > > + > > + JOURNAL_LOCK (ext2_journal); > > + journal_buffer_t *jb = journal_map_lookup (&txn->t_buffer_map, > fs_blocknr); > > + if (!jb) > > + { > > + /* STRICT JBD2 BEHAVIOR: > > + If we hit this, the filesystem modified a block without calling > > + journal_get_write_access() first! */ > > + JRNL_LOG_WARN ("mark_dirty called on unreserved block %u! Missing > get_write_access?", > > + fs_blocknr); > > + goto out; > > + } > > + > > + /* We don't delete the intercepted chunk here, its the only copy of > the data > > + we have. It won't be used for hydration, or for post commit > flushing but > > + we still need it for now for the journal_store_read(). We will > instead > > + instruct the hydration that this needs a fresh copy. */ > > + jb->needs_copy = 1; > > + > > + /* Reset the physical write flag so the Checkpoint thread knows to > > + overwrite the disk with our pristine shadow buffer later. */ > > + if (jb->jb_is_written) > > + { > > + jb->jb_is_written = 0; > > + txn->t_outstanding_io++; > > + } > > + > > +out: > > + JOURNAL_UNLOCK (ext2_journal); > > +} > > + > > /** > > * Adds a modified filesystem block to the SPECIFIC transaction handle. > > * Defers the actual memory copy until the transaction stops. > > */ > > static error_t > > -journal_dirty_block_locked (diskfs_transaction_t *txn, block_t > fs_blocknr) > > +journal_get_write_access_locked (diskfs_transaction_t *txn, block_t > fs_blocknr) > > { > > journal_buffer_t *jb; > > journal_buffer_t *new_jb; > > error_t err = 0; > > > > - if (!txn) > > - { > > - JRNL_LOG_DEBUG ("[TRX] Transaction null but block dirty."); > > - goto out; > > - } > > - > > assert_backtrace (txn->t_state == T_RUNNING || txn->t_state == > T_LOCKED); > > jb = journal_map_lookup (&txn->t_buffer_map, fs_blocknr); > > > > @@ -1644,11 +1670,6 @@ journal_dirty_block_locked (diskfs_transaction_t > *txn, block_t fs_blocknr) > > jb->jb_is_written = 0; > > txn->t_outstanding_io++; > > } > > - if (jb->jb_intercepted_data) > > - { > > - journal_free_intercept_chunk (ext2_journal, > jb->jb_intercepted_data); > > - jb->jb_intercepted_data = NULL; > > - } > > goto out; > > } > > > > @@ -1786,17 +1807,61 @@ journal_thread_release (diskfs_transaction_t > *txn) > > * Defers the actual memory copy until the transaction stops. > > */ > > error_t > > -journal_dirty_block (diskfs_transaction_t *txn, block_t fs_blocknr) > > +journal_get_write_access (diskfs_transaction_t *txn, block_t fs_blocknr) > > { > > - error_t err; > > + error_t err = 0; > > if (!ext2_journal) > > - return EINVAL; > > + goto out; > > + > > + if (!txn) > > + { > > + JRNL_LOG_WARN > > + ("Transaction null so no write access granted for block %u.", > > + fs_blocknr); > > + goto out; > > + } > > + > > JOURNAL_LOCK (ext2_journal); > > - err = journal_dirty_block_locked (txn, fs_blocknr); > > + err = journal_get_write_access_locked (txn, fs_blocknr); > > + if (err) > > + JRNL_LOG_WARN > > + ("Didn't manage to get journal write access for block %u. (%s).", > > + fs_blocknr, strerror (err)); > > JOURNAL_UNLOCK (ext2_journal); > > + > > +out: > > return err; > > } > > > > +/* Return the transaction of the calling thread's open handle. > > + > > + While the journal is live the caller must hold a handle, taken with > > + diskfs_journal_start_transaction; calling without one is a bug and > > + asserts. The result is then never NULL, and the handle's > > + t_active_threads count keeps it T_RUNNING or T_LOCKED until the > handle > > + is released, so blocks may be added to it. > > + > > + NULL is returned only when the thread holds no handle because there > is > > + no journal, or because the journal is shutting down (j_must_exit) and > > + diskfs_journal_start_transaction declined to open one. Callers must > > + accept NULL and treat it as "do not journal". > > + > > + Under NDEBUG the missing-handle check is compiled out: a caller that > > + forgot its handle gets NULL and its edits silently go unjournaled. > */ > > +diskfs_transaction_t * > > +journal_thread_transaction (void) > > +{ > > + diskfs_transaction_t *txn = journal_thread_txn; > > + > > + assert_backtrace (!ext2_journal || ext2_journal->j_must_exit > > + || journal_thread_depth > 0); > > + /* The handle's t_active_threads count keeps commit from moving the > > + transaction past T_LOCKED. */ > > + assert_backtrace (!txn || txn->t_state == T_RUNNING > > + || txn->t_state == T_LOCKED); > > + return txn; > > +} > > + > > /* Two flags. journal_thread_sync tells this thread's RPC tail to > commit > > rather than stop. txn->sync_needed tells the outermost stop to wake > > kjournald when t_active_threads reaches 0, for callers that never > commit > > diff --git a/ext2fs/journal.h b/ext2fs/journal.h > > index ab7ab1e6b..3cc8212b1 100644 > > --- a/ext2fs/journal.h > > +++ b/ext2fs/journal.h > > @@ -84,7 +84,7 @@ journal_store_read (block_t start_block, size_t > length, void **buf, > > * Records a range of deleted blocks so they can be unpinned from older > > * checkpoint lists AFTER this transaction safely commits. > > */ > > -void journal_record_freed_blocks (block_t start, unsigned long count); > > +void journal_record_freed_blocks (diskfs_transaction_t *txn, block_t > start, unsigned long count); > > > > /** > > * Marks the calling thread as running a pager callback (ON = 1) or done > > diff --git a/ext2fs/orphan.c b/ext2fs/orphan.c > > index 56782d39b..24f10e7d9 100644 > > --- a/ext2fs/orphan.c > > +++ b/ext2fs/orphan.c > > @@ -43,11 +43,12 @@ diskfs_orphan_add (struct node *np) > > { > > ino_t inum = np->cache_id; > > struct ext2_inode *di; > > - diskfs_transaction_t *txn = NULL; > > + diskfs_transaction_t *txn; > > > > assert_backtrace (!diskfs_readonly); > > assert_backtrace (np->dn_stat.st_nlink == 0); > > > > + /* The orphan list is exclusively an ext3/journaling feature. */ > > if (!ext2_journal) > > return; > > > > @@ -59,15 +60,13 @@ diskfs_orphan_add (struct node *np) > > 2. global_lock: Protects the in-memory superblock modifications. > > 3. Journal Transaction (txn): Guarantees that the superblock > pointer and the > > inode pointer hit the physical disk as a single, atomic > operation. */ > > - txn = diskfs_journal_start_transaction (); > > + txn = journal_thread_transaction (); > > > > pthread_mutex_lock (&orphan_lock); > > > > if (diskfs_node_disknode (np)->on_orphan_list) > > { > > pthread_mutex_unlock (&orphan_lock); > > - if (txn) > > - diskfs_journal_stop_transaction (txn); > > return; > > } > > > > @@ -80,7 +79,9 @@ diskfs_orphan_add (struct node *np) > > > > di = dino_ref (inum); > > > > - /* Atomically link the new orphan to the head of the on-disk list. */ > > + /* Reserve the inode block in the journal */ > > + journal_get_write_access (txn, boffs_block (bptr_offs (di))); > > + > > pthread_spin_lock (&global_lock); > > di->i_dtime = sblock->s_last_orphan; > > sblock->s_last_orphan = htole32 (inum); > > @@ -97,16 +98,8 @@ diskfs_orphan_add (struct node *np) > > di->i_size_high = 0; > > memset (di->i_block, 0, EXT2_N_BLOCKS * sizeof di->i_block[0]); > > > > - if (txn) > > - { > > - /* Atomically bundle the superblock and the placeholder inode. > > - By dirtying both blocks in the same transaction, we guarantee > that a > > - crash cannot leave a severed list chain. */ > > - memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE); > > - journal_dirty_block (txn, boffs_block (bptr_offs (di))); > > - journal_dirty_block (txn, boffs_block (SBLOCK_OFFS)); > > - } > > - > > + /* Let the journal know we are done editing. */ > > + journal_mark_dirty (txn, boffs_block (bptr_offs (di))); > > dino_deref (di); > > > > /* Maintain the in-memory doubly linked list for O(1) removals. */ > > @@ -118,10 +111,8 @@ diskfs_orphan_add (struct node *np) > > > > pthread_mutex_unlock (&orphan_lock); > > > > - if (txn) > > - diskfs_journal_stop_transaction (txn); > > - else > > - diskfs_set_hypermetadata (0, 0); > > + /* hyper.c handles the superblock's get_write_access -> memcpy -> > mark_dirty! */ > > + diskfs_set_hypermetadata (0, 0); > > } > > > > /* Remove inode NP from the orphan list. NP is locked by the caller. */ > > @@ -129,7 +120,7 @@ void > > diskfs_orphan_del (struct node *np) > > { > > ino_t inum = np->cache_id; > > - diskfs_transaction_t *txn = NULL; > > + diskfs_transaction_t *txn; > > int update_super = 0; > > > > if (!ext2_journal) > > @@ -138,15 +129,13 @@ diskfs_orphan_del (struct node *np) > > if (!diskfs_node_disknode (np)->on_orphan_list) > > return; > > > > - txn = diskfs_journal_start_transaction (); > > + txn = journal_thread_transaction (); > > > > pthread_mutex_lock (&orphan_lock); > > > > if (!diskfs_node_disknode (np)->on_orphan_list) > > { > > pthread_mutex_unlock (&orphan_lock); > > - if (txn) > > - diskfs_journal_stop_transaction (txn); > > return; > > } > > > > @@ -154,12 +143,16 @@ diskfs_orphan_del (struct node *np) > > > > struct ext2_inode *my_di = dino_ref (inum); > > __u32 my_next = le32toh (my_di->i_dtime); > > + block_t blocknr = boffs_block (bptr_offs (my_di)); > > > > /* This inode is leaving the list. i_dtime becomes a normal deletion > > stamp in the caller's following write_node (mode is already 0). > > - We do not journal_dirty it here: that copy can run before > write_node > > - stores the cleared block map. */ > > + We let the journal know we are about to modify the associated > block. */ > > + journal_get_write_access (txn, blocknr); > > my_di->i_dtime = 0; > > + /* Done editing. */ > > + journal_mark_dirty (txn, blocknr); > > + > > dino_deref (my_di); > > > > struct node *prev = diskfs_node_disknode (np)->orphan_prev; > > @@ -172,14 +165,7 @@ diskfs_orphan_del (struct node *np) > > sblock_dirty = 1; > > pthread_spin_unlock (&global_lock); > > > > - if (txn) > > - { > > - memcpy (boffs_ptr (SBLOCK_OFFS), sblock, SBLOCK_SIZE); > > - journal_dirty_block (txn, boffs_block (SBLOCK_OFFS)); > > - } > > - else > > - update_super = 1; > > - > > + update_super = 1; > > ram_orphan_head = next; > > } > > else > > @@ -188,9 +174,9 @@ diskfs_orphan_del (struct node *np) > > > > /* prev stays on the list, so its cached i_block[] is already the > > placeholder (zeros). Only i_dtime changes. */ > > + journal_get_write_access (txn, boffs_block (bptr_offs (prev_di))); > > prev_di->i_dtime = htole32 (my_next); > > - if (txn) > > - journal_dirty_block (txn, boffs_block (bptr_offs (prev_di))); > > + journal_mark_dirty (txn, boffs_block (bptr_offs (prev_di))); > > dino_deref (prev_di); > > > > diskfs_node_disknode (prev)->orphan_next = next; > > @@ -205,10 +191,9 @@ diskfs_orphan_del (struct node *np) > > > > pthread_mutex_unlock (&orphan_lock); > > > > + /* Bundle the superblock modification into the transaction if needed > */ > > if (update_super) > > diskfs_set_hypermetadata (0, 0); > > - else > > - diskfs_journal_stop_transaction (txn); > > } > > > > /* Recover (clean up) the orphan list at mount time. > > @@ -229,6 +214,7 @@ ext2_recover_orphan_list (void) > > int count = 0; > > struct ext2_inode *di; > > struct node *np = NULL; > > + diskfs_transaction_t *txn; > > error_t err; > > __u32 max_inodes = le32toh (sblock->s_inodes_count); > > > > @@ -270,9 +256,13 @@ ext2_recover_orphan_list (void) > > next_orphan = le32toh (di->i_dtime); > > dino_deref (di); > > > > + /* One handle per orphan, taken before the node lock like an > RPC's: > > + orphan_del and the drop in diskfs_nput edit metadata in it. */ > > + txn = diskfs_journal_start_transaction (); > > err = diskfs_cached_lookup (inum, &np); > > if (err || !np) > > { > > + diskfs_journal_stop_transaction (txn); > > ext2_warning ("cannot look up orphan inode %lu: %s", > > (unsigned long) inum, > > err ? strerror (err) : "not found"); > > @@ -298,6 +288,7 @@ ext2_recover_orphan_list (void) > > (unsigned long) inum); > > diskfs_orphan_del (np); > > diskfs_nput (np); > > + diskfs_journal_stop_transaction (txn); > > continue; > > } > > > > @@ -305,6 +296,7 @@ ext2_recover_orphan_list (void) > > truncate the file and call diskfs_orphan_del (np), which > > advances s_last_orphan and clears the head. */ > > diskfs_nput (np); > > + diskfs_journal_stop_transaction (txn); > > count++; > > } > > > > diff --git a/ext2fs/truncate.c b/ext2fs/truncate.c > > index 3dd9f8dfb..9c458f46e 100644 > > --- a/ext2fs/truncate.c > > +++ b/ext2fs/truncate.c > > @@ -120,6 +120,8 @@ trunc_indirect (struct node *node, block_t end, > > void (*free_block)(block_t *p, unsigned index), > > struct free_block_run *fbr) > > { > > + diskfs_transaction_t *txn = journal_thread_transaction (); > > + > > if (*p) > > { > > unsigned index; > > @@ -127,6 +129,7 @@ trunc_indirect (struct node *node, block_t end, > > block_t *ind_bh = (block_t *) disk_cache_block_ref (*p); > > unsigned first = end < offset ? 0 : end - offset; > > > > + journal_get_write_access (txn, *p); > > for (index = first; index < addr_per_block; index++) > > if (ind_bh[index]) > > { > > @@ -139,6 +142,10 @@ trunc_indirect (struct node *node, block_t end, > > > > if (first == 0 && all_freed) > > { > > + /* We modified this block before killing it. > > + Use *p (the block number), not ind_bh (the RAM pointer). */ > > + if (modified && ext2_journal) > > + journal_mark_dirty (txn, *p); > > pager_flush_some (diskfs_disk_pager, > > bptr_index (ind_bh) << log2_block_size, > > block_size, 1); > > diff --git a/ext2fs/xattr.c b/ext2fs/xattr.c > > index a1e02368f..1d007ef11 100644 > > --- a/ext2fs/xattr.c > > +++ b/ext2fs/xattr.c > > @@ -436,6 +436,7 @@ ext2_free_xattr_block (struct node *np) > > void *block; > > struct ext2_inode *ei; > > struct ext2_xattr_header *header; > > + diskfs_transaction_t *txn; > > > > if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR)) > > { > > @@ -443,6 +444,7 @@ ext2_free_xattr_block (struct node *np) > > return EOPNOTSUPP; > > } > > > > + txn = journal_thread_transaction (); > > err = 0; > > block = NULL; > > > > @@ -484,11 +486,13 @@ ext2_free_xattr_block (struct node *np) > > { > > ext2_debug("h_refcount: %d", le32toh (header->h_refcount)); > > > > + journal_get_write_access (txn, blkno); > > header->h_refcount = htole32 (le32toh (header->h_refcount) - 1); > > record_global_poke (block); > > } > > > > > > + journal_get_write_access (txn, boffs_block (bptr_offs (ei))); > > ei->i_file_acl = 0; > > record_global_poke (ei); > > > > @@ -680,6 +684,7 @@ ext2_set_xattr (struct node *np, const char *name, > const char *value, > > struct ext2_xattr_header *header; > > struct ext2_xattr_entry *entry; > > struct ext2_xattr_entry *location; > > + diskfs_transaction_t *txn; > > > > if (!EXT2_HAS_COMPAT_FEATURE (sblock, EXT2_FEATURE_COMPAT_EXT_ATTR)) > > { > > @@ -692,6 +697,7 @@ ext2_set_xattr (struct node *np, const char *name, > const char *value, > > > > if (strlen(name) > 255 || len > block_size) > > return ERANGE; > > + txn = journal_thread_transaction (); > > > > ei = dino_ref (np->cache_id); > > blkno = ei->i_file_acl; > > @@ -722,6 +728,7 @@ ext2_set_xattr (struct node *np, const char *name, > const char *value, > > } > > > > block = disk_cache_block_ref (blkno); > > + journal_get_write_access (txn, blkno); > > memset (block, 0, block_size); > > > > header = EXT2_XATTR_HEADER (block); > > @@ -739,6 +746,7 @@ ext2_set_xattr (struct node *np, const char *name, > const char *value, > > err = EIO; > > goto cleanup; > > } > > + journal_get_write_access (txn, blkno); > > } > > > > entry = EXT2_XATTR_ENTRY_FIRST (header); > > @@ -864,6 +872,7 @@ ext2_set_xattr (struct node *np, const char *name, > const char *value, > > np->dn_stat.st_blocks += 1 << log2_stat_blocks_per_fs_block; > > np->dn_set_ctime = 1; > > > > + journal_get_write_access (txn, boffs_block (bptr_offs (ei))); > > ei->i_file_acl = blkno; > > record_global_poke (ei); > > } > > -- > > 2.56.0 > > > > > > -- > Samuel > +#if defined(__alpha__) && defined(CONFIG_PCI) > + /* > + * The meaning of life, the universe, and everything. Plus > + * this makes the year come out right. > + */ > + year -= 42; > +#endif > (From the patch for 1.3.2: (kernel/time.c), submitted by Marcus Meissner) >
