Applied, thanks!
Milos Nikic, le mer. 30 sept. 2026 21:00:08 -0700, a ecrit:
> This patch replaces the fixed-size, bitmask-managed Lifeboat array with an
> O(1) intrusive linked-list allocator tied directly to the transaction map.
>
> Previously, when the Mach VM pager rushed a block locked by an active
> transaction, the payload was intercepted and stored in a global array
> indexed via an 8-slot `uint64_t` bitmask. This approach required expensive
> index lookups, O(N) allocation loops, and compare-and-swap checks during
> physical flushing to avoid race conditions.
>
> By embedding an intercepted_data pointer directly into journal_buffer_t,
> the intercepted memory is now exclusively owned by the transaction itself.
> The global array is replaced with a simple, pre-allocated linked list of
> 4KB chunks, granting O(1) push/pop allocations and entirely eliminating
> index collision hazards.
>
> Cache coherence in journal_store_read is preserved by checking the
> `RUNNING` transaction's pointers before the COMMITTING transaction's
> pointers, natively shadowing older intercepted data.
> ---
> ext2fs/journal.c | 356 +++++++++++++++++++++--------------------------
> 1 file changed, 155 insertions(+), 201 deletions(-)
>
> diff --git a/ext2fs/journal.c b/ext2fs/journal.c
> index cc3aeb35a..05448b2e4 100644
> --- a/ext2fs/journal.c
> +++ b/ext2fs/journal.c
> @@ -121,28 +121,8 @@
> #define IS_RUNNING_TID(j, tid) \
> ((j)->j_running_transaction && (j)->j_running_transaction->t_tid ==
> (tid))
>
> -#define JRNL_LIFEBOAT_CAPACITY 512
> +#define JRNL_INTERCEPT_CAPACITY 512
>
> -#define JRNL_LIFEBOAT_ALLOC_MASK_LEN 8
> -
> -/* Temporary storage for blocks rushed by the Mach VM pager.
> - * Because we cannot block or delay the pager when it needs to flush a page
> - * belonging to an active (RUNNING/COMMITTING) transaction, this cache
> - * absorbs the write. This prevents a permanent deadlock while preserving
> - * Write-Ahead Log (WAL) ordering.
> - * The payloads are flushed to disk as soon as the transaction safely
> commits.
> - */
> -struct journal_lifeboat
> -{
> - /* 512 bits total: 0 means free, 1 means occupied.
> - Protected by the main ext2_journal->j_state_lock. */
> - uint64_t alloc_mask[JRNL_LIFEBOAT_ALLOC_MASK_LEN];
> -
> - /* The pre-allocated payload pool (512 * 4KB = 2MB) */
> - char *payloads;
> -};
> -
> -static struct journal_lifeboat ext2_lifeboat;
> static pthread_t kjournald_tid;
>
> /* The handle this thread currently holds, and how many starts are open
> @@ -181,8 +161,8 @@ typedef struct journal_buffer
> uint8_t needs_copy; /* Whether this buffer needs a new copy
> from
> from the live Mach VM cache. Should be 1
> when new. */
> - /* -1 if normal, 0-127 if holding a spoofed payload in the lifeboat */
> - int16_t lifeboat_index;
> + /* Pointer to the intercepted data */
> + char *jb_intercepted_data;
> uint8_t jb_is_flushing; /* 1 if commit thread is actively flushing it.
> */
> uint8_t jb_escaped;
> } journal_buffer_t;
> @@ -310,53 +290,40 @@ typedef struct journal
> /* Pre-allocated buffers for (near) zero-allocation journal_dirty_block */
> journal_buffer_t *j_pool_memory; /* The raw contiguous block */
> journal_buffer_t *j_free_buffers; /* The linked list head */
> +
> + /* Intercepted data pool pointers */
> + char *j_intercept_pool;
> + void *j_free_intercept_chunks;
> } journal_t;
>
> /**
> - * Returns 0-127 on success, or -1 if the lifeboat is full.
> - * MUST be called with JOURNAL_LOCK(ext2_journal) held.
> + * O(1) Intercept Chunk Allocation.
> + * MUST be called with JOURNAL_LOCK held.
> */
> -static inline int
> -lifeboat_alloc_slot (void)
> +static inline char *
> +journal_alloc_intercept_chunk (journal_t *journal)
> {
> - for (int i = 0; i < JRNL_LIFEBOAT_ALLOC_MASK_LEN; i++)
> + if (journal->j_free_intercept_chunks)
> {
> - /* Invert mask: 1s now represent FREE slots */
> - uint64_t free_bits = ~ext2_lifeboat.alloc_mask[i];
> -
> - if (free_bits != 0)
> - {
> - /* __builtin_ffsll returns 1-64, so we subtract 1 for 0-based index */
> - int bit = __builtin_ffsll ((long long) free_bits) - 1;
> - ext2_lifeboat.alloc_mask[i] |= (1ULL << bit);
> - return (int) (i * sizeof (ext2_lifeboat.alloc_mask[0]) * 8) + bit;
> - }
> + char *chunk = (char *) journal->j_free_intercept_chunks;
> + /* The first 8 bytes of the free chunk hold the 'next' pointer */
> + journal->j_free_intercept_chunks = *(void **) chunk;
> + return chunk;
> }
> - return -1;
> + return NULL; /* Starvation fallback */
> }
>
> /**
> - * Frees a raw slot back to the pool.
> - */
> -static inline void
> -lifeboat_free_slot (int index)
> -{
> - if (index >= 0 && index < JRNL_LIFEBOAT_CAPACITY)
> - ext2_lifeboat.alloc_mask[index / 64] &= ~(1ULL << (index % 64));
> -}
> -
> -/**
> - * Frees a slot back to the pool.
> - * MUST be called with JOURNAL_LOCK(ext2_journal) held.
> + * O(1) Intercept Chunk Deallocation.
> + * MUST be called with JOURNAL_LOCK held.
> */
> static inline void
> -lifeboat_release_buffer (journal_buffer_t *jb)
> +journal_free_intercept_chunk (journal_t *journal, char *chunk)
> {
> - int index = jb->lifeboat_index;
> - if (index >= 0 && index < JRNL_LIFEBOAT_CAPACITY)
> + if (chunk)
> {
> - ext2_lifeboat.alloc_mask[index / 64] &= ~(1ULL << (index % 64));
> - jb->lifeboat_index = -1;
> + *(void **) chunk = journal->j_free_intercept_chunks;
> + journal->j_free_intercept_chunks = chunk;
> }
> }
>
> @@ -600,7 +567,7 @@ journal_alloc_buffer (journal_t *journal)
> if (!jb)
> return NULL;
> out:
> - jb->lifeboat_index = -1;
> + jb->jb_intercepted_data = NULL;
> return jb;
> }
>
> @@ -611,14 +578,15 @@ out:
> static inline void
> journal_free_buffer (journal_t *journal, journal_buffer_t *jb)
> {
> - /* Check if this pointer falls inside our contiguous pool block */
> - if (jb->lifeboat_index >= 0)
> - lifeboat_release_buffer (jb);
> -
> uintptr_t ptr = (uintptr_t) jb;
> uintptr_t start = (uintptr_t) journal->j_pool_memory;
> uintptr_t end = (uintptr_t) &
> journal->j_pool_memory[JRNL_MAX_FREE_BUFFERS];
>
> + if (jb->jb_intercepted_data)
> + {
> + journal_free_intercept_chunk (journal, jb->jb_intercepted_data);
> + jb->jb_intercepted_data = NULL;
> + }
> if (ptr >= start && ptr < end)
> {
> /* It belongs to the permanent pool. Link it back up! */
> @@ -1062,7 +1030,7 @@ journal_stop_transaction_locked (journal_t *journal,
> {
> if (jb_exp->needs_copy)
> {
> - /* ALWAYS hydrate from the live VM cache. The lifeboat is for
> + /* ALWAYS hydrate from the live VM cache. The intercept is for
> delayed physical I/O, not for sourcing WAL shadow data! */
> /* Clear the flag here, under the lock, before the copy. A
> dirty that lands while we copy unlocked sets it again and
> @@ -1196,7 +1164,7 @@ journal_notify_blocks_written_locked (block_t
> start_block, size_t n_blocks)
> * stale relative to the final shadow copy.
> *
> * Blocks in the running transaction that legitimately need to be marked as
> - * written are handled by journal_flush_lifeboat_payloads(), which runs
> AFTER
> + * written are handled by journal_flush_intercepted_payloads(), which runs
> AFTER
> * the WAL barrier is crossed (T_COMMITTED) and does its own marking.
> *
> * The committing transaction and checkpoint list are safe to notify: their
> @@ -1376,6 +1344,61 @@ journal_is_block_in_newer_transaction_locked
> (journal_t *journal,
> return 0;
> }
>
> +/* Flush any intercepted VM pager blocks to the primary disk
> + Executes completely outside the global journal lock (manages its own lock
> per-block).
> + Called immediately after a transaction is safely committed to the WAL. */
> +static void
> +journal_flush_intercepted_payloads (journal_t *journal,
> + diskfs_transaction_t *txn)
> +{
> + size_t iter = 0;
> + journal_buffer_t *jb;
> +
> + while ((jb = journal_map_iterate (&txn->t_buffer_map, &iter)) != NULL)
> + {
> + char *flush_data = NULL;
> +
> + JOURNAL_LOCK (journal);
> + if (jb->jb_intercepted_data)
> + {
> + flush_data = jb->jb_intercepted_data;
> + jb->jb_is_flushing = 1;
> + }
> + JOURNAL_UNLOCK (journal);
> +
> + if (flush_data)
> + {
> + store_offset_t dev_block =
> + (store_offset_t) jb->jb_blocknr << log2_dev_blocks_per_fs_block;
> + size_t amount;
> + error_t err;
> +
> + /* Shadow data is fully settled and committed. */
> + err =
> + store_write (store, dev_block, jb->jb_shadow_data, block_size,
> &amount);
> +
> + JOURNAL_LOCK (journal);
> + if (err)
> + JRNL_LOG_WARN ("Intercept flush failed for block %u: %s",
> + jb->jb_blocknr, strerror (err));
> +
> + jb->jb_is_flushing = 0;
> +
> + /* We always free the chunk we just finished using */
> + journal_free_intercept_chunk (journal, flush_data);
> +
> + /* If the pager didn't replace it with a new write, clear it. */
> + if (jb->jb_intercepted_data == flush_data)
> + jb->jb_intercepted_data = NULL;
> +
> + if (!err)
> + journal_notify_blocks_written_locked (jb->jb_blocknr, 1);
> +
> + JOURNAL_UNLOCK (journal);
> + }
> + }
> +}
> +
> /**
> * Internal helper to flush checkpoint transactions to the main filesystem.
> * If target_free is UINT32_MAX, it flushes ALL transactions in the list.
> @@ -1607,9 +1630,11 @@ journal_dirty_block_locked (diskfs_transaction_t *txn,
> block_t fs_blocknr)
> jb->jb_is_written = 0;
> txn->t_outstanding_io++;
> }
> - if (jb->lifeboat_index >= 0)
> - lifeboat_release_buffer (jb);
> -
> + if (jb->jb_intercepted_data)
> + {
> + journal_free_intercept_chunk (ext2_journal, jb->jb_intercepted_data);
> + jb->jb_intercepted_data = NULL;
> + }
> goto out;
> }
>
> @@ -1945,77 +1970,6 @@ journal_write_commit_record (journal_t *journal,
> return journal_write_block (journal, commit_loc, commit_buf);
> }
>
> -/**
> - * Flushes any intercepted pager writes (Lifeboat payloads) to the main
> filesystem.
> - * Executes completely outside the global journal lock (manages its own lock
> per-block).
> - * Called immediately after a transaction is safely committed to the WAL.
> - */
> -static void
> -journal_flush_lifeboat_payloads (journal_t *journal,
> - diskfs_transaction_t *txn)
> -{
> - size_t iter = 0;
> - journal_buffer_t *jb_lb;
> -
> - while ((jb_lb = journal_map_iterate (&txn->t_buffer_map, &iter)) != NULL)
> - {
> - int lb_idx = -1;
> -
> - JOURNAL_LOCK (journal);
> - if (jb_lb->lifeboat_index >= 0)
> - {
> - lb_idx = jb_lb->lifeboat_index;
> - jb_lb->jb_is_flushing = 1; /* Mark as actively flushing! */
> - }
> - JOURNAL_UNLOCK (journal);
> -
> - if (lb_idx >= 0)
> - {
> - store_offset_t dev_block =
> - (store_offset_t) jb_lb->jb_blocknr <<
> - log2_dev_blocks_per_fs_block;
> - size_t amount;
> - error_t err;
> -
> - /* We do the I/O using our safely captured, privately owned index */
> - err = store_write (store, dev_block,
> - &(ext2_lifeboat.payloads)[lb_idx * block_size],
> - block_size, &amount);
> -
> - JOURNAL_LOCK (journal);
> - if (err)
> - {
> - JRNL_LOG_WARN
> - ("Lifeboat flush failed for block %u: %s",
> - jb_lb->jb_blocknr, strerror (err));
> - }
> -
> - jb_lb->jb_is_flushing = 0; /* Done flushing (even if it failed) */
> -
> - /* Compare-and-Swap: Did the pager replace our slot with a new one? */
> - if (jb_lb->lifeboat_index == lb_idx)
> - {
> - /* No, it didn't. We can safely detach it now. */
> - jb_lb->lifeboat_index = -1;
> - }
> -
> - /* We always free the raw slot we just finished using */
> - lifeboat_free_slot (lb_idx);
> -
> - /* Mark it as written so checkpointing can advance!
> - Use the global notification system so ALL transactions
> - that contain this block are marked as written, preventing
> - the Active Checkpointer from overwriting fresh data with
> - stale shadow metadata from older checkpoint transactions. */
> - if (!err)
> - {
> - journal_notify_blocks_written_locked (jb_lb->jb_blocknr, 1);
> - }
> - JOURNAL_UNLOCK (journal);
> - }
> - }
> -}
> -
> /**
> * Ensures there is enough free space in the ring buffer to commit the
> * transaction. If space is dangerously low, it forces a synchronous
> checkpoint.
> @@ -2133,7 +2087,7 @@ journal_commit_running_transaction_locked (journal_t
> *journal)
> /* The WAL barrier is crossed! Tell the Pager it can write safely! */
> txn->t_state = T_COMMITTED;
> /* Flush any intercepted VM pager blocks to the primary disk */
> - journal_flush_lifeboat_payloads (journal, txn);
> + journal_flush_intercepted_payloads (journal, txn);
>
> int need_sb_flush = 0;
> /* IO done, lock again and finalize Metadata */
> @@ -2307,11 +2261,23 @@ journal_create (struct node *journal_inode)
> j->j_pool_memory[JRNL_MAX_FREE_BUFFERS - 1].jb_next = NULL;
> j->j_free_buffers = &j->j_pool_memory[0];
>
> - ext2_lifeboat.payloads =
> - mmap (NULL, JRNL_LIFEBOAT_CAPACITY * block_size, PROT_READ | PROT_WRITE,
> + j->j_intercept_pool =
> + mmap (NULL, JRNL_INTERCEPT_CAPACITY * block_size, PROT_READ | PROT_WRITE,
> MAP_ANON | MAP_PRIVATE, -1, 0);
> - if (ext2_lifeboat.payloads == MAP_FAILED)
> - ext2_panic ("[JOURNAL] No RAM for lifeboat cache!");
> + if (j->j_intercept_pool == MAP_FAILED)
> + ext2_panic ("[JOURNAL] No RAM for intercept cache!");
> +
> + /* Chain the 4KB chunks into a simple pointer-linked free list */
> + j->j_free_intercept_chunks = j->j_intercept_pool;
> + char *curr = j->j_intercept_pool;
> + for (int i = 0; i < JRNL_INTERCEPT_CAPACITY - 1; i++)
> + {
> + char *next = curr + block_size;
> + *(void **) curr = next;
> + curr = next;
> + }
> + *(void **) curr = NULL;
> +
> return j;
> }
>
> @@ -2460,8 +2426,8 @@ out:
>
> /**
> * Checks a block against active transactions and handles deadlock hazards.
> - * Returns 1 if the block was intercepted (written to lifeboat), 0 if it
> - * should be written to physical disk.
> + * Returns 1 if the block was intercepted, 0 if it should be written to
> + * physical disk.
> * MUST be called with JOURNAL_LOCK held.
> */
> static int
> @@ -2487,53 +2453,43 @@ retry:
> jb_commit = NULL;
>
> /* Hazard Interception: If it's trapped in an active transaction,
> - send it straight to the lifeboat and RETURN EARLY. Do not touch
> checkpoints. */
> + send it straight to the intercepted chunks and RETURN EARLY.
> + Do not touch checkpoints. */
> if (jb_run || jb_commit)
> {
> - int lb_idx_run = jb_run ? lifeboat_alloc_slot () : -1;
> - int lb_idx_commit = jb_commit ? lifeboat_alloc_slot () : -1;
> + if (jb_run)
> + {
> + char *chunk_run = jb_run->jb_intercepted_data;
> + if (!chunk_run)
> + {
> + chunk_run = journal_alloc_intercept_chunk (ext2_journal);
> + if (!chunk_run) goto pool_full;
> + }
> + memcpy (chunk_run, b_data, block_size);
> + jb_run->jb_intercepted_data = chunk_run;
> + }
> + else if (jb_commit)
> + {
> + char *chunk_commit = jb_commit->jb_intercepted_data;
> + if (!chunk_commit)
> + {
> + chunk_commit = journal_alloc_intercept_chunk (ext2_journal);
> + if (!chunk_commit) goto pool_full;
> + }
> + memcpy (chunk_commit, b_data, block_size);
> + jb_commit->jb_intercepted_data = chunk_commit;
> + }
> +
> + intercepted = 1;
> + JRNL_LOG_DEBUG ("Intercepted rushed pager write for block %u", b);
> + return intercepted;
>
> - if ((jb_run && lb_idx_run < 0) || (jb_commit && lb_idx_commit < 0))
> - {
> - if (lb_idx_run >= 0)
> - lifeboat_free_slot (lb_idx_run);
> - if (lb_idx_commit >= 0)
> - lifeboat_free_slot (lb_idx_commit);
> -
> - /* Failure: Lifeboat full. Trigger V4 WAL bypass */
> - JRNL_LOG_WARN
> - ("VM Deadlock & Lifeboat Full! Bypassing WAL for block %u.", b);
> - }
> - else
> - {
> - /* Success: Spoof the write directly into the Lifeboat */
> - if (jb_run)
> - {
> - memcpy (&(ext2_lifeboat.payloads)[lb_idx_run * block_size],
> - b_data, block_size);
> - /* If the old slot is NOT being flushed, we must free it to avoid
> a leak.
> - If it IS being flushed, the commit thread owns it and will
> free it. */
> - if (jb_run->lifeboat_index >= 0)
> - lifeboat_free_slot (jb_run->lifeboat_index);
> - jb_run->lifeboat_index = (int16_t) lb_idx_run;
> - }
> - if (jb_commit)
> - {
> - memcpy (&(ext2_lifeboat.payloads)[lb_idx_commit * block_size],
> - b_data, block_size);
> - if (jb_commit->lifeboat_index >= 0
> - && !jb_commit->jb_is_flushing)
> - lifeboat_free_slot (jb_commit->lifeboat_index);
> - jb_commit->lifeboat_index = (int16_t) lb_idx_commit;
> - }
> +pool_full:
> + JRNL_LOG_WARN
> + ("VM Deadlock & Intercept Pool Full! Bypassing WAL for block %u.",
> b);
>
> - intercepted = 1;
> - JRNL_LOG_DEBUG
> - ("Intercepted rushed pager write for block %u into Lifeboat slots
> (run:%d, commit:%d)",
> - b, lb_idx_run, lb_idx_commit);
> - }
> /* Do not claim any flushing flags! */
> - return intercepted;
> + return 0;
> }
>
> /* We are definitively going to physical disk.
> @@ -2674,7 +2630,7 @@ journal_clear_flushing_locked (block_t start_block,
> size_t n_blocks)
> * Replaces raw store_write calls to safely intercept deadlock hazards.
> * When the pager attempts to write a block that is actively locked by a
> * running or committing transaction (rushing the transaction commit cycle),
> - * this function redirects the payload into temporary storage (the Lifeboat
> + * this function redirects the payload into temporary storage (the intercept
> * cache). This preserves the Write-Ahead Log (WAL) ordering and prevents
> * the VM pager from deadlocking the filesystem. Safe blocks are coalesced
> * and written normally to disk.
> @@ -2709,7 +2665,7 @@ journal_store_write (block_t start_block, size_t
> length, void *buf,
> if (intercepted)
> {
> /* We successfully handled this block in RAM. Move to the next. */
> - JRNL_LOG_DEBUG ("Lifeboat intercepted hazard for block %u", b);
> + JRNL_LOG_DEBUG ("Intercepted hazard for block %u", b);
> i++;
> total_written += block_size;
> }
> @@ -2774,13 +2730,13 @@ journal_store_write (block_t start_block, size_t
> length, void *buf,
> }
>
> /**
> - * Overlays fresh Lifeboat cache data on top of a buffer that was just read
> + * Overlays fresh Intercept cache data on top of a buffer that was just read
> * from disk. This ensures Mach VM pointer and sub-block offset contracts
> * remain unbroken. Safely handles "short reads" at the end of devices
> without
> * overflowing the buffer.
> */
> static void
> -journal_overlay_lifeboat (block_t start_block, size_t length, void *buf)
> +journal_overlay_intercepted (block_t start_block, size_t length, void *buf)
> {
> if (!ext2_journal || length == 0 || !buf)
> return;
> @@ -2800,27 +2756,25 @@ journal_overlay_lifeboat (block_t start_block, size_t
> length, void *buf)
> journal_buffer_t *jb =
> run ? journal_map_lookup (&run->t_buffer_map, b) : NULL;
>
> - if (!jb && commit)
> + /* If the running txn doesn't have an intercept chunk,
> + fall back to checking the committing one! */
> + if (!(jb && jb->jb_intercepted_data) && commit)
> jb = journal_map_lookup (&commit->t_buffer_map, b);
>
> - if (jb && jb->lifeboat_index >= 0)
> + if (jb && jb->jb_intercepted_data)
> {
> /* Calculate exact offset and bounds for this specific block */
> size_t offset = i << log2_block_size;
> size_t copy_len = block_size;
> -
> /* If this is the final, partially-read block, clamp the copy length
> */
> if (length - offset < block_size)
> copy_len = length - offset;
>
> /* Overlay the fresh RAM data safely! */
> - memcpy (out_ptr + offset,
> - &(ext2_lifeboat.payloads)[jb->lifeboat_index * block_size],
> - copy_len);
> -
> + memcpy (out_ptr + offset, jb->jb_intercepted_data, copy_len);
> JRNL_LOG_DEBUG
> - ("Lifeboat Overlay successful for block %u (copied %zu bytes)", b,
> - copy_len);
> + ("Intercept Overlay successful for block %u (copied %zu bytes)",
> b,
> + copy_len);
> }
> }
> JOURNAL_UNLOCK (ext2_journal);
> @@ -2829,7 +2783,7 @@ journal_overlay_lifeboat (block_t start_block, size_t
> length, void *buf)
> /**
> * A block device filter layer for the VFS pager's read path.
> * Passes the read through to the underlying physical disk, and then
> - * transparently overlays any fresh data from the temporary Lifeboat cache.
> + * transparently overlays any fresh data from the temporary Intercept cache.
> * This ensures that reads of blocks which were recently intercepted and
> * redirected to temporary storage (due to rushing the transaction cycle)
> * return the most up-to-date data, maintaining strict cache coherence
> @@ -2845,7 +2799,7 @@ journal_store_read (block_t start_block, size_t length,
> void **buf,
> error_t err = store_read (store, dev_block, length, buf, read_amount);
> if (!err && ext2_journal && *read_amount > 0)
> /* Pass the actual amount read, just in case it was a short read */
> - journal_overlay_lifeboat (start_block, *read_amount, *buf);
> + journal_overlay_intercepted (start_block, *read_amount, *buf);
>
> return err;
> }
> --
> 2.55.0
>
--
Samuel
Who wants to remember that escape-x-alt-control-left shift-b puts you into
super-edit-debug-compile mode?
(Discussion in comp.os.linux.misc on the intuitiveness of commands, especially
Emacs.)