This patch replaces the fixed-size, bitmask-managed Lifeboat array with an
O(1) intrusive linked-list allocator tied directly to the transaction map.

Previously, when the Mach VM pager rushed a block locked by an active
transaction, the payload was intercepted and stored in a global array
indexed via an 8-slot `uint64_t` bitmask. This approach required expensive
index lookups, O(N) allocation loops, and compare-and-swap checks during
physical flushing to avoid race conditions.

By embedding an intercepted_data pointer directly into journal_buffer_t,
the intercepted memory is now exclusively owned by the transaction itself.
The global array is replaced with a simple, pre-allocated linked list of
4KB chunks, granting O(1) push/pop allocations and entirely eliminating
index collision hazards.

Cache coherence in journal_store_read is preserved by checking the
`RUNNING` transaction's pointers before the COMMITTING transaction's
pointers, natively shadowing older intercepted data.
---
 ext2fs/journal.c | 356 +++++++++++++++++++++--------------------------
 1 file changed, 155 insertions(+), 201 deletions(-)

diff --git a/ext2fs/journal.c b/ext2fs/journal.c
index cc3aeb35a..05448b2e4 100644
--- a/ext2fs/journal.c
+++ b/ext2fs/journal.c
@@ -121,28 +121,8 @@
 #define IS_RUNNING_TID(j, tid) \
     ((j)->j_running_transaction && (j)->j_running_transaction->t_tid == (tid))
 
-#define JRNL_LIFEBOAT_CAPACITY 512
+#define JRNL_INTERCEPT_CAPACITY 512
 
-#define JRNL_LIFEBOAT_ALLOC_MASK_LEN 8
-
-/* Temporary storage for blocks rushed by the Mach VM pager.
- * Because we cannot block or delay the pager when it needs to flush a page
- * belonging to an active (RUNNING/COMMITTING) transaction, this cache
- * absorbs the write. This prevents a permanent deadlock while preserving
- * Write-Ahead Log (WAL) ordering.
- * The payloads are flushed to disk as soon as the transaction safely commits.
- */
-struct journal_lifeboat
-{
-  /* 512 bits total: 0 means free, 1 means occupied.
-     Protected by the main ext2_journal->j_state_lock. */
-  uint64_t alloc_mask[JRNL_LIFEBOAT_ALLOC_MASK_LEN];
-
-  /* The pre-allocated payload pool (512 * 4KB = 2MB) */
-  char *payloads;
-};
-
-static struct journal_lifeboat ext2_lifeboat;
 static pthread_t kjournald_tid;
 
 /* The handle this thread currently holds, and how many starts are open
@@ -181,8 +161,8 @@ typedef struct journal_buffer
   uint8_t needs_copy;          /* Whether this buffer needs a new copy from
                                   from the live Mach VM cache. Should be 1
                                   when new. */
-  /* -1 if normal, 0-127 if holding a spoofed payload in the lifeboat */
-  int16_t lifeboat_index;
+  /* Pointer to the intercepted data */
+  char *jb_intercepted_data;
   uint8_t jb_is_flushing;      /* 1 if commit thread is actively flushing it. 
*/
   uint8_t jb_escaped;
 } journal_buffer_t;
@@ -310,53 +290,40 @@ typedef struct journal
   /* Pre-allocated buffers for (near) zero-allocation journal_dirty_block */
   journal_buffer_t *j_pool_memory;     /* The raw contiguous block */
   journal_buffer_t *j_free_buffers;    /* The linked list head */
+
+  /* Intercepted data pool pointers */
+  char *j_intercept_pool;
+  void *j_free_intercept_chunks;
 } journal_t;
 
 /**
- * Returns 0-127 on success, or -1 if the lifeboat is full.
- * MUST be called with JOURNAL_LOCK(ext2_journal) held.
+ * O(1) Intercept Chunk Allocation.
+ * MUST be called with JOURNAL_LOCK held.
  */
-static inline int
-lifeboat_alloc_slot (void)
+static inline char *
+journal_alloc_intercept_chunk (journal_t *journal)
 {
-  for (int i = 0; i < JRNL_LIFEBOAT_ALLOC_MASK_LEN; i++)
+  if (journal->j_free_intercept_chunks)
     {
-      /* Invert mask: 1s now represent FREE slots */
-      uint64_t free_bits = ~ext2_lifeboat.alloc_mask[i];
-
-      if (free_bits != 0)
-       {
-         /* __builtin_ffsll returns 1-64, so we subtract 1 for 0-based index */
-         int bit = __builtin_ffsll ((long long) free_bits) - 1;
-         ext2_lifeboat.alloc_mask[i] |= (1ULL << bit);
-         return (int) (i * sizeof (ext2_lifeboat.alloc_mask[0]) * 8) + bit;
-       }
+      char *chunk = (char *) journal->j_free_intercept_chunks;
+      /* The first 8 bytes of the free chunk hold the 'next' pointer */
+      journal->j_free_intercept_chunks = *(void **) chunk;
+      return chunk;
     }
-  return -1;
+  return NULL;                 /* Starvation fallback */
 }
 
 /**
- * Frees a raw slot back to the pool.
- */
-static inline void
-lifeboat_free_slot (int index)
-{
-  if (index >= 0 && index < JRNL_LIFEBOAT_CAPACITY)
-    ext2_lifeboat.alloc_mask[index / 64] &= ~(1ULL << (index % 64));
-}
-
-/**
- * Frees a slot back to the pool.
- * MUST be called with JOURNAL_LOCK(ext2_journal) held.
+ * O(1) Intercept Chunk Deallocation.
+ * MUST be called with JOURNAL_LOCK held.
  */
 static inline void
-lifeboat_release_buffer (journal_buffer_t *jb)
+journal_free_intercept_chunk (journal_t *journal, char *chunk)
 {
-  int index = jb->lifeboat_index;
-  if (index >= 0 && index < JRNL_LIFEBOAT_CAPACITY)
+  if (chunk)
     {
-      ext2_lifeboat.alloc_mask[index / 64] &= ~(1ULL << (index % 64));
-      jb->lifeboat_index = -1;
+      *(void **) chunk = journal->j_free_intercept_chunks;
+      journal->j_free_intercept_chunks = chunk;
     }
 }
 
@@ -600,7 +567,7 @@ journal_alloc_buffer (journal_t *journal)
   if (!jb)
     return NULL;
 out:
-  jb->lifeboat_index = -1;
+  jb->jb_intercepted_data = NULL;
   return jb;
 }
 
@@ -611,14 +578,15 @@ out:
 static inline void
 journal_free_buffer (journal_t *journal, journal_buffer_t *jb)
 {
-  /* Check if this pointer falls inside our contiguous pool block */
-  if (jb->lifeboat_index >= 0)
-    lifeboat_release_buffer (jb);
-
   uintptr_t ptr = (uintptr_t) jb;
   uintptr_t start = (uintptr_t) journal->j_pool_memory;
   uintptr_t end = (uintptr_t) & journal->j_pool_memory[JRNL_MAX_FREE_BUFFERS];
 
+  if (jb->jb_intercepted_data)
+    {
+      journal_free_intercept_chunk (journal, jb->jb_intercepted_data);
+      jb->jb_intercepted_data = NULL;
+    }
   if (ptr >= start && ptr < end)
     {
       /* It belongs to the permanent pool. Link it back up! */
@@ -1062,7 +1030,7 @@ journal_stop_transaction_locked (journal_t *journal,
        {
          if (jb_exp->needs_copy)
            {
-             /* ALWAYS hydrate from the live VM cache. The lifeboat is for
+             /* ALWAYS hydrate from the live VM cache. The intercept is for
                 delayed physical I/O, not for sourcing WAL shadow data! */
              /* Clear the flag here, under the lock, before the copy.  A
                 dirty that lands while we copy unlocked sets it again and
@@ -1196,7 +1164,7 @@ journal_notify_blocks_written_locked (block_t 
start_block, size_t n_blocks)
    * stale relative to the final shadow copy.
    *
    * Blocks in the running transaction that legitimately need to be marked as
-   * written are handled by journal_flush_lifeboat_payloads(), which runs AFTER
+   * written are handled by journal_flush_intercepted_payloads(), which runs 
AFTER
    * the WAL barrier is crossed (T_COMMITTED) and does its own marking.
    *
    * The committing transaction and checkpoint list are safe to notify: their
@@ -1376,6 +1344,61 @@ journal_is_block_in_newer_transaction_locked (journal_t 
*journal,
   return 0;
 }
 
+/* Flush any intercepted VM pager blocks to the primary disk
+   Executes completely outside the global journal lock (manages its own lock 
per-block).
+   Called immediately after a transaction is safely committed to the WAL.  */
+static void
+journal_flush_intercepted_payloads (journal_t *journal,
+                                   diskfs_transaction_t *txn)
+{
+  size_t iter = 0;
+  journal_buffer_t *jb;
+
+  while ((jb = journal_map_iterate (&txn->t_buffer_map, &iter)) != NULL)
+    {
+      char *flush_data = NULL;
+
+      JOURNAL_LOCK (journal);
+      if (jb->jb_intercepted_data)
+       {
+         flush_data = jb->jb_intercepted_data;
+         jb->jb_is_flushing = 1;
+       }
+      JOURNAL_UNLOCK (journal);
+
+      if (flush_data)
+       {
+         store_offset_t dev_block =
+           (store_offset_t) jb->jb_blocknr << log2_dev_blocks_per_fs_block;
+         size_t amount;
+         error_t err;
+
+         /* Shadow data is fully settled and committed. */
+         err =
+           store_write (store, dev_block, jb->jb_shadow_data, block_size, 
&amount);
+
+         JOURNAL_LOCK (journal);
+         if (err)
+           JRNL_LOG_WARN ("Intercept flush failed for block %u: %s",
+                          jb->jb_blocknr, strerror (err));
+
+         jb->jb_is_flushing = 0;
+
+         /* We always free the chunk we just finished using */
+         journal_free_intercept_chunk (journal, flush_data);
+
+         /* If the pager didn't replace it with a new write, clear it. */
+         if (jb->jb_intercepted_data == flush_data)
+           jb->jb_intercepted_data = NULL;
+
+         if (!err)
+           journal_notify_blocks_written_locked (jb->jb_blocknr, 1);
+
+         JOURNAL_UNLOCK (journal);
+       }
+    }
+}
+
 /**
  * Internal helper to flush checkpoint transactions to the main filesystem.
  * If target_free is UINT32_MAX, it flushes ALL transactions in the list.
@@ -1607,9 +1630,11 @@ journal_dirty_block_locked (diskfs_transaction_t *txn, 
block_t fs_blocknr)
          jb->jb_is_written = 0;
          txn->t_outstanding_io++;
        }
-      if (jb->lifeboat_index >= 0)
-       lifeboat_release_buffer (jb);
-
+      if (jb->jb_intercepted_data)
+       {
+         journal_free_intercept_chunk (ext2_journal, jb->jb_intercepted_data);
+         jb->jb_intercepted_data = NULL;
+       }
       goto out;
     }
 
@@ -1945,77 +1970,6 @@ journal_write_commit_record (journal_t *journal,
   return journal_write_block (journal, commit_loc, commit_buf);
 }
 
-/**
- * Flushes any intercepted pager writes (Lifeboat payloads) to the main 
filesystem.
- * Executes completely outside the global journal lock (manages its own lock 
per-block).
- * Called immediately after a transaction is safely committed to the WAL.
- */
-static void
-journal_flush_lifeboat_payloads (journal_t *journal,
-                                diskfs_transaction_t *txn)
-{
-  size_t iter = 0;
-  journal_buffer_t *jb_lb;
-
-  while ((jb_lb = journal_map_iterate (&txn->t_buffer_map, &iter)) != NULL)
-    {
-      int lb_idx = -1;
-
-      JOURNAL_LOCK (journal);
-      if (jb_lb->lifeboat_index >= 0)
-       {
-         lb_idx = jb_lb->lifeboat_index;
-         jb_lb->jb_is_flushing = 1;    /* Mark as actively flushing! */
-       }
-      JOURNAL_UNLOCK (journal);
-
-      if (lb_idx >= 0)
-       {
-         store_offset_t dev_block =
-           (store_offset_t) jb_lb->jb_blocknr <<
-           log2_dev_blocks_per_fs_block;
-         size_t amount;
-         error_t err;
-
-         /* We do the I/O using our safely captured, privately owned index */
-         err = store_write (store, dev_block,
-                            &(ext2_lifeboat.payloads)[lb_idx * block_size],
-                            block_size, &amount);
-
-         JOURNAL_LOCK (journal);
-         if (err)
-           {
-             JRNL_LOG_WARN
-               ("Lifeboat flush failed for block %u: %s",
-                jb_lb->jb_blocknr, strerror (err));
-           }
-
-         jb_lb->jb_is_flushing = 0;    /* Done flushing (even if it failed) */
-
-         /* Compare-and-Swap: Did the pager replace our slot with a new one? */
-         if (jb_lb->lifeboat_index == lb_idx)
-           {
-             /* No, it didn't. We can safely detach it now. */
-             jb_lb->lifeboat_index = -1;
-           }
-
-         /* We always free the raw slot we just finished using */
-         lifeboat_free_slot (lb_idx);
-
-         /* Mark it as written so checkpointing can advance!
-            Use the global notification system so ALL transactions
-            that contain this block are marked as written, preventing
-            the Active Checkpointer from overwriting fresh data with
-            stale shadow metadata from older checkpoint transactions. */
-         if (!err)
-           {
-             journal_notify_blocks_written_locked (jb_lb->jb_blocknr, 1);
-           }
-         JOURNAL_UNLOCK (journal);
-       }
-    }
-}
-
 /**
  * Ensures there is enough free space in the ring buffer to commit the
  * transaction. If space is dangerously low, it forces a synchronous 
checkpoint.
@@ -2133,7 +2087,7 @@ journal_commit_running_transaction_locked (journal_t 
*journal)
   /* The WAL barrier is crossed! Tell the Pager it can write safely! */
   txn->t_state = T_COMMITTED;
   /* Flush any intercepted VM pager blocks to the primary disk */
-  journal_flush_lifeboat_payloads (journal, txn);
+  journal_flush_intercepted_payloads (journal, txn);
 
   int need_sb_flush = 0;
   /* IO done, lock again and finalize Metadata */
@@ -2307,11 +2261,23 @@ journal_create (struct node *journal_inode)
   j->j_pool_memory[JRNL_MAX_FREE_BUFFERS - 1].jb_next = NULL;
   j->j_free_buffers = &j->j_pool_memory[0];
 
-  ext2_lifeboat.payloads =
-    mmap (NULL, JRNL_LIFEBOAT_CAPACITY * block_size, PROT_READ | PROT_WRITE,
+  j->j_intercept_pool =
+    mmap (NULL, JRNL_INTERCEPT_CAPACITY * block_size, PROT_READ | PROT_WRITE,
          MAP_ANON | MAP_PRIVATE, -1, 0);
-  if (ext2_lifeboat.payloads == MAP_FAILED)
-    ext2_panic ("[JOURNAL] No RAM for lifeboat cache!");
+  if (j->j_intercept_pool == MAP_FAILED)
+    ext2_panic ("[JOURNAL] No RAM for intercept cache!");
+
+  /* Chain the 4KB chunks into a simple pointer-linked free list */
+  j->j_free_intercept_chunks = j->j_intercept_pool;
+  char *curr = j->j_intercept_pool;
+  for (int i = 0; i < JRNL_INTERCEPT_CAPACITY - 1; i++)
+    {
+      char *next = curr + block_size;
+      *(void **) curr = next;
+      curr = next;
+    }
+  *(void **) curr = NULL;
+
   return j;
 }
 
@@ -2460,8 +2426,8 @@ out:
 
 /**
  * Checks a block against active transactions and handles deadlock hazards.
- * Returns 1 if the block was intercepted (written to lifeboat), 0 if it
- * should be written to physical disk.
+ * Returns 1 if the block was intercepted, 0 if it should be written to
+ * physical disk.
  * MUST be called with JOURNAL_LOCK held.
  */
 static int
@@ -2487,53 +2453,43 @@ retry:
     jb_commit = NULL;
 
   /* Hazard Interception: If it's trapped in an active transaction,
-     send it straight to the lifeboat and RETURN EARLY. Do not touch 
checkpoints. */
+     send it straight to the intercepted chunks and RETURN EARLY.
+     Do not touch checkpoints. */
   if (jb_run || jb_commit)
     {
-      int lb_idx_run = jb_run ? lifeboat_alloc_slot () : -1;
-      int lb_idx_commit = jb_commit ? lifeboat_alloc_slot () : -1;
+      if (jb_run)
+        {
+          char *chunk_run = jb_run->jb_intercepted_data;
+          if (!chunk_run)
+            {
+              chunk_run = journal_alloc_intercept_chunk (ext2_journal);
+              if (!chunk_run) goto pool_full;
+            }
+          memcpy (chunk_run, b_data, block_size);
+          jb_run->jb_intercepted_data = chunk_run;
+        }
+      else if (jb_commit)
+        {
+          char *chunk_commit = jb_commit->jb_intercepted_data;
+          if (!chunk_commit)
+            {
+              chunk_commit = journal_alloc_intercept_chunk (ext2_journal);
+              if (!chunk_commit) goto pool_full;
+            }
+          memcpy (chunk_commit, b_data, block_size);
+          jb_commit->jb_intercepted_data = chunk_commit;
+        }
+
+      intercepted = 1;
+      JRNL_LOG_DEBUG ("Intercepted rushed pager write for block %u", b);
+      return intercepted;
 
-      if ((jb_run && lb_idx_run < 0) || (jb_commit && lb_idx_commit < 0))
-       {
-         if (lb_idx_run >= 0)
-           lifeboat_free_slot (lb_idx_run);
-         if (lb_idx_commit >= 0)
-           lifeboat_free_slot (lb_idx_commit);
-
-         /* Failure: Lifeboat full. Trigger V4 WAL bypass */
-         JRNL_LOG_WARN
-           ("VM Deadlock & Lifeboat Full! Bypassing WAL for block %u.", b);
-       }
-      else
-       {
-         /* Success: Spoof the write directly into the Lifeboat */
-         if (jb_run)
-           {
-             memcpy (&(ext2_lifeboat.payloads)[lb_idx_run * block_size],
-                     b_data, block_size);
-             /* If the old slot is NOT being flushed, we must free it to avoid 
a leak.
-                If it IS being flushed, the commit thread owns it and will 
free it. */
-             if (jb_run->lifeboat_index >= 0)
-               lifeboat_free_slot (jb_run->lifeboat_index);
-             jb_run->lifeboat_index = (int16_t) lb_idx_run;
-           }
-         if (jb_commit)
-           {
-             memcpy (&(ext2_lifeboat.payloads)[lb_idx_commit * block_size],
-                     b_data, block_size);
-             if (jb_commit->lifeboat_index >= 0
-                 && !jb_commit->jb_is_flushing)
-               lifeboat_free_slot (jb_commit->lifeboat_index);
-             jb_commit->lifeboat_index = (int16_t) lb_idx_commit;
-           }
+pool_full:
+      JRNL_LOG_WARN
+        ("VM Deadlock & Intercept Pool Full! Bypassing WAL for block %u.", b);
 
-         intercepted = 1;
-         JRNL_LOG_DEBUG
-           ("Intercepted rushed pager write for block %u into Lifeboat slots 
(run:%d, commit:%d)",
-            b, lb_idx_run, lb_idx_commit);
-       }
       /* Do not claim any flushing flags! */
-      return intercepted;
+      return 0;
     }
 
   /* We are definitively going to physical disk.
@@ -2674,7 +2630,7 @@ journal_clear_flushing_locked (block_t start_block, 
size_t n_blocks)
  * Replaces raw store_write calls to safely intercept deadlock hazards.
  * When the pager attempts to write a block that is actively locked by a
  * running or committing transaction (rushing the transaction commit cycle),
- * this function redirects the payload into temporary storage (the Lifeboat
+ * this function redirects the payload into temporary storage (the intercept
  * cache). This preserves the Write-Ahead Log (WAL) ordering and prevents
  * the VM pager from deadlocking the filesystem. Safe blocks are coalesced
  * and written normally to disk.
@@ -2709,7 +2665,7 @@ journal_store_write (block_t start_block, size_t length, 
void *buf,
       if (intercepted)
        {
          /* We successfully handled this block in RAM. Move to the next. */
-         JRNL_LOG_DEBUG ("Lifeboat intercepted hazard for block %u", b);
+         JRNL_LOG_DEBUG ("Intercepted hazard for block %u", b);
          i++;
          total_written += block_size;
        }
@@ -2774,13 +2730,13 @@ journal_store_write (block_t start_block, size_t 
length, void *buf,
 }
 
 /**
- * Overlays fresh Lifeboat cache data on top of a buffer that was just read
+ * Overlays fresh Intercept cache data on top of a buffer that was just read
  * from disk. This ensures Mach VM pointer and sub-block offset contracts
  * remain unbroken. Safely handles "short reads" at the end of devices without
  * overflowing the buffer.
  */
 static void
-journal_overlay_lifeboat (block_t start_block, size_t length, void *buf)
+journal_overlay_intercepted (block_t start_block, size_t length, void *buf)
 {
   if (!ext2_journal || length == 0 || !buf)
     return;
@@ -2800,27 +2756,25 @@ journal_overlay_lifeboat (block_t start_block, size_t 
length, void *buf)
       journal_buffer_t *jb =
        run ? journal_map_lookup (&run->t_buffer_map, b) : NULL;
 
-      if (!jb && commit)
+      /* If the running txn doesn't have an intercept chunk,
+        fall back to checking the committing one! */
+      if (!(jb && jb->jb_intercepted_data) && commit)
        jb = journal_map_lookup (&commit->t_buffer_map, b);
 
-      if (jb && jb->lifeboat_index >= 0)
+      if (jb && jb->jb_intercepted_data)
        {
          /* Calculate exact offset and bounds for this specific block */
          size_t offset = i << log2_block_size;
          size_t copy_len = block_size;
-
          /* If this is the final, partially-read block, clamp the copy length 
*/
          if (length - offset < block_size)
            copy_len = length - offset;
 
          /* Overlay the fresh RAM data safely! */
-         memcpy (out_ptr + offset,
-                 &(ext2_lifeboat.payloads)[jb->lifeboat_index * block_size],
-                 copy_len);
-
+         memcpy (out_ptr + offset, jb->jb_intercepted_data, copy_len);
          JRNL_LOG_DEBUG
-           ("Lifeboat Overlay successful for block %u (copied %zu bytes)", b,
-            copy_len);
+           ("Intercept Overlay successful for block %u (copied %zu bytes)", b,
+            copy_len);
        }
     }
   JOURNAL_UNLOCK (ext2_journal);
@@ -2829,7 +2783,7 @@ journal_overlay_lifeboat (block_t start_block, size_t 
length, void *buf)
 /**
  * A block device filter layer for the VFS pager's read path.
  * Passes the read through to the underlying physical disk, and then
- * transparently overlays any fresh data from the temporary Lifeboat cache.
+ * transparently overlays any fresh data from the temporary Intercept cache.
  * This ensures that reads of blocks which were recently intercepted and
  * redirected to temporary storage (due to rushing the transaction cycle)
  * return the most up-to-date data, maintaining strict cache coherence
@@ -2845,7 +2799,7 @@ journal_store_read (block_t start_block, size_t length, 
void **buf,
   error_t err = store_read (store, dev_block, length, buf, read_amount);
   if (!err && ext2_journal && *read_amount > 0)
     /* Pass the actual amount read, just in case it was a short read */
-    journal_overlay_lifeboat (start_block, *read_amount, *buf);
+    journal_overlay_intercepted (start_block, *read_amount, *buf);
 
   return err;
 }
-- 
2.55.0


Reply via email to