When journaling is active, forcing metadata to the physical
platter via sync_global in diskfs_set_hypermetadata fights the Write-Ahead Log
and artificially floods the journal's intercept cache with rushed pager writes.

This patch skips sync_global when the journal is present, instead using
diskfs_journal_set_sync() to guarantee durability through the WAL.

Relying on lazy pager evictions, however, exposes a race during unmount.
If a pageout completes after diskfs_journal_shutdown sets j_must_exit,
journal_notify_blocks_written_locked previously returned immediately.
This left jb_is_flushing set, causing journal_quiesce_checkpoints to wait
on j_flush_wait forever.

The notification path is updated to properly release flushing claims while
j_must_exit is set, without altering the checkpoint list that the quiesce
routine is actively walking.
---
 ext2fs/hyper.c   | 30 ++++++++++++++++++++++--------
 ext2fs/journal.c | 16 +++++++++++++++-
 2 files changed, 37 insertions(+), 9 deletions(-)

diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c
index 6d6fe7d71..35cdb7ed0 100644
--- a/ext2fs/hyper.c
+++ b/ext2fs/hyper.c
@@ -193,6 +193,9 @@ map_hypermetadata (void)
 error_t
 diskfs_set_hypermetadata (int wait, int clean)
 {
+  error_t err = 0;
+  diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
+
   if (clean)
     {
       /* Always clear recovery flag on clean unmount if journal is present */
@@ -229,7 +232,10 @@ diskfs_set_hypermetadata (int wait, int clean)
   if (sblock_dirty)
     {
       if (diskfs_readonly)
-        return EROFS; /* impossible to write */
+        {
+          err = EROFS; /* impossible to write */
+          goto out;
+        }
 
       /* Before writing, set the time of write */
       sblock->s_wtime = htole32 (diskfs_mtime->seconds);
@@ -239,15 +245,23 @@ diskfs_set_hypermetadata (int wait, int clean)
       record_global_poke (mapped_sblock);
     }
 
-  sync_global (wait);
-  if (wait)
+  if (!ext2_journal)
     {
-      error_t err = store_sync (store);
-      /* Ignore EOPNOTSUPP (legacy drivers), but warn on real I/O errors */
-      if (err && err != EOPNOTSUPP && err != D_INVALID_OPERATION)
-        ext2_warning ("device flush failed: %s", strerror (err));
+      sync_global (wait);
+      if (wait)
+       {
+         error_t err = store_sync (store);
+         /* Ignore EOPNOTSUPP (legacy drivers), but warn on real I/O errors */
+         if (err && err != EOPNOTSUPP && err != D_INVALID_OPERATION)
+           ext2_warning ("device flush failed: %s", strerror (err));
+       }
     }
-  return 0;
+  else if (wait)
+    diskfs_journal_set_sync (txn);
+
+out:
+  diskfs_journal_stop_transaction (txn);
+  return err;
 }
 
 void
diff --git a/ext2fs/journal.c b/ext2fs/journal.c
index 05448b2e4..6828aa4d8 100644
--- a/ext2fs/journal.c
+++ b/ext2fs/journal.c
@@ -1143,9 +1143,16 @@ journal_notify_blocks_written_locked (block_t 
start_block, size_t n_blocks)
 {
   int sb_changed = 0;
   error_t err = 0;
-  if (!ext2_journal || n_blocks == 0 || ext2_journal->j_must_exit)
+  if (!ext2_journal || n_blocks == 0)
     return 0;
 
+  /* On shutdown journal_quiesce_checkpoints owns the checkpoint list and
+     drops the lock for I/O while it walks it, so nothing here may unlink or
+     free a transaction.  The blocks are still marked written and their
+     jb_is_flushing claims released: quiesce waits on j_flush_wait for any
+     block a pager write has claimed.  */
+  int exiting = ext2_journal->j_must_exit;
+
   JRNL_LOG_DEBUG ("Got notification for %zu blocks starting at %u",
                  n_blocks, start_block);
 
@@ -1179,6 +1186,13 @@ journal_notify_blocks_written_locked (block_t 
start_block, size_t n_blocks)
   diskfs_transaction_t *txn = ext2_journal->j_checkpoint_list;
   while (txn)
     {
+      if (exiting)
+       {
+         journal_notify_txn_locked (txn, start_block, n_blocks);
+         txn = txn->t_checkpoint_next;
+         continue;
+       }
+
       /* Fast-path cleanup for empty transactions lingering at the head */
       if (txn->t_outstanding_io == 0
          && txn == ext2_journal->j_checkpoint_list)
-- 
2.56.0


Reply via email to