When journaling is active, forcing metadata to the physical
platter via sync_global in diskfs_set_hypermetadata fights the Write-Ahead Log
and artificially floods the journal's intercept cache with rushed pager writes.
This patch skips sync_global when the journal is present, instead using
diskfs_journal_set_sync() to guarantee durability through the WAL.
Relying on lazy pager evictions, however, exposes a race during unmount.
If a pageout completes after diskfs_journal_shutdown sets j_must_exit,
journal_notify_blocks_written_locked previously returned immediately.
This left jb_is_flushing set, causing journal_quiesce_checkpoints to wait
on j_flush_wait forever.
The notification path is updated to properly release flushing claims while
j_must_exit is set, without altering the checkpoint list that the quiesce
routine is actively walking.
---
ext2fs/hyper.c | 30 ++++++++++++++++++++++--------
ext2fs/journal.c | 16 +++++++++++++++-
2 files changed, 37 insertions(+), 9 deletions(-)
diff --git a/ext2fs/hyper.c b/ext2fs/hyper.c
index 6d6fe7d71..35cdb7ed0 100644
--- a/ext2fs/hyper.c
+++ b/ext2fs/hyper.c
@@ -193,6 +193,9 @@ map_hypermetadata (void)
error_t
diskfs_set_hypermetadata (int wait, int clean)
{
+ error_t err = 0;
+ diskfs_transaction_t *txn = diskfs_journal_start_transaction ();
+
if (clean)
{
/* Always clear recovery flag on clean unmount if journal is present */
@@ -229,7 +232,10 @@ diskfs_set_hypermetadata (int wait, int clean)
if (sblock_dirty)
{
if (diskfs_readonly)
- return EROFS; /* impossible to write */
+ {
+ err = EROFS; /* impossible to write */
+ goto out;
+ }
/* Before writing, set the time of write */
sblock->s_wtime = htole32 (diskfs_mtime->seconds);
@@ -239,15 +245,23 @@ diskfs_set_hypermetadata (int wait, int clean)
record_global_poke (mapped_sblock);
}
- sync_global (wait);
- if (wait)
+ if (!ext2_journal)
{
- error_t err = store_sync (store);
- /* Ignore EOPNOTSUPP (legacy drivers), but warn on real I/O errors */
- if (err && err != EOPNOTSUPP && err != D_INVALID_OPERATION)
- ext2_warning ("device flush failed: %s", strerror (err));
+ sync_global (wait);
+ if (wait)
+ {
+ error_t err = store_sync (store);
+ /* Ignore EOPNOTSUPP (legacy drivers), but warn on real I/O errors */
+ if (err && err != EOPNOTSUPP && err != D_INVALID_OPERATION)
+ ext2_warning ("device flush failed: %s", strerror (err));
+ }
}
- return 0;
+ else if (wait)
+ diskfs_journal_set_sync (txn);
+
+out:
+ diskfs_journal_stop_transaction (txn);
+ return err;
}
void
diff --git a/ext2fs/journal.c b/ext2fs/journal.c
index 05448b2e4..6828aa4d8 100644
--- a/ext2fs/journal.c
+++ b/ext2fs/journal.c
@@ -1143,9 +1143,16 @@ journal_notify_blocks_written_locked (block_t
start_block, size_t n_blocks)
{
int sb_changed = 0;
error_t err = 0;
- if (!ext2_journal || n_blocks == 0 || ext2_journal->j_must_exit)
+ if (!ext2_journal || n_blocks == 0)
return 0;
+ /* On shutdown journal_quiesce_checkpoints owns the checkpoint list and
+ drops the lock for I/O while it walks it, so nothing here may unlink or
+ free a transaction. The blocks are still marked written and their
+ jb_is_flushing claims released: quiesce waits on j_flush_wait for any
+ block a pager write has claimed. */
+ int exiting = ext2_journal->j_must_exit;
+
JRNL_LOG_DEBUG ("Got notification for %zu blocks starting at %u",
n_blocks, start_block);
@@ -1179,6 +1186,13 @@ journal_notify_blocks_written_locked (block_t
start_block, size_t n_blocks)
diskfs_transaction_t *txn = ext2_journal->j_checkpoint_list;
while (txn)
{
+ if (exiting)
+ {
+ journal_notify_txn_locked (txn, start_block, n_blocks);
+ txn = txn->t_checkpoint_next;
+ continue;
+ }
+
/* Fast-path cleanup for empty transactions lingering at the head */
if (txn->t_outstanding_io == 0
&& txn == ext2_journal->j_checkpoint_list)
--
2.56.0