ext2_free_blocks clears the bitmap bit at once, so ext2_new_block could
hand the block to a new owner while the transaction that freed it was
still running or committing. The journal can hold copies of the
block's old contents in that window: in the freeing transaction, in the
committing one, and in older checkpoint transactions. A home write of
such a copy lands over the new owner's data. A crash before the free
commits also gives the block back to its old owner after the new owner
has overwritten it.
ext3 and ext4 close this window in the allocator: a block freed in a
transaction is allocated again only after that transaction commits. Do
the same. The bitmap and the free counts still change at free time, so
the transaction carries the free as before. The block is also marked
in a per-group busy bitmap kept in memory, and ext2_new_block searches
the block bitmap with the busy blocks marked in use. Group selection
counts only the free blocks that are not busy.
journal_record_freed_blocks now reports whether it recorded the range.
After the commit, journal_forget_freed_blocks marks the copies in the
freeing transaction and in older checkpoint transactions written, waits
for any write of them already in flight, and hands the blocks back with
ext2_release_busy_blocks. Since no freed block has a new owner before
that point, every copy it finds holds the old contents.
When a freed range cannot be tracked for lack of memory, it stays
allocated: a leaked block only costs space until e2fsck frees it, while
reusing it could put an old copy over its new owner.
Discarded preallocations never held data. They go through the new
ext2_free_unused_blocks, which leaves them allocatable at once.
When every free block is busy, ext2_new_block waits for the transaction
committing now to hand its blocks back, and retries. It cannot wait for
the running transaction, which the caller's handle holds open, so it
wakes kjournald to start that commit as soon as the handle drains. A
pager thread never waits. If a commit fails, its freed blocks stay busy
until the next mount.
Without a journal nothing is recorded, and freed blocks are reusable at
once as before.
A copy of a freed block in an older transaction that is still in the
log can be replayed over the block's new owner after a crash. Revoke
records are needed for that; the XXX comment in trunc_indirect now says
so.
---
ext2fs/balloc.c | 206 ++++++++++++++++++++++++++++++++++++++++------
ext2fs/ext2fs.h | 8 ++
ext2fs/getblk.c | 3 +-
ext2fs/journal.c | 164 ++++++++++++++++++++++++++++--------
ext2fs/journal.h | 13 ++-
ext2fs/truncate.c | 19 ++---
6 files changed, 336 insertions(+), 77 deletions(-)
diff --git a/ext2fs/balloc.c b/ext2fs/balloc.c
index 48273e14e..1a1418b9c 100644
--- a/ext2fs/balloc.c
+++ b/ext2fs/balloc.c
@@ -40,6 +40,7 @@
* when a file system is mounted (see ext2_read_super).
*/
+#include <stdlib.h>
#include <string.h>
#include "journal.h"
#include "ext2fs.h"
@@ -55,8 +56,117 @@ memscan (void *buf, unsigned char ch, size_t len)
#define in_range(b, first, len) ((b) >= (first) && (b) <= (first) + (len) - 1)
+/* Blocks freed in a transaction that has not committed yet are busy. The
+ bitmap shows them free, but the journal can still hold copies of their
+ old contents, and a crash before the free commits gives them back to
+ their old owner. ext2_new_block skips them until the journal hands
+ them back with ext2_release_busy_blocks. Protected by global_lock. */
+static unsigned char **busy_maps; /* A bitmap per group, or NULL */
+static uint32_t *busy_counts; /* Busy blocks in each group */
+static unsigned long busy_total;
+static unsigned char *busy_scratch; /* Bitmap ext2_new_block searches */
+
+/* Make sure group BLOCK_GROUP has a busy bitmap. Returns 0 without
+ memory for it. */
+static int
+busy_map_ready (unsigned long block_group)
+{
+ if (!busy_maps)
+ {
+ busy_maps = calloc (groups_count, sizeof *busy_maps);
+ busy_counts = calloc (groups_count, sizeof *busy_counts);
+ busy_scratch = malloc (block_size);
+ if (!busy_maps || !busy_counts || !busy_scratch)
+ {
+ free (busy_maps);
+ free (busy_counts);
+ free (busy_scratch);
+ busy_maps = NULL;
+ busy_counts = NULL;
+ busy_scratch = NULL;
+ return 0;
+ }
+ }
+
+ if (!busy_maps[block_group])
+ busy_maps[block_group] = calloc (1, block_size);
+ return busy_maps[block_group] != NULL;
+}
+
+/* Mark bit BIT of group BLOCK_GROUP busy. busy_map_ready must have
+ succeeded for the group. */
+static void
+mark_busy (unsigned long block_group, unsigned long bit)
+{
+ if (!set_bit (bit, busy_maps[block_group]))
+ {
+ busy_counts[block_group]++;
+ busy_total++;
+ }
+}
+
+/* Returns the free blocks of group BLOCK_GROUP that ext2_new_block can
+ hand out. */
+static inline unsigned long
+group_allocatable (unsigned long block_group, struct ext2_group_desc *gdp)
+{
+ unsigned long free_count = le16toh (gdp->bg_free_blocks_count);
+ unsigned long busy = busy_counts ? busy_counts[block_group] : 0;
+
+ return free_count > busy ? free_count - busy : 0;
+}
+
+/* Returns the bitmap ext2_new_block searches in group BLOCK_GROUP: the
+ block bitmap BH, or a copy of it with the group's busy blocks marked
+ in use. */
+static unsigned char *
+search_map (unsigned long block_group, unsigned char *bh)
+{
+ uint32_t *dst, *src, *busy;
+ size_t i;
+
+ if (!busy_counts || busy_counts[block_group] == 0)
+ return bh;
+
+ dst = (uint32_t *) busy_scratch;
+ src = (uint32_t *) bh;
+ busy = (uint32_t *) busy_maps[block_group];
+ for (i = 0; i < block_size / sizeof (uint32_t); i++)
+ dst[i] = src[i] | busy[i];
+ return busy_scratch;
+}
+
void
-ext2_free_blocks (block_t block, unsigned long count)
+ext2_release_busy_blocks (block_t block, unsigned long count)
+{
+ pthread_spin_lock (&global_lock);
+ if (busy_maps)
+ for (; count > 0; block++, count--)
+ {
+ unsigned long block_group =
+ (block - le32toh (sblock->s_first_data_block)) /
+ le32toh (sblock->s_blocks_per_group);
+ unsigned long bit =
+ (block - le32toh (sblock->s_first_data_block)) %
+ le32toh (sblock->s_blocks_per_group);
+
+ if (busy_maps[block_group]
+ && clear_bit (bit, busy_maps[block_group]))
+ {
+ busy_counts[block_group]--;
+ busy_total--;
+ }
+ }
+ pthread_spin_unlock (&global_lock);
+}
+
+/* Free COUNT blocks starting at BLOCK. With BUSY set and a journal, the
+ blocks stay busy until the transaction freeing them commits. When they
+ cannot be kept busy, they stay allocated instead: reusing them could put
+ a journal copy of their old contents over their new owner, while a
+ leaked block only costs space until e2fsck frees it. */
+static void
+free_blocks (block_t block, unsigned long count, int busy)
{
unsigned char *bh;
unsigned long block_group;
@@ -127,19 +237,27 @@ ext2_free_blocks (block_t block, unsigned long count)
"block = %u, count = %lu",
block, count);
- journal_record_freed_blocks (txn, block, gcount);
- for (i = 0; i < gcount; i++)
- {
- if (!clear_bit (bit + i, bh))
- ext2_warning ("bit already cleared for block %lu", block + i);
- else
- {
- gdp->bg_free_blocks_count =
- htole16 (le16toh (gdp->bg_free_blocks_count) + 1);
- sblock->s_free_blocks_count =
- htole32 (le32toh (sblock->s_free_blocks_count) + 1);
- }
- }
+ int hold = busy && ext2_journal && txn;
+ int leak = hold && (!busy_map_ready (block_group)
+ || !journal_record_freed_blocks (txn, block, gcount));
+ if (leak)
+ ext2_warning ("no memory to keep freed blocks busy; leaving "
+ "blocks %u[%lu] allocated", block, gcount);
+ else
+ for (i = 0; i < gcount; i++)
+ {
+ if (!clear_bit (bit + i, bh))
+ ext2_warning ("bit already cleared for block %lu", block + i);
+ else
+ {
+ gdp->bg_free_blocks_count =
+ htole16 (le16toh (gdp->bg_free_blocks_count) + 1);
+ sblock->s_free_blocks_count =
+ htole32 (le32toh (sblock->s_free_blocks_count) + 1);
+ if (hold)
+ mark_busy (block_group, bit + i);
+ }
+ }
record_global_poke (bh);
disk_cache_block_ref_ptr (gdp);
@@ -156,6 +274,18 @@ ext2_free_blocks (block_t block, unsigned long count)
alloc_sync (0);
}
+void
+ext2_free_blocks (block_t block, unsigned long count)
+{
+ free_blocks (block, count, 1);
+}
+
+void
+ext2_free_unused_blocks (block_t block, unsigned long count)
+{
+ free_blocks (block, count, 0);
+}
+
/*
* ext2_new_block uses a goal block to assist allocation. If the goal is
* free, or there is a free block within 32 blocks of the goal, that block
@@ -169,6 +299,7 @@ ext2_new_block (block_t goal,
block_t *prealloc_count, block_t *prealloc_block)
{
unsigned char *bh = NULL;
+ unsigned char *map = NULL;
unsigned char *p, *r;
int i, j, k, tmp;
uint32_t lmap;
@@ -205,7 +336,7 @@ repeat:
i = (goal - le32toh (sblock->s_first_data_block)) /
le32toh (sblock->s_blocks_per_group);
gdp = group_desc (i);
- if (le16toh (gdp->bg_free_blocks_count) > 0)
+ if (group_allocatable (i, gdp) > 0)
{
j = ((goal - le32toh (sblock->s_first_data_block))
% le32toh (sblock->s_blocks_per_group));
@@ -214,10 +345,11 @@ repeat:
goal_attempts++;
#endif
bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap));
+ map = search_map (i, bh);
ext2_debug ("goal is at %d:%d", i, j);
- if (!test_bit (j, bh))
+ if (!test_bit (j, map))
{
#ifdef EXT2FS_DEBUG
goal_hits++;
@@ -234,11 +366,11 @@ repeat:
if ((j & 31) == 31)
lmap = 0;
else
- lmap = ((((uint32_t *) bh)[j >> 5]) >>
+ lmap = ((((uint32_t *) map)[j >> 5]) >>
((j & 31) + 1));
if (j < le32toh (sblock->s_blocks_per_group) - 32)
- lmap |= (((uint32_t *) bh)[(j >> 5) + 1]) <<
+ lmap |= (((uint32_t *) map)[(j >> 5) + 1]) <<
(31 - (j & 31));
else
lmap |= 0xffffffffu << (31 - (j & 31));
@@ -264,15 +396,15 @@ repeat:
* Search first in the remainder of the current group; then,
* cyclicly search through the rest of the groups.
*/
- p = bh + (j >> 3);
+ p = map + (j >> 3);
r = memscan (p, 0, (le32toh (sblock->s_blocks_per_group) - j + 7) >> 3);
- k = (r - bh) << 3;
+ k = (r - map) << 3;
if (k < le32toh (sblock->s_blocks_per_group))
{
j = k;
goto search_back;
}
- k = find_next_zero_bit ((uint32_t *) bh,
+ k = find_next_zero_bit ((uint32_t *) map,
le32toh (sblock->s_blocks_per_group),
j);
if (k < le32toh (sblock->s_blocks_per_group))
@@ -296,22 +428,42 @@ repeat:
if (i >= groups_count)
i = 0;
gdp = group_desc (i);
- if (le16toh (gdp->bg_free_blocks_count) > 0)
+ if (group_allocatable (i, gdp) > 0)
break;
}
if (k >= groups_count)
{
+ unsigned long busy = busy_total;
+
pthread_spin_unlock (&global_lock);
+ /* The free space can be busy until the transaction committing now
+ finishes.
+
+ XXX TODO: The caller can hold node locks here, such as alloc_lock
+ from diskfs_grow. A pager thread that joined the committing
+ transaction (pager_unlock_page) and then waits for that alloc_lock
+ keeps the commit from draining, and this wait never ends. Either
+ wait only when the caller holds no node lock, or return ENOSPC and
+ let the caller retry after dropping its locks, as ext4 does. When
+ the busy blocks all belong to the caller's own transaction, nothing
+ waits and the caller gets ENOSPC although space comes back at the
+ next commit. */
+ if (busy > 0 && journal_wait_freed_blocks (txn))
+ {
+ pthread_spin_lock (&global_lock);
+ goto repeat;
+ }
return 0;
}
assert_backtrace (bh == NULL);
bh = disk_cache_block_ref (le32toh (gdp->bg_block_bitmap));
- r = memscan (bh, 0, le32toh (sblock->s_blocks_per_group) >> 3);
- j = (r - bh) << 3;
+ map = search_map (i, bh);
+ r = memscan (map, 0, le32toh (sblock->s_blocks_per_group) >> 3);
+ j = (r - map) << 3;
if (j < le32toh (sblock->s_blocks_per_group))
goto search_back;
else
- j = find_first_zero_bit ((uint32_t *) bh,
+ j = find_first_zero_bit ((uint32_t *) map,
le32toh (sblock->s_blocks_per_group));
if (j >= le32toh (sblock->s_blocks_per_group))
{
@@ -328,7 +480,7 @@ search_back:
* bitmap. Now search backwards up to 7 bits to find the
* start of this group of free blocks.
*/
- for (k = 0; k < 7 && j > 0 && !test_bit (j - 1, bh); k++, j--);
+ for (k = 0; k < 7 && j > 0 && !test_bit (j - 1, map); k++, j--);
got_block:
assert_backtrace (bh != NULL);
@@ -376,7 +528,7 @@ got_block:
for (k = 1;
k < prealloc_goal && (j + k) < le32toh (sblock->s_blocks_per_group);
k++)
{
- if (set_bit (j + k, bh))
+ if (test_bit (j + k, map) || set_bit (j + k, bh))
break;
(*prealloc_count)++;
diff --git a/ext2fs/ext2fs.h b/ext2fs/ext2fs.h
index 458446984..8aabc6434 100644
--- a/ext2fs/ext2fs.h
+++ b/ext2fs/ext2fs.h
@@ -700,6 +700,14 @@ block_t ext2_new_block (block_t goal,
block_t *prealloc_count, block_t *prealloc_block);
void ext2_free_blocks (block_t block, unsigned long count);
+
+/* Free COUNT blocks starting at BLOCK that never held data, such as
+ discarded preallocations. ext2_new_block can hand them out at once. */
+void ext2_free_unused_blocks (block_t block, unsigned long count);
+
+/* Make COUNT blocks starting at BLOCK, freed by a transaction that has
+ now committed, available to ext2_new_block again. */
+void ext2_release_busy_blocks (block_t block, unsigned long count);
/* ---------------------------------------------------------------- */
diff --git a/ext2fs/getblk.c b/ext2fs/getblk.c
index 39ad31da0..404aa62e1 100644
--- a/ext2fs/getblk.c
+++ b/ext2fs/getblk.c
@@ -55,7 +55,8 @@ ext2_discard_prealloc (struct node *node)
ext2_debug ("discarding %d prealloced blocks for inode %d",
i, node->cache_id);
diskfs_node_disknode (node)->info.i_prealloc_count = 0;
- ext2_free_blocks (diskfs_node_disknode (node)->info.i_prealloc_block, i);
+ ext2_free_unused_blocks
+ (diskfs_node_disknode (node)->info.i_prealloc_block, i);
}
#endif
}
diff --git a/ext2fs/journal.c b/ext2fs/journal.c
index 85dde2ad5..91323f3da 100644
--- a/ext2fs/journal.c
+++ b/ext2fs/journal.c
@@ -279,6 +279,8 @@ typedef struct journal
uint32_t j_min_free;
uint32_t j_last_committed_tid; /* Transaction ID of the last committed
txn. */
+ uint32_t j_released_tid; /* Last txn whose freed blocks went back
+ to the allocator. */
pthread_cond_t j_commit_done; /* Cond. var. while waiting for the tx
to be committed. */
int j_must_exit; /* variable that tells journal thread when to
stop. */
@@ -874,21 +876,24 @@ journal_get_oldest_transaction_locked (journal_t *journal)
/**
* Records a range of deleted blocks so they can be unpinned from older
- * checkpoint lists AFTER this transaction safely commits.
+ * checkpoint lists AFTER this transaction safely commits. Returns 1 if
+ * the range was recorded; journal_forget_freed_blocks then hands it back
+ * to the allocator. Returns 0 if it was not, and the caller must then not
+ * let the blocks be reused.
*/
-void
+int
journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
unsigned long count)
{
if (!ext2_journal || !txn)
- return;
+ return 0;
journal_freed_extent_t *ext = malloc (sizeof (journal_freed_extent_t));
if (!ext)
{
JRNL_LOG_WARN
- ("ENOMEM tracking freed blocks. Harmless I/O overhead may occur.");
- return;
+ ("ENOMEM tracking freed blocks %u[%lu].", start, count);
+ return 0;
}
ext->fe_start = start;
@@ -902,11 +907,13 @@ journal_record_freed_blocks (diskfs_transaction_t *txn,
block_t start,
We just drop the recording, since the block is already forgotten. */
JOURNAL_UNLOCK (ext2_journal);
free (ext);
- return;
+ return 0;
}
ext->fe_next = txn->t_freed_blocks;
txn->t_freed_blocks = ext;
+
JOURNAL_UNLOCK (ext2_journal);
+ return 1;
}
/**
@@ -1131,6 +1138,31 @@ journal_notify_txn_locked (diskfs_transaction_t *txn,
return txn->t_outstanding_io == 0;
}
+/* Record the tail in the journal superblock after
+ journal_try_advance_tail_locked moved it. MUST be called with
+ JOURNAL_LOCK held. Returns 1 if the superblock needs a flush. */
+static int
+journal_record_tail_locked (journal_t *journal)
+{
+ uint32_t tail_seq;
+ diskfs_transaction_t *oldest =
+ journal_get_oldest_transaction_locked (journal);
+
+ if (oldest)
+ tail_seq = oldest->t_tid;
+ else
+ tail_seq = journal->j_transaction_sequence;
+
+ /* Update Superblock Persistently */
+ error_t err = journal_update_superblock (journal, tail_seq);
+ if (err)
+ {
+ JRNL_LOG_WARN ("Failed to update superblock. %s", strerror (err));
+ return 0;
+ }
+ return 1;
+}
+
/**
* Called just after blocks have been written to the main disk.
*/
@@ -1138,7 +1170,6 @@ static int
journal_notify_blocks_written_locked (block_t start_block, size_t n_blocks)
{
int sb_changed = 0;
- error_t err = 0;
if (!ext2_journal || n_blocks == 0)
return 0;
@@ -1210,44 +1241,73 @@ journal_notify_blocks_written_locked (block_t
start_block, size_t n_blocks)
txn = txn->t_checkpoint_next;
}
- if (sb_changed)
- {
- uint32_t tail_seq;
- diskfs_transaction_t *oldest =
- journal_get_oldest_transaction_locked (ext2_journal);
-
- if (oldest)
- tail_seq = oldest->t_tid;
- else
- tail_seq = ext2_journal->j_transaction_sequence;
-
- /* Update Superblock Persistently */
- err = journal_update_superblock (ext2_journal, tail_seq);
- if (err)
- JRNL_LOG_WARN ("Failed to update superblock. %s", strerror (err));
- }
- return (sb_changed && !err) ? 1 : 0;
+ return sb_changed ? journal_record_tail_locked (ext2_journal) : 0;
}
/**
- * Consumes the freed blocks list and deallocates them.
+ * Consumes the freed blocks list of transaction FREED_TID once it has
+ * committed. The copies of those blocks in FREED_TID and in older
+ * checkpoint transactions hold the old contents, so they need no home
+ * write anymore and are marked written. The allocator keeps the blocks
+ * busy until here, so no copy can belong to a new owner yet. The blocks
+ * then go back to the allocator.
*/
static void
-journal_forget_freed_blocks (journal_t *journal, journal_freed_extent_t *ext)
+journal_forget_freed_blocks (journal_t *journal, uint32_t freed_tid,
+ journal_freed_extent_t *freed)
{
int flush_needed = 0;
+ journal_freed_extent_t *ext;
+
JOURNAL_LOCK (journal);
- while (ext)
- {
- journal_freed_extent_t *next = ext->fe_next;
- if (journal_notify_blocks_written_locked (ext->fe_start, ext->fe_count))
- flush_needed = 1;
- free (ext);
- ext = next;
- }
+ for (ext = freed; ext; ext = ext->fe_next)
+ {
+ restart:
+ for (diskfs_transaction_t *txn = journal->j_checkpoint_list;
+ txn && (int32_t) (txn->t_tid - freed_tid) <= 0;
+ txn = txn->t_checkpoint_next)
+ for (unsigned long i = 0; i < ext->fe_count; i++)
+ {
+ block_t b = ext->fe_start + i;
+ journal_buffer_t *jb =
+ journal_map_lookup (&txn->t_buffer_map, b);
+ if (!jb)
+ continue;
+ if (jb->jb_is_flushing)
+ {
+ /* A write of the old contents is in flight. It must land
+ before the block has a new owner. The wait drops the
+ lock, and the checkpoint list can change meanwhile. */
+ pthread_cond_wait (&journal->j_flush_wait,
+ &journal->j_state_lock);
+ goto restart;
+ }
+ journal_notify_txn_locked (txn, b, 1);
+ }
+ }
+
+ /* On shutdown journal_quiesce_checkpoints owns the checkpoint list. */
+ if (!journal->j_must_exit && journal_try_advance_tail_locked (journal)
+ && journal_record_tail_locked (journal))
+ flush_needed = 1;
JOURNAL_UNLOCK (journal);
if (flush_needed)
flush_to_disk ();
+
+ /* ext2_new_block takes the journal lock inside global_lock, so the
+ blocks go back to the allocator with the journal lock dropped. */
+ while (freed)
+ {
+ ext = freed->fe_next;
+ ext2_release_busy_blocks (freed->fe_start, freed->fe_count);
+ free (freed);
+ freed = ext;
+ }
+
+ JOURNAL_LOCK (journal);
+ journal->j_released_tid = freed_tid;
+ pthread_cond_broadcast (&journal->j_commit_done);
+ JOURNAL_UNLOCK (journal);
}
/* Install a running transaction with t_active_threads == 0.
@@ -1607,6 +1667,35 @@ journal_wait_on_tid_locked (journal_t *journal, uint32_t
target_tid)
JOURNAL_WAIT (&journal->j_commit_done, journal);
}
+/* A pager thread never waits. Neither does a thread whose handle is on
+ the committing transaction, since that commit waits for the handle to
+ drain. The running transaction's frees wait for its commit, which the
+ caller's handle holds back; kjournald is woken to start that commit as
+ soon as the handle drains. */
+int
+journal_wait_freed_blocks (diskfs_transaction_t *txn)
+{
+ diskfs_transaction_t *commit;
+ int waited = 0;
+
+ if (!ext2_journal || journal_thread_is_pager)
+ return 0;
+
+ JOURNAL_LOCK (ext2_journal);
+ pthread_cond_signal (&ext2_journal->j_flusher_wakeup);
+ commit = ext2_journal->j_committing_transaction;
+ if (commit && commit != txn)
+ {
+ uint32_t tid = commit->t_tid;
+
+ while (tid_gt (tid, ext2_journal->j_released_tid))
+ JOURNAL_WAIT (&ext2_journal->j_commit_done, ext2_journal);
+ waited = 1;
+ }
+ JOURNAL_UNLOCK (ext2_journal);
+ return waited;
+}
+
void
journal_mark_dirty (diskfs_transaction_t *txn, block_t fs_blocknr)
{
@@ -2187,6 +2276,7 @@ journal_commit_running_transaction_locked (journal_t
*journal)
txn->t_checkpoint_next = NULL;
journal->j_committing_transaction = NULL;
journal_freed_extent_t *freed_extents = txn->t_freed_blocks;
+ uint32_t freed_tid = txn->t_tid;
txn->t_freed_blocks = NULL;
if (journal->j_checkpoint_last)
@@ -2201,7 +2291,7 @@ journal_commit_running_transaction_locked (journal_t
*journal)
if (need_sb_flush)
flush_to_disk ();
- journal_forget_freed_blocks (journal, freed_extents);
+ journal_forget_freed_blocks (journal, freed_tid, freed_extents);
JOURNAL_LOCK (journal);
goto out;
abort_commit:
@@ -2210,6 +2300,9 @@ abort_commit:
JOURNAL_LOCK (journal);
journal->j_committing_transaction = NULL;
journal->j_last_committed_tid = txn->t_tid;
+ /* The frees did not commit, and older copies of their blocks were not
+ forgotten. The blocks stay busy until the next mount. */
+ journal->j_released_tid = txn->t_tid;
pthread_cond_broadcast (&journal->j_commit_done);
journal_free_transaction (txn);
out:
@@ -2305,6 +2398,7 @@ journal_create (struct node *journal_inode)
if (journal_load_superblock (j) != 0)
ext2_panic ("[JOURNAL] Failed to load superblock!");
j->j_last_committed_tid = j->j_transaction_sequence - 1;
+ j->j_released_tid = j->j_last_committed_tid;
pthread_cond_init (&j->j_commit_done, NULL);
pthread_mutex_init (&j->j_state_lock, NULL);
pthread_cond_init (&j->j_commit_wait, NULL);
diff --git a/ext2fs/journal.h b/ext2fs/journal.h
index 3cc8212b1..06460445d 100644
--- a/ext2fs/journal.h
+++ b/ext2fs/journal.h
@@ -82,9 +82,18 @@ journal_store_read (block_t start_block, size_t length, void
**buf,
/**
* Records a range of deleted blocks so they can be unpinned from older
- * checkpoint lists AFTER this transaction safely commits.
+ * checkpoint lists AFTER this transaction safely commits. Returns 1 if
+ * the range was recorded: the journal then hands it back with
+ * ext2_release_busy_blocks once the commit is done.
*/
-void journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
unsigned long count);
+int journal_record_freed_blocks (diskfs_transaction_t *txn, block_t start,
unsigned long count);
+
+/**
+ * Waits until the transaction committing now has handed its freed blocks
+ * back to the allocator. TXN is the caller's handle. Returns 1 after
+ * such a wait, 0 if there was nothing the caller can wait for.
+ */
+int journal_wait_freed_blocks (diskfs_transaction_t *txn);
/**
* Marks the calling thread as running a pager callback (ON = 1) or done
diff --git a/ext2fs/truncate.c b/ext2fs/truncate.c
index 665b83443..837fd02c3 100644
--- a/ext2fs/truncate.c
+++ b/ext2fs/truncate.c
@@ -140,21 +140,16 @@ trunc_indirect (struct node *node, block_t end,
modified = 1;
}
- /* Reserve the block only if it survives. A block freed here must
- not be in the transaction: ext2_new_block can hand it to a new
- file before the free commits, the new owner's pager writes are then
- intercepted, and the post-commit flush writes the shadow, the old
- block pointers, over the file's data. Reserving after the edit is
+ /* Reserve the block only if it survives. A block freed here needs
+ no copy in the journal: its contents are dead, and ext2_free_blocks
+ keeps it busy until the free commits. Reserving after the edit is
safe because this thread's handle keeps the stop-time sweep from
copying the block before the handle is released.
- XXX TODO: Freed blocks are reusable before their free commits, so
- a copy of an old owner's contents can still reach a reused block
- (a block allocated and freed in one transaction, or a copy in the
- committing or checkpoint transactions). Like ext3/ext4, the
- allocator should keep freed blocks busy until the freeing
- transaction commits, and the journal should write revoke records
- so replay does not overwrite a reused block. */
+ XXX TODO: Older transactions still in the log can hold copies of a
+ freed block, and replay after a crash writes them over the block's
+ new owner. The journal should write revoke records, as ext3/ext4
+ do. */
if (first == 0 && all_freed)
{
pager_flush_some (diskfs_disk_pager,
--
2.56.0