diskfs_orphan_add cleared i_size, i_size_high, i_blocks and i_block[] of
the on-disk inode, and write_node left those fields alone for as long as
the node stayed on the orphan list. The orphan's blocks then appeared
nowhere on disk except in the bitmaps. At the next mount
ext2_recover_orphan_list read an inode with no size and no blocks, so
diskfs_drop_node freed the inode and left its blocks allocated; e2fsck
cleared the orphan the same way and freed the blocks only in pass 5.
Every file still open when the filesystem shut down leaked its blocks
until the next full check.
ext3 and ext4 keep the orphan's block map, size and block count on disk
and overload only i_dtime as the link to the next orphan. Do the same.
diskfs_orphan_add now only links the inode into the list and clears
i_links_count; write_node writes i_size, i_blocks and i_block[] for an
orphan as for any other inode and leaves only i_dtime to orphan.c. The
truncate at the last close rewrites the block map in the transaction
that frees the blocks, and recovery truncates whatever the orphan still
owns.
The fields were cleared so that a block freed and reused while its
inode was on the orphan list could not be reached through the orphan's
block map. Freed blocks now stay busy until their free commits, so the
block map of an orphan only ever names blocks it still owns, and the
workaround is no longer needed.
---
ext2fs/inode.c | 38 ++++++++++++++++++++------------------
ext2fs/orphan.c | 19 ++++++-------------
2 files changed, 26 insertions(+), 31 deletions(-)
diff --git a/ext2fs/inode.c b/ext2fs/inode.c
index 21256f843..e525d306f 100644
--- a/ext2fs/inode.c
+++ b/ext2fs/inode.c
@@ -494,31 +494,33 @@ write_node (struct node *np)
info->i_flags |= EXT2_IMMUTABLE_FL;
di->i_flags = htole32 (info->i_flags);
- /* The i_dtime and other fields here are used by the orphan machinery
- so we don't need to touch them here if a node is an orphan. */
+ /* While NP is on the orphan list, i_dtime holds the next orphan's
+ inode number, and only orphan.c writes it. */
if (!diskfs_node_disknode (np)->on_orphan_list)
{
if (st->st_mode == 0)
/* Set dtime non-zero to indicate a deleted file. */
di->i_dtime = htole32 (di->i_mtime);
else
- {
- /* We don't clear i_size, i_blocks, and i_translator if mode is
0,
- to give "undeletion" utilities a chance. */
- di->i_dtime = htole32 (0);
- di->i_size = htole32 (st->st_size);
- if (sizeof (off_t) >= 8 && !S_ISDIR (st->st_mode))
- /* 64bit file size */
- di->i_size_high = htole32 (st->st_size >> 32);
- di->i_blocks = htole32 (st->st_blocks);
- }
-
- if (S_ISCHR(st->st_mode) || S_ISBLK(st->st_mode))
- di->i_block[0] = htole32 (st->st_rdev);
- else
- memcpy (di->i_block, diskfs_node_disknode (np)->info.i_data,
- EXT2_N_BLOCKS * sizeof di->i_block[0]);
+ di->i_dtime = htole32 (0);
}
+
+ /* We don't clear i_size, i_blocks, and i_translator if mode is 0,
+ to give "undeletion" utilities a chance. */
+ if (st->st_mode != 0)
+ {
+ di->i_size = htole32 (st->st_size);
+ if (sizeof (off_t) >= 8 && !S_ISDIR (st->st_mode))
+ /* 64bit file size */
+ di->i_size_high = htole32 (st->st_size >> 32);
+ di->i_blocks = htole32 (st->st_blocks);
+ }
+
+ if (S_ISCHR(st->st_mode) || S_ISBLK(st->st_mode))
+ di->i_block[0] = htole32 (st->st_rdev);
+ else
+ memcpy (di->i_block, diskfs_node_disknode (np)->info.i_data,
+ EXT2_N_BLOCKS * sizeof di->i_block[0]);
diskfs_end_catch_exception ();
np->dn_stat_dirty = 0;
diff --git a/ext2fs/orphan.c b/ext2fs/orphan.c
index 24f10e7d9..9923d4331 100644
--- a/ext2fs/orphan.c
+++ b/ext2fs/orphan.c
@@ -70,9 +70,11 @@ diskfs_orphan_add (struct node *np)
return;
}
- /* write_node must see this before it copies info.i_data. While set:
- leave i_dtime alone, and do not copy i_data, i_size, or i_blocks.
- libdiskfs must call diskfs_node_update as soon as this function returns.
*/
+ /* write_node must see this before it next writes the inode. While set,
+ it leaves i_dtime alone and writes the rest of the inode as usual, so
+ the block map on disk always matches the blocks NP still owns and
+ recovery can truncate it. libdiskfs must call diskfs_node_update as
+ soon as this function returns. */
diskfs_node_disknode (np)->on_orphan_list = 1;
ext2_debug ("adding inode %lu to orphan list", (unsigned long) inum);
@@ -88,15 +90,7 @@ diskfs_orphan_add (struct node *np)
sblock_dirty = 1;
pthread_spin_unlock (&global_lock);
- /* Isolate the inode from standard file system deletion logic.
- Zeroing the block map here prevents the Mach pager from flushing garbage
or
- cross-linked block pointers to the disk before the journal commits. */
di->i_links_count = 0;
- di->i_size = 0;
- di->i_blocks = 0;
- if (!S_ISDIR (np->dn_stat.st_mode))
- di->i_size_high = 0;
- memset (di->i_block, 0, EXT2_N_BLOCKS * sizeof di->i_block[0]);
/* Let the journal know we are done editing. */
journal_mark_dirty (txn, boffs_block (bptr_offs (di)));
@@ -172,8 +166,7 @@ diskfs_orphan_del (struct node *np)
{
struct ext2_inode *prev_di = dino_ref (prev->cache_id);
- /* prev stays on the list, so its cached i_block[] is already the
- placeholder (zeros). Only i_dtime changes. */
+ /* prev stays on the list; only its i_dtime changes. */
journal_get_write_access (txn, boffs_block (bptr_offs (prev_di)));
prev_di->i_dtime = htole32 (my_next);
journal_mark_dirty (txn, boffs_block (bptr_offs (prev_di)));
--
2.56.0