mirror of
https://github.com/versity/scoutfs.git
synced 2026-08-21 06:36:49 +00:00
scoutfs: fix btree split/join setting parent keys
Before the introduction of the AVL tree to sort btree items, the items were sorted by sorting a small packed array of offsets. The final offset in that array pointed to the item in the block with the greatest key. With the move to sorting items in an AVL tree by nodes embedded in item structs, we now don't have the array of offsets and instead have a dense array of items. Creation and deletion of items always works with the final item in the array. last_item() used to return the item with the greatest key by returning the item pointed to by the final entry in the sorted offset array, then it returned the final entry in the item array for creation and deletion but that was no longer the item with the greatest key. But spliting and joining still used last_item() to find the item in the block with the greatest key for updating references to blocks in parents. Since the introduction of the AVL tree splitting and joining has been corrrupting the tree by setting parent block reference keys to whatever item happened to be at the end of the array, not the item with the greatest key. The extent code recently pushed hard enough to hit this by working with relatively random extent items in the core allocation btrees. Eventually the parent block reference keys got out of sync and we'd fail to find items by descending into the wrong children when looking for them. Extent deletion hit this during allocation, returned -ENOENT, and the allocator turned that into -ENOSPC. With this fixed we can repetedly create and delte millions of files with heavily fragmented extents in a tiny metadata device. Eventually it actually runs out of space instead of spuriously returning ENOSPC in a matter of minutes. Signed-off-by: Zach Brown <zab@versity.com>
This commit is contained in:
+25
-15
@@ -140,7 +140,12 @@ off_item(struct scoutfs_btree_block *bt, __le16 off)
|
||||
return (void *)bt + le16_to_cpu(off);
|
||||
}
|
||||
|
||||
static struct scoutfs_btree_item *last_item(struct scoutfs_btree_block *bt)
|
||||
|
||||
/*
|
||||
* The item at the end of the item array. This is *not* the item in the
|
||||
* block with the greatest key.
|
||||
*/
|
||||
static struct scoutfs_btree_item *end_item(struct scoutfs_btree_block *bt)
|
||||
{
|
||||
BUG_ON(bt->nr_items == 0);
|
||||
|
||||
@@ -183,6 +188,11 @@ static struct scoutfs_btree_item *node_item(struct scoutfs_avl_node *node)
|
||||
return container_of(node, struct scoutfs_btree_item, node);
|
||||
}
|
||||
|
||||
static struct scoutfs_btree_item *last_item(struct scoutfs_btree_block *bt)
|
||||
{
|
||||
return node_item(scoutfs_avl_last(&bt->item_root));
|
||||
}
|
||||
|
||||
static struct scoutfs_btree_item *prev_item(struct scoutfs_btree_block *bt,
|
||||
struct scoutfs_btree_item *item)
|
||||
{
|
||||
@@ -543,7 +553,7 @@ static void create_item(struct scoutfs_btree_block *bt,
|
||||
le16_add_cpu(&bt->mid_free_len,
|
||||
-(u16)sizeof(struct scoutfs_btree_item));
|
||||
le16_add_cpu(&bt->nr_items, 1);
|
||||
item = last_item(bt);
|
||||
item = end_item(bt);
|
||||
|
||||
item->key = *key;
|
||||
|
||||
@@ -568,14 +578,14 @@ static void delete_item(struct scoutfs_btree_block *bt,
|
||||
struct scoutfs_btree_item *item,
|
||||
struct scoutfs_btree_item **use_after)
|
||||
{
|
||||
struct scoutfs_btree_item *last;
|
||||
struct scoutfs_btree_item *end;
|
||||
unsigned int val_off;
|
||||
unsigned int val_len;
|
||||
|
||||
/* save some values before we delete the item */
|
||||
val_off = le16_to_cpu(item->val_off);
|
||||
val_len = le16_to_cpu(item->val_len);
|
||||
last = last_item(bt);
|
||||
end = end_item(bt);
|
||||
|
||||
/* delete the item */
|
||||
scoutfs_avl_delete(&bt->item_root, &item->node);
|
||||
@@ -585,18 +595,18 @@ static void delete_item(struct scoutfs_btree_block *bt,
|
||||
le16_add_cpu(&bt->total_item_bytes, -item_bytes(item));
|
||||
|
||||
/* move the final item into the deleted space */
|
||||
if (last != item) {
|
||||
item->key = last->key;
|
||||
item->val_off = last->val_off;
|
||||
item->val_len = last->val_len;
|
||||
if (last->val_len)
|
||||
set_val_owner(bt, le16_to_cpu(last->val_off),
|
||||
val_bytes(le16_to_cpu(last->val_len)),
|
||||
if (end != item) {
|
||||
item->key = end->key;
|
||||
item->val_off = end->val_off;
|
||||
item->val_len = end->val_len;
|
||||
if (end->val_len)
|
||||
set_val_owner(bt, le16_to_cpu(end->val_off),
|
||||
val_bytes(le16_to_cpu(end->val_len)),
|
||||
ptr_off(bt, item));
|
||||
leaf_item_hash_change(bt, &last->key, ptr_off(bt, item),
|
||||
ptr_off(bt, last));
|
||||
scoutfs_avl_relocate(&bt->item_root, &item->node,&last->node);
|
||||
if (use_after && *use_after == last)
|
||||
leaf_item_hash_change(bt, &end->key, ptr_off(bt, item),
|
||||
ptr_off(bt, end));
|
||||
scoutfs_avl_relocate(&bt->item_root, &item->node,&end->node);
|
||||
if (use_after && *use_after == end)
|
||||
*use_after = item;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user