mirror of
https://github.com/versity/scoutfs.git
synced 2026-09-28 18:55:39 +00:00
scoutfs: use packed extents and bitmaps
The btree forest item storage doesn't have as much item granular state as the item cache did. The item cache could tell if a cached item was populated from persistent storage or was created in memory. It could simply remove created items rather than leaving behind a deletion item. The cached btree blocks in the btree forest item storage mechanism can't do this. It has to create deletion items when deleting newly created items because it doesn't know if the item already exists in the persistent record or not. This created a problem with the extent storage we were using. The individual extent items were stored with a key set to the last logical block of their extent. As extents grew or shrank they often were deleted and created at different key values during a transaction. In the btree forest log trees this left a huge stream of deletion items beind, one for every previous version of the extent. Then searches for an extent covering a block would have to skip over all these deleted items before hitting the current stored extent. Streaming writes would operate on O(n) for every extent operation. It got to be out of hand. This large change solves the problem by using more coarse and stable item storage to track free blocks and blocks mapped into file data. For file data we now have large packed extent items which store packed representations of all the logical mappings of a fixed region of a file. The data code has loading and storage functions which transfer that persistent version to and from the version that is modified in memory. Free blocks are stored in bitmaps that are similarly efficiently packed into fixed size items. The client is no longer working with free extent items managed by the forest, it's working with free block bitmap btrees directly. It needs access to the client's metadata block allocator and block write contexts so we move those two out of the forest code and up into the transaction. Previously the client and server would exchange extents with network messages. Now the roots of the btrees that store the free block bitmap items are communicated along with the roots of the other trees involved in a transaction. The client doesn't need to send free extents back to the server so we can remove those tasks and rpcs. The server no longer has to manage free extents. It transfers block bitmap items between trees around commits. All of its extent manipulation can be removed. The item size portion of transaction item counts are removed because we're not using that level of granularity now that metadata transactions are dirty btree blocks instead of dirty items we pack into fixed sized segments. Signed-off-by: Zach Brown <zab@versity.com>
This commit is contained in:
+57
-6
@@ -25,6 +25,8 @@
|
||||
#include "counters.h"
|
||||
#include "client.h"
|
||||
#include "inode.h"
|
||||
#include "balloc.h"
|
||||
#include "block.h"
|
||||
#include "scoutfs_trace.h"
|
||||
|
||||
/*
|
||||
@@ -60,6 +62,10 @@ struct trans_info {
|
||||
unsigned reserved_vals;
|
||||
unsigned holders;
|
||||
bool writing;
|
||||
|
||||
struct scoutfs_log_trees lt;
|
||||
struct scoutfs_balloc_allocator alloc;
|
||||
struct scoutfs_block_writer wri;
|
||||
};
|
||||
|
||||
#define DECLARE_TRANS_INFO(sb, name) \
|
||||
@@ -77,6 +83,48 @@ static bool drained_holders(struct trans_info *tri)
|
||||
return drained;
|
||||
}
|
||||
|
||||
static int commit_btrees(struct super_block *sb)
|
||||
{
|
||||
DECLARE_TRANS_INFO(sb, tri);
|
||||
struct scoutfs_log_trees lt;
|
||||
|
||||
lt = tri->lt;
|
||||
lt.alloc_root = tri->alloc.alloc_root;
|
||||
lt.free_root = tri->alloc.free_root;
|
||||
scoutfs_forest_get_btrees(sb, <);
|
||||
scoutfs_data_get_btrees(sb, <);
|
||||
|
||||
return scoutfs_client_commit_log_trees(sb, <);
|
||||
}
|
||||
|
||||
/*
|
||||
* This gets all the resources from the server that the client will
|
||||
* need during the transaction.
|
||||
*/
|
||||
int scoutfs_trans_get_log_trees(struct super_block *sb)
|
||||
{
|
||||
DECLARE_TRANS_INFO(sb, tri);
|
||||
struct scoutfs_log_trees lt;
|
||||
int ret = 0;
|
||||
|
||||
ret = scoutfs_client_get_log_trees(sb, <);
|
||||
if (ret == 0) {
|
||||
tri->lt = lt;
|
||||
scoutfs_balloc_init(&tri->alloc, <.alloc_root, <.free_root);
|
||||
scoutfs_block_writer_init(sb, &tri->wri);
|
||||
|
||||
scoutfs_forest_init_btrees(sb, &tri->alloc, &tri->wri, <);
|
||||
scoutfs_data_init_btrees(sb, &tri->alloc, &tri->wri, <);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
bool scoutfs_trans_has_dirty(struct super_block *sb)
|
||||
{
|
||||
DECLARE_TRANS_INFO(sb, tri);
|
||||
|
||||
return scoutfs_block_writer_has_dirty(sb, &tri->wri);
|
||||
}
|
||||
/*
|
||||
* This work func is responsible for writing out all the dirty blocks
|
||||
* that make up the current dirty transaction. It prevents writers from
|
||||
@@ -113,18 +161,19 @@ void scoutfs_trans_write_func(struct work_struct *work)
|
||||
|
||||
wait_event(sbi->trans_hold_wq, drained_holders(tri));
|
||||
|
||||
trace_scoutfs_trans_write_func(sb, scoutfs_forest_dirty_bytes(sb));
|
||||
trace_scoutfs_trans_write_func(sb,
|
||||
scoutfs_block_writer_dirty_bytes(sb, &tri->wri));
|
||||
|
||||
if (scoutfs_forest_has_dirty(sb)) {
|
||||
if (scoutfs_block_writer_has_dirty(sb, &tri->wri)) {
|
||||
if (sbi->trans_deadline_expired)
|
||||
scoutfs_inc_counter(sb, trans_commit_timer);
|
||||
|
||||
ret = scoutfs_inode_walk_writeback(sb, true) ?:
|
||||
scoutfs_forest_write(sb) ?:
|
||||
scoutfs_block_writer_write(sb, &tri->wri) ?:
|
||||
scoutfs_inode_walk_writeback(sb, false) ?:
|
||||
scoutfs_forest_commit(sb) ?:
|
||||
commit_btrees(sb) ?:
|
||||
scoutfs_client_advance_seq(sb, &sbi->trans_seq) ?:
|
||||
scoutfs_forest_get_log_trees(sb);
|
||||
scoutfs_trans_get_log_trees(sb);
|
||||
if (ret)
|
||||
goto out;
|
||||
|
||||
@@ -297,7 +346,8 @@ static bool acquired_hold(struct super_block *sb,
|
||||
vals = tri->reserved_vals + cnt->vals;
|
||||
|
||||
/* XXX arbitrarily limit to 8 meg transactions */
|
||||
if (scoutfs_forest_dirty_bytes(sb) >= (8 * 1024 * 1024)) {
|
||||
if (scoutfs_block_writer_dirty_bytes(sb, &tri->wri) >=
|
||||
(8 * 1024 * 1024)) {
|
||||
scoutfs_inc_counter(sb, trans_commit_full);
|
||||
queue_trans_work(sbi);
|
||||
goto out;
|
||||
@@ -481,6 +531,7 @@ void scoutfs_shutdown_trans(struct super_block *sb)
|
||||
DECLARE_TRANS_INFO(sb, tri);
|
||||
|
||||
if (tri) {
|
||||
scoutfs_block_writer_forget_all(sb, &tri->wri);
|
||||
if (sbi->trans_write_workq) {
|
||||
cancel_delayed_work_sync(&sbi->trans_write_work);
|
||||
destroy_workqueue(sbi->trans_write_workq);
|
||||
|
||||
Reference in New Issue
Block a user