scoutfs: add data waiting

One of the core features of scoutfs is the ability to transparently
migrate file contents to and from an archive tier.  For this to be
transparent we need file system operations to trigger staging the file
contents back into the file system as needed.

This adds the infrastructure which operations use to wait for offline
extents to come online and which provides userspace with a list of
blocks that the operations are waiting for.

We add some waiting infrastructure that callers use to lock, check for
offline extents, and unlock and wait before checking again to see if
they're still offline.  We add these checks and waiting to data io
operations that could encounter offline extents.

This has to be done carefully so that we don't wait while holding locks
that would prevent staging.  We use per-task structures to discover when
we are the first user of a cluster lock on an inode, indicating that
it's safe for us to wait because we don't hold any locks.

And while we're waiting our operation is tracked and reported to
userspace through an ioctl.  This is a non-blocking ioctl, it's up to
userspace to decide how often to check and how large a region to stage.

Waiters are woken up when the file contents could have changed, not
specifically when we know that the extent has come online.  This lets us
wake waiters when their lock is revoked so that they can block waiting
to reacquire the lock and test the extents again.  It lets us provide
coherent demand staging across the cluster without fine grained waiting
protocols sent betwen the nodes.  It may result in some spurious wakeups
and work but hopefully it won't, and it's a very simple and functional
first pass.

Signed-off-by: Zach Brown <zab@versity.com>
This commit is contained in:
Zach Brown
2019-05-21 11:33:26 -07:00
committed by Zach Brown
parent cfa563a4a4
commit a6782fc03f
12 changed files with 576 additions and 16 deletions
+304 -7
View File
@@ -732,9 +732,9 @@ static int scoutfs_get_block(struct inode *inode, sector_t iblock,
if (ext.len)
trace_scoutfs_data_get_block_intersection(sb, &ext);
/* fail read and write if it's offline and we're not staging */
if ((ext.flags & SEF_OFFLINE) && !si->staging) {
ret = -EINVAL;
/* non-staging callers should have waited on offline blocks */
if (WARN_ON_ONCE((ext.flags & SEF_OFFLINE) && !si->staging)) {
ret = -EIO;
goto out;
}
@@ -780,14 +780,28 @@ out:
/*
* This is almost never used. We can't block on a cluster lock while
* holding the page lock because lock invalidation gets the page lock
* while blocking locks. If we can't use an existing lock then we drop
* the page lock and try again.
* while blocking locks. If a non blocking lock attempt fails we unlock
* the page and block acquiring the lock. We unlocked the page so it
* could have been truncated away, or whatever, so we return
* AOP_TRUNCATED_PAGE to have the caller try again.
*
* A similar process happens if we try to read from an offline extent
* that a caller hasn't already waited for. Instead of blocking
* acquiring the lock we block waiting for the offline extent. The page
* lock protects the page from release while we're checking and
* reading the extent.
*
* We can return errors from locking and checking offline extents. The
* page is unlocked if we return an error.
*/
static int scoutfs_readpage(struct file *file, struct page *page)
{
struct inode *inode = file->f_inode;
struct scoutfs_inode_info *si = SCOUTFS_I(inode);
struct super_block *sb = inode->i_sb;
struct scoutfs_lock *inode_lock = NULL;
SCOUTFS_DECLARE_PER_TASK_ENTRY(pt_ent);
DECLARE_DATA_WAIT(dw);
int flags;
int ret;
@@ -809,27 +823,77 @@ static int scoutfs_readpage(struct file *file, struct page *page)
return ret;
}
if (scoutfs_per_task_add_excl(&si->pt_data_lock, &pt_ent, inode_lock)) {
ret = scoutfs_data_wait_check(inode, page_offset(page),
PAGE_CACHE_SIZE, SEF_OFFLINE,
SCOUTFS_IOC_DWO_READ, &dw,
inode_lock);
if (ret != 0) {
unlock_page(page);
scoutfs_per_task_del(&si->pt_data_lock, &pt_ent);
scoutfs_unlock(sb, inode_lock, SCOUTFS_LOCK_READ);
}
if (ret > 0) {
ret = scoutfs_data_wait(inode, &dw);
if (ret == 0)
ret = AOP_TRUNCATED_PAGE;
}
if (ret != 0)
return ret;
}
ret = mpage_readpage(page, scoutfs_get_block);
scoutfs_unlock(sb, inode_lock, SCOUTFS_LOCK_READ);
scoutfs_per_task_del(&si->pt_data_lock, &pt_ent);
return ret;
}
/*
* This is used for opportunistic read-ahead which can throw the pages
* away if it needs to. If the caller didn't deal with offline extents
* then we drop those pages rather than trying to wait. Whoever is
* staging offline extents should be doing it in enormous chunks so that
* read-ahead can ramp up within each staged region. The check for
* offline extents is cheap when the inode has no offline extents.
*/
static int scoutfs_readpages(struct file *file, struct address_space *mapping,
struct list_head *pages, unsigned nr_pages)
{
struct inode *inode = file->f_inode;
struct super_block *sb = inode->i_sb;
struct scoutfs_lock *inode_lock = NULL;
struct page *page;
struct page *tmp;
int ret;
ret = scoutfs_lock_inode(sb, SCOUTFS_LOCK_READ,
SCOUTFS_LKF_REFRESH_INODE, inode, &inode_lock);
if (ret)
return ret;
goto out;
list_for_each_entry_safe(page, tmp, pages, lru) {
ret = scoutfs_data_wait_check(inode, page_offset(page),
PAGE_CACHE_SIZE, SEF_OFFLINE,
SCOUTFS_IOC_DWO_READ, NULL,
inode_lock);
if (ret < 0)
goto out;
if (ret > 0) {
list_del(&page->lru);
page_cache_release(page);
if (--nr_pages == 0) {
ret = 0;
goto out;
}
}
}
ret = mpage_readpages(mapping, pages, nr_pages, scoutfs_get_block);
out:
scoutfs_unlock(sb, inode_lock, SCOUTFS_LOCK_READ);
BUG_ON(!list_empty(pages));
return ret;
}
@@ -1249,6 +1313,239 @@ out:
return ret;
}
/*
* Insert a new waiter. This supports multiple tasks waiting for the
* same ino and iblock by also comparing waiters by their addresses.
*/
static void insert_offline_waiting(struct rb_root *root,
struct scoutfs_data_wait *ins)
{
struct rb_node **node = &root->rb_node;
struct rb_node *parent = NULL;
struct scoutfs_data_wait *dw;
int cmp;
while (*node) {
parent = *node;
dw = rb_entry(*node, struct scoutfs_data_wait, node);
cmp = scoutfs_cmp_u64s(ins->ino, dw->ino) ?:
scoutfs_cmp_u64s(ins->iblock, dw->iblock) ?:
scoutfs_cmp(ins, dw);
if (cmp < 0)
node = &(*node)->rb_left;
else
node = &(*node)->rb_right;
}
rb_link_node(&ins->node, parent, node);
rb_insert_color(&ins->node, root);
}
static struct scoutfs_data_wait *next_data_wait(struct rb_root *root, u64 ino,
u64 iblock)
{
struct rb_node **node = &root->rb_node;
struct rb_node *parent = NULL;
struct scoutfs_data_wait *next = NULL;
struct scoutfs_data_wait *dw;
int cmp;
while (*node) {
parent = *node;
dw = rb_entry(*node, struct scoutfs_data_wait, node);
/* go left when ino/iblock are equal to get first task */
cmp = scoutfs_cmp_u64s(ino, dw->ino) ?:
scoutfs_cmp_u64s(iblock, dw->iblock);
if (cmp <= 0) {
node = &(*node)->rb_left;
next = dw;
} else if (cmp > 0) {
node = &(*node)->rb_right;
}
}
return next;
}
static struct scoutfs_data_wait *dw_next(struct scoutfs_data_wait *dw)
{
struct rb_node *node = rb_next(&dw->node);
if (node)
return container_of(node, struct scoutfs_data_wait, node);
return NULL;
}
/*
* Check if we should wait by looking for extents whose flags match.
* Returns 0 if no extents were found or any error encountered.
*
* The caller must have locked the extents before calling, both across
* mounts and within this mount.
*
* Returns 1 if any file extents in the caller's region matched. If the
* wait struct is provided then it is initialized to be woken when the
* extents change after the caller unlocks after the check. The caller
* must come through _data_wait() to clean up the wait struct if we set
* it up.
*/
int scoutfs_data_wait_check(struct inode *inode, loff_t pos, loff_t len,
u8 sef, u8 op, struct scoutfs_data_wait *dw,
struct scoutfs_lock *lock)
{
struct super_block *sb = inode->i_sb;
DECLARE_DATA_WAIT_ROOT(sb, rt);
DECLARE_DATA_WAITQ(inode, wq);
struct scoutfs_extent ext = {0,};
u64 iblock;
u64 last_block;
u64 on;
u64 off;
int ret = 0;
if (WARN_ON_ONCE(sef & SEF_UNKNOWN) ||
WARN_ON_ONCE(op & SCOUTFS_IOC_DWO_UNKNOWN) ||
WARN_ON_ONCE(dw && !RB_EMPTY_NODE(&dw->node)) ||
WARN_ON_ONCE(pos + len < pos)) {
ret = -EINVAL;
goto out;
}
if ((sef & SEF_OFFLINE)) {
scoutfs_inode_get_onoff(inode, &on, &off);
if (off == 0) {
ret = 0;
goto out;
}
}
iblock = pos >> SCOUTFS_BLOCK_SHIFT;
last_block = (pos + len - 1) >> SCOUTFS_BLOCK_SHIFT;
while(iblock <= last_block) {
scoutfs_extent_init(&ext, SCOUTFS_FILE_EXTENT_TYPE,
scoutfs_ino(inode), iblock, 1, 0, 0);
ret = scoutfs_extent_next(sb, data_extent_io, &ext, lock);
if (ret < 0) {
if (ret == -ENOENT)
ret = 0;
break;
}
if (ext.start > last_block)
break;
if (sef & ext.flags) {
if (dw) {
dw->chg = atomic64_read(&wq->changed);
dw->ino = scoutfs_ino(inode);
dw->iblock = max(iblock, ext.start);
dw->op = op;
spin_lock(&rt->lock);
insert_offline_waiting(&rt->root, dw);
spin_unlock(&rt->lock);
}
ret = 1;
break;
}
iblock = ext.start + ext.len;
}
out:
trace_scoutfs_data_wait_check(sb, scoutfs_ino(inode), pos, len,
sef, op, ext.start, ext.len, ext.flags,
ret);
return ret;
}
bool scoutfs_data_wait_found(struct scoutfs_data_wait *dw)
{
return !RB_EMPTY_NODE(&dw->node);
}
int scoutfs_data_wait_check_iov(struct inode *inode, const struct iovec *iov,
unsigned long nr_segs, loff_t pos, u8 sef,
u8 op, struct scoutfs_data_wait *dw,
struct scoutfs_lock *lock)
{
unsigned long i;
int ret = 0;
for (i = 0; i < nr_segs; i++) {
if (iov[i].iov_len == 0)
continue;
ret = scoutfs_data_wait_check(inode, pos, iov[i].iov_len, sef,
op, dw, lock);
if (ret != 0)
break;
pos += iov[i].iov_len;
}
return ret;
}
int scoutfs_data_wait(struct inode *inode, struct scoutfs_data_wait *dw)
{
DECLARE_DATA_WAIT_ROOT(inode->i_sb, rt);
DECLARE_DATA_WAITQ(inode, wq);
int ret;
ret = wait_event_interruptible(wq->waitq,
atomic64_read(&wq->changed) != dw->chg);
spin_lock(&rt->lock);
rb_erase(&dw->node, &rt->root);
RB_CLEAR_NODE(&dw->node);
spin_unlock(&rt->lock);
return ret;
}
void scoutfs_data_wait_changed(struct inode *inode)
{
DECLARE_DATA_WAITQ(inode, wq);
atomic64_inc(&wq->changed);
wake_up(&wq->waitq);
}
int scoutfs_data_waiting(struct super_block *sb, u64 ino, u64 iblock,
struct scoutfs_ioctl_data_waiting_entry *dwe,
unsigned int nr)
{
DECLARE_DATA_WAIT_ROOT(sb, rt);
struct scoutfs_data_wait *dw;
int ret = 0;
spin_lock(&rt->lock);
dw = next_data_wait(&rt->root, ino, iblock);
while (dw && ret < nr) {
dwe->ino = dw->ino;
dwe->iblock = dw->iblock;
dwe->op = dw->op;
while ((dw = dw_next(dw)) &&
(dw->ino == dwe->ino && dw->iblock == dwe->iblock)) {
dwe->op |= dw->op;
}
dwe++;
ret++;
}
spin_unlock(&rt->lock);
return ret;
}
const struct address_space_operations scoutfs_file_aops = {
.readpage = scoutfs_readpage,
.readpages = scoutfs_readpages,