From 750e998e40f5019a836c9b0bbfab10e2511b6e19 Mon Sep 17 00:00:00 2001 From: Zach Brown Date: Wed, 11 Feb 2026 15:13:02 -0800 Subject: [PATCH] Add raw_read_meta_seq ioctl Add an ioctl for reading the meta_seq index without cluster locking. Signed-off-by: Zach Brown --- kmod/src/Makefile | 1 + kmod/src/ioctl.c | 38 ++++++++ kmod/src/ioctl.h | 61 ++++++++++++ kmod/src/raw.c | 236 ++++++++++++++++++++++++++++++++++++++++++++++ kmod/src/raw.h | 8 ++ 5 files changed, 344 insertions(+) create mode 100644 kmod/src/raw.c create mode 100644 kmod/src/raw.h diff --git a/kmod/src/Makefile b/kmod/src/Makefile index fa632aa1..c9f6a961 100644 --- a/kmod/src/Makefile +++ b/kmod/src/Makefile @@ -36,6 +36,7 @@ scoutfs-y += \ per_task.o \ quorum.o \ quota.o \ + raw.o \ recov.o \ scoutfs_trace.o \ server.o \ diff --git a/kmod/src/ioctl.c b/kmod/src/ioctl.c index 0a5fc4c7..5dda2d94 100644 --- a/kmod/src/ioctl.c +++ b/kmod/src/ioctl.c @@ -49,6 +49,7 @@ #include "quota.h" #include "scoutfs_trace.h" #include "util.h" +#include "raw.h" /* * We make inode index items coherent by locking fixed size regions of @@ -1739,6 +1740,41 @@ out: return ret; } +static long scoutfs_ioc_raw_read_meta_seq(struct file *file, unsigned long arg) +{ + struct super_block *sb = file_inode(file)->i_sb; + struct scoutfs_ioctl_raw_read_meta_seq __user *urms = (void __user *)arg; + struct scoutfs_ioctl_raw_read_meta_seq rms; + int ret; + + if (!capable(CAP_SYS_ADMIN)) { + ret = -EPERM; + goto out; + } + + if (copy_from_user(&rms, urms, sizeof(rms))) { + ret = -EFAULT; + goto out; + } + + if (rms.results_size == 0) { + ret = 0; + goto out; + } + + if (rms.results_size < sizeof(struct scoutfs_ioctl_meta_seq) || + rms.results_size > INT_MAX) { + ret = -EINVAL; + goto out; + } + + ret = scoutfs_raw_read_meta_seq(sb, &rms, &rms.last); + if (ret >= 0 && copy_to_user(&urms->last, &rms.last, sizeof(rms.last))) + ret = -EFAULT; +out: + return ret; +} + long scoutfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg) { switch (cmd) { @@ -1790,6 +1826,8 @@ long scoutfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg) return scoutfs_ioc_read_xattr_index(file, arg); case SCOUTFS_IOC_PUNCH_OFFLINE: return scoutfs_ioc_punch_offline(file, arg); + case SCOUTFS_IOC_RAW_READ_META_SEQ: + return scoutfs_ioc_raw_read_meta_seq(file, arg); } return -ENOTTY; diff --git a/kmod/src/ioctl.h b/kmod/src/ioctl.h index 93533931..41b79fb6 100644 --- a/kmod/src/ioctl.h +++ b/kmod/src/ioctl.h @@ -862,4 +862,65 @@ struct scoutfs_ioctl_punch_offline { #define SCOUTFS_IOC_PUNCH_OFFLINE \ _IOW(SCOUTFS_IOCTL_MAGIC, 24, struct scoutfs_ioctl_punch_offline) +/* + * Read meta_seq items without cluster locking. + * + * @start is the first meta_seq item value that could be returned. + * {0,0} is the minimum. + * + * @end is the last meta_seq item value that could be returned. + * {U64_MAX, U64_MAX} is the maximum. + * + * @last is only set on success from the call. It's the last meta_seq + * item that could have been returned. This lets the caller detect that + * the full input range wasn't explored. Another call can be made with + * start set to just after this. + * + * @results_ptr is a pointer to an array of (struct + * scoutfs_ioctl_meta_seq) elements that were found in the input range. + * + * @results_size is the count of elements in the results_ptr array and + * the maximum number of results that can be returned. There must be + * room for at least one result. + * + * Return existing meta_seq items starting from @start until @last. + * Partial results can be returned and is indicated by @last being set + * to an item before @last. + * + * The results are sorted first by increasing meta_seq and then by + * increasing ino. All of the results are from one version of file + * system metadata. This means that an inode can not be found multiple + * times within the results of one call. + * + * This call ignores currently dirty transactions and reads persistent + * items directly. A transaction can be written after this call and + * cause meta_seq items to appear before or within the results from this + * call. + * + * The number of meta_seq items stored in the results buffer is returned + * and @last is updated. 0 items can be returned if none are found + * within the input range. + * + * Unique errors: + * + * -EINVAL: The result count was 0 or greater than INT_MAX. + * + * -ESTALE: The results could not be read from one stable version of + * file system metadata. Decrease the number of inodes requested. + */ +struct scoutfs_ioctl_meta_seq { + __u64 meta_seq; + __u64 ino; +}; +struct scoutfs_ioctl_raw_read_meta_seq { + struct scoutfs_ioctl_meta_seq start; + struct scoutfs_ioctl_meta_seq end; + struct scoutfs_ioctl_meta_seq last; + __u64 results_ptr; + __u32 results_size; + __u32 _pad; +}; +#define SCOUTFS_IOC_RAW_READ_META_SEQ \ + _IOR(SCOUTFS_IOCTL_MAGIC, 25, struct scoutfs_ioctl_raw_read_meta_seq) + #endif diff --git a/kmod/src/raw.c b/kmod/src/raw.c new file mode 100644 index 00000000..9515608e --- /dev/null +++ b/kmod/src/raw.c @@ -0,0 +1,236 @@ +/* + * Copyright (C) 2026 Versity Software, Inc. All rights reserved. + * + * This program is free software; you can redistribute it and/or + * modify it under the terms of the GNU General Public + * License v2 as published by the Free Software Foundation. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + * General Public License for more details. + */ +#include +#include +#include + +#include "format.h" +#include "key.h" +#include "block.h" +#include "inode.h" +#include "forest.h" +#include "client.h" +#include "ioctl.h" +#include "raw.h" + +struct fs_item { + struct list_head head; + struct scoutfs_key key; + u64 seq; + int val_len; + bool deletion; + /* val is aligned so we can deref structs in vals */ + u8 val[0] __aligned(ARCH_KMALLOC_MINALIGN); +}; + +static int save_fs_item(struct list_head *list, struct scoutfs_key *key, u64 seq, u8 flags, + void *val, int val_len) +{ + struct fs_item *fsi; + + /* max btree val len is hundreds of bytes */ + fsi = kmalloc(offsetof(struct fs_item, val[val_len]), GFP_NOFS); + if (!fsi) + return -ENOMEM; + + fsi->key = *key; + fsi->seq = seq; + fsi->val_len = val_len; + fsi->deletion = !!(flags & SCOUTFS_ITEM_FLAG_DELETION); + if (val_len > 0) + memcpy(fsi->val, val, val_len); + list_add_tail(&fsi->head, list); + + return 0; +} + +static void free_fs_item(struct fs_item *fsi) +{ + if (!list_empty(&fsi->head)) + list_del_init(&fsi->head); + kfree(fsi); +} + +static void free_fs_items(struct list_head *list) +{ + struct fs_item *fsi; + struct fs_item *tmp; + + list_for_each_entry_safe(fsi, tmp, list, head) + free_fs_item(fsi); +} + +static int cmp_fs_items(void *priv, KC_LIST_CMP_CONST struct list_head *A, + KC_LIST_CMP_CONST struct list_head *B) +{ + KC_LIST_CMP_CONST struct fs_item *a = + container_of(A, KC_LIST_CMP_CONST struct fs_item, head); + KC_LIST_CMP_CONST struct fs_item *b = + container_of(B, KC_LIST_CMP_CONST struct fs_item, head); + + return scoutfs_key_compare(&a->key, &b->key) ?: -scoutfs_cmp(a->seq, b->seq); +} + +static void sort_and_remove(struct list_head *list, struct scoutfs_key *end) +{ + struct fs_item *prev; + struct fs_item *fsi; + struct fs_item *tmp; + + list_sort(NULL, list, cmp_fs_items); + + /* start by removing any items read before end was decreased by later blocks */ + list_for_each_entry_safe_reverse(fsi, tmp, list, head) { + if (scoutfs_key_compare(&fsi->key, end) > 0) + free_fs_item(fsi); + else + break; + } + + prev = NULL; + list_for_each_entry_safe(fsi, tmp, list, head) { + /* remove this item if it's an older version of previous item */ + if (prev && scoutfs_key_compare(&prev->key, &fsi->key) == 0) { + free_fs_item(fsi); + continue; + } + + /* remove previous deletion item once it has removed all older versions */ + if (prev && prev->deletion) + free_fs_item(prev); + + /* next item might match this, record to compare */ + prev = fsi; + } + + /* remove the last item if it's a deletion */ + list_for_each_entry_reverse(fsi, list, head) { + if (fsi->deletion) + free_fs_item(fsi); + break; + } +} + +static int save_all_items(struct super_block *sb, struct scoutfs_key *key, u64 seq, u8 flags, + void *val, int val_len, int fic, void *arg) +{ + struct list_head *list = arg; + + return save_fs_item(list, key, seq, flags, val, val_len); +} + +/* -------------- */ + +static void ms_from_key(struct scoutfs_ioctl_meta_seq *ms, struct scoutfs_key *key) +{ + ms->meta_seq = le64_to_cpu(key->skii_major); + ms->ino = le64_to_cpu(key->skii_ino); +} + +/* + * Increment the key's ino->meta_seq so that we don't land between items. + */ +static void inc_meta_seq(struct scoutfs_key *key) +{ + le64_add_cpu(&key->skii_ino, 1); + if (key->skii_ino == 0) + le64_add_cpu(&key->skii_major, 1); +} + +int scoutfs_raw_read_meta_seq(struct super_block *sb, + struct scoutfs_ioctl_raw_read_meta_seq *rms, + struct scoutfs_ioctl_meta_seq *last_ret) +{ + struct scoutfs_ioctl_meta_seq __user *ums; + struct scoutfs_ioctl_meta_seq ms; + struct scoutfs_net_roots roots; + DECLARE_SAVED_REFS(saved); + struct scoutfs_key start; + struct scoutfs_key last; + struct scoutfs_key key; + struct scoutfs_key end; + struct fs_item *fsi; + struct fs_item *tmp; + LIST_HEAD(list); + int retries; + int copied; + int count; + int ret; + + ums = (void __user *)rms->results_ptr; + count = rms->results_size / sizeof(struct scoutfs_ioctl_meta_seq); + retries = 10; + copied = 0; + + scoutfs_inode_init_index_key(&last, SCOUTFS_INODE_INDEX_META_SEQ_TYPE, + rms->end.meta_seq, 0, rms->end.ino); + +retry: + ret = scoutfs_client_get_roots(sb, &roots); + if (ret) + goto out; + + scoutfs_inode_init_index_key(&key, SCOUTFS_INODE_INDEX_META_SEQ_TYPE, + rms->start.meta_seq, 0, rms->start.ino); + + for (;;) { + start = key; + end = last; + ret = scoutfs_forest_read_items_roots(sb, &roots, 0, &key, NULL, &start, &end, + save_all_items, &list); + if (ret < 0) + goto out; + + sort_and_remove(&list, &end); + + list_for_each_entry_safe(fsi, tmp, &list, head) { + + if (copied == count) { + /* results are full, set end to before item can't return */ + end = fsi->key; + le64_add_cpu(&end.skii_ino, -1ULL); + ret = 0; + goto out; + } + + ms_from_key(&ms, &fsi->key); + if (copy_to_user(&ums[copied], &ms, sizeof(ms))) { + ret = -EFAULT; + goto out; + } + + free_fs_item(fsi); + copied++; + } + + if (scoutfs_key_compare(&end, &last) >= 0) { + end = last; + break; + } + + key = end; + inc_meta_seq(&key); + } + + ret = 0; +out: + free_fs_items(&list); + + ret = scoutfs_block_check_stale(sb, ret, &saved, &roots.fs_root.ref, &roots.logs_root.ref); + if (ret == -ESTALE && copied == 0 && retries-- > 0) + goto retry; + + ms_from_key(last_ret, &end); + + return ret ?: copied; +} diff --git a/kmod/src/raw.h b/kmod/src/raw.h new file mode 100644 index 00000000..7fb8fc49 --- /dev/null +++ b/kmod/src/raw.h @@ -0,0 +1,8 @@ +#ifndef _SCOUTFS_RAW_H_ +#define _SCOUTFS_RAW_H_ + +int scoutfs_raw_read_meta_seq(struct super_block *sb, + struct scoutfs_ioctl_raw_read_meta_seq *rms, + struct scoutfs_ioctl_meta_seq *last_ret); + +#endif