diff --git a/utils/Makefile b/utils/Makefile index e0f76142..7f819d40 100644 --- a/utils/Makefile +++ b/utils/Makefile @@ -7,7 +7,7 @@ FMTIOC_H := format.h ioctl.h FMTIOC_KMOD := $(addprefix ../kmod/src/,$(FMTIOC_H)) CFLAGS := -Wall -O2 -Werror -D_FILE_OFFSET_BITS=64 -g -msse4.2 \ - -fno-strict-aliasing \ + -I src/ -fno-strict-aliasing \ -DSCOUTFS_FORMAT_HASH=0x$(SCOUTFS_FORMAT_HASH)LLU ifneq ($(wildcard $(firstword $(FMTIOC_KMOD))),) @@ -15,8 +15,9 @@ CFLAGS += -I../kmod/src endif BIN := src/scoutfs -OBJ := $(patsubst %.c,%.o,$(wildcard src/*.c)) -DEPS := $(wildcard */*.d) +OBJ_DIRS := src src/check +OBJ := $(foreach dir,$(OBJ_DIRS),$(patsubst %.c,%.o,$(wildcard $(dir)/*.c))) +DEPS := $(foreach dir,$(OBJ_DIRS),$(wildcard $(dir)/*.d)) all: $(BIN) diff --git a/utils/src/check/alloc.c b/utils/src/check/alloc.c new file mode 100644 index 00000000..f67b6660 --- /dev/null +++ b/utils/src/check/alloc.c @@ -0,0 +1,159 @@ +#include +#include +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" +#include "bitmap.h" +#include "key.h" + +#include "alloc.h" +#include "block.h" +#include "btree.h" +#include "extent.h" +#include "iter.h" +#include "sns.h" + +/* + * We check the list blocks serially. + * + * XXX: + * - compare ref seqs + * - detect cycles? + */ +int alloc_list_meta_iter(struct scoutfs_alloc_list_head *lhead, extent_cb_t cb, void *cb_arg) +{ + struct scoutfs_alloc_list_block *lblk; + struct scoutfs_block_ref ref; + struct block *blk = NULL; + u64 blkno; + int ret; + + ref = lhead->ref; + + while (ref.blkno) { + blkno = le64_to_cpu(ref.blkno); + + ret = cb(blkno, 1, cb_arg); + if (ret < 0) { + ret = xlate_iter_errno(ret); + goto out; + } + + ret = block_get(&blk, blkno, 0); + if (ret < 0) + goto out; + + lblk = block_buf(blk); + /* XXX verify block */ + /* XXX sort? maybe */ + + ref = lblk->next; + + block_put(&blk); + } + + ret = 0; +out: + return ret; +} + +int alloc_root_meta_iter(struct scoutfs_alloc_root *root, extent_cb_t cb, void *cb_arg) +{ + return btree_meta_iter(&root->root, cb, cb_arg); +} + +int alloc_list_extent_iter(struct scoutfs_alloc_list_head *lhead, extent_cb_t cb, void *cb_arg) +{ + struct scoutfs_alloc_list_block *lblk; + struct scoutfs_block_ref ref; + struct block *blk = NULL; + u64 blkno; + int ret; + int i; + + ref = lhead->ref; + + while (ref.blkno) { + blkno = le64_to_cpu(ref.blkno); + + ret = block_get(&blk, blkno, 0); + if (ret < 0) + goto out; + + sns_push("alloc_list_block", blkno, 0); + + lblk = block_buf(blk); + /* XXX verify block */ + /* XXX sort? maybe */ + + ret = 0; + for (i = 0; i < le32_to_cpu(lblk->nr); i++) { + blkno = le64_to_cpu(lblk->blknos[le32_to_cpu(lblk->start) + i]); + + ret = cb(blkno, 1, cb_arg); + if (ret < 0) + break; + } + + ref = lblk->next; + + block_put(&blk); + sns_pop(); + if (ret < 0) { + ret = xlate_iter_errno(ret); + goto out; + } + } + + ret = 0; +out: + return ret; +} + +static bool valid_free_extent_key(struct scoutfs_key *key) +{ + return (key->sk_zone == SCOUTFS_FREE_EXTENT_BLKNO_ZONE || + key->sk_zone == SCOUTFS_FREE_EXTENT_ORDER_ZONE) && + (!key->_sk_fourth && !key->sk_type && + (key->sk_zone == SCOUTFS_FREE_EXTENT_ORDER_ZONE || !key->_sk_third)); +} + +static int free_item_cb(struct scoutfs_key *key, void *val, u16 val_len, void *cb_arg) +{ + struct extent_cb_arg_t *ecba = cb_arg; + u64 start; + u64 len; + + /* XXX not sure these eios are what we want */ + + if (val_len != 0) + return -EIO; + + if (!valid_free_extent_key(key)) + return -EIO; + + if (key->sk_zone == SCOUTFS_FREE_EXTENT_ORDER_ZONE) + return -ECHECK_ITER_DONE; + + start = le64_to_cpu(key->skfb_end) - le64_to_cpu(key->skfb_len) + 1; + len = le64_to_cpu(key->skfb_len); + + return ecba->cb(start, len, ecba->cb_arg); +} + +/* + * Call the callback with each of the primary BLKNO free extents stored + * in item in the given alloc root. It doesn't visit the secondary + * ORDER extents. + */ +int alloc_root_extent_iter(struct scoutfs_alloc_root *root, extent_cb_t cb, void *cb_arg) +{ + struct extent_cb_arg_t ecba = { .cb = cb, .cb_arg = cb_arg }; + + return btree_item_iter(&root->root, free_item_cb, &ecba); +} diff --git a/utils/src/check/alloc.h b/utils/src/check/alloc.h new file mode 100644 index 00000000..f0273e4a --- /dev/null +++ b/utils/src/check/alloc.h @@ -0,0 +1,12 @@ +#ifndef _SCOUTFS_UTILS_CHECK_ALLOC_H +#define _SCOUTFS_UTILS_CHECK_ALLOC_H + +#include "extent.h" + +int alloc_list_meta_iter(struct scoutfs_alloc_list_head *lhead, extent_cb_t cb, void *cb_arg); +int alloc_root_meta_iter(struct scoutfs_alloc_root *root, extent_cb_t cb, void *cb_arg); + +int alloc_list_extent_iter(struct scoutfs_alloc_list_head *lhead, extent_cb_t cb, void *cb_arg); +int alloc_root_extent_iter(struct scoutfs_alloc_root *root, extent_cb_t cb, void *cb_arg); + +#endif diff --git a/utils/src/check/block.c b/utils/src/check/block.c new file mode 100644 index 00000000..53b6eed0 --- /dev/null +++ b/utils/src/check/block.c @@ -0,0 +1,564 @@ +#define _ISOC11_SOURCE /* aligned_alloc */ +#define _DEFAULT_SOURCE /* syscall() */ +#include +#include +#include +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" +#include "list.h" +#include "cmp.h" +#include "hash.h" + +#include "block.h" +#include "debug.h" +#include "eno.h" + +static struct block_data { + struct list_head *hash_lists; + size_t hash_nr; + + struct list_head active_head; + struct list_head inactive_head; + struct list_head dirty_list; + size_t nr_active; + size_t nr_inactive; + size_t nr_dirty; + + int meta_fd; + size_t max_cached; + size_t nr_events; + + aio_context_t ctx; + struct iocb *iocbs; + struct iocb **iocbps; + struct io_event *events; +} global_bdat; + +struct block { + struct list_head hash_head; + struct list_head lru_head; + struct list_head dirty_head; + struct list_head submit_head; + unsigned long refcount; + unsigned long uptodate:1, + active:1; + u64 blkno; + void *buf; + size_t size; +}; + +#define BLK_FMT \ + "blkno %llu rc %ld d %u a %u" +#define BLK_ARG(blk) \ + (blk)->blkno, (blk)->refcount, !list_empty(&(blk)->dirty_head), blk->active +#define debug_blk(blk, fmt, args...) \ + debug(fmt " " BLK_FMT, ##args, BLK_ARG(blk)) + +/* + * This just allocates and initialzies the block. The caller is + * responsible for putting it on the appropriate initial lists and + * managing refcounts. + */ +static struct block *alloc_block(struct block_data *bdat, u64 blkno, size_t size) +{ + struct block *blk; + + blk = calloc(1, sizeof(struct block)); + if (blk) { + blk->buf = aligned_alloc(4096, size); /* XXX static alignment :/ */ + if (!blk->buf) { + free(blk); + blk = NULL; + } else { + INIT_LIST_HEAD(&blk->hash_head); + INIT_LIST_HEAD(&blk->lru_head); + INIT_LIST_HEAD(&blk->dirty_head); + INIT_LIST_HEAD(&blk->submit_head); + blk->blkno = blkno; + blk->size = size; + } + } + + return blk; +} + +static void free_block(struct block_data *bdat, struct block *blk) +{ + debug_blk(blk, "free"); + + if (!list_empty(&blk->lru_head)) { + if (blk->active) + bdat->nr_active--; + else + bdat->nr_inactive--; + list_del(&blk->lru_head); + } + + if (!list_empty(&blk->dirty_head)) { + bdat->nr_dirty--; + list_del(&blk->dirty_head); + } + + if (!list_empty(&blk->hash_head)) + list_del(&blk->hash_head); + + if (!list_empty(&blk->submit_head)) + list_del(&blk->submit_head); + + free(blk->buf); + free(blk); +} + +static bool blk_is_dirty(struct block *blk) +{ + return !list_empty(&blk->dirty_head); +} + +/* + * Rebalance the cache. + * + * First we shrink the cache to limit it to max_cached blocks. + * Logically, we walk from oldest to newest in the inactive list and + * then in the active list. Since these lists are physically one + * list_head list we achieve this with a reverse walk starting from the + * active head. + * + * Then we rebalnace the size of the two lists. The constraint is that + * we don't let the active list grow larger than the inactive list. We + * move blocks from the oldest tail of the active list to the newest + * head of the inactive list. + * + * <- [active head] <-> [ .. active list .. ] <-> [inactive head] <-> [ .. inactive list .. ] -> + */ +static void rebalance_cache(struct block_data *bdat) +{ + struct block *blk; + struct block *blk_; + + list_for_each_entry_safe_reverse(blk, blk_, &bdat->active_head, lru_head) { + if ((bdat->nr_active + bdat->nr_inactive) < bdat->max_cached) + break; + + if (&blk->lru_head == &bdat->inactive_head || blk->refcount > 0 || + blk_is_dirty(blk)) + continue; + + free_block(bdat, blk); + } + + list_for_each_entry_safe_reverse(blk, blk_, &bdat->inactive_head, lru_head) { + if (bdat->nr_active <= bdat->nr_inactive || &blk->lru_head == &bdat->active_head) + break; + + list_move(&blk->lru_head, &bdat->inactive_head); + blk->active = 0; + bdat->nr_active--; + bdat->nr_inactive++; + } +} + +static void make_active(struct block_data *bdat, struct block *blk) +{ + if (!blk->active) { + if (!list_empty(&blk->lru_head)) { + list_move(&blk->lru_head, &bdat->active_head); + bdat->nr_inactive--; + } else { + list_add(&blk->lru_head, &bdat->active_head); + } + + blk->active = 1; + bdat->nr_active++; + } +} + +static int compar_iocbp(const void *A, const void *B) +{ + struct iocb *a = *(struct iocb **)A; + struct iocb *b = *(struct iocb **)B; + + return scoutfs_cmp(a->aio_offset, b->aio_offset); +} + +static int submit_and_wait(struct block_data *bdat, struct list_head *list) +{ + struct io_event *event; + struct iocb *iocb; + struct block *blk; + int ret; + int err; + int nr; + int i; + + err = 0; + nr = 0; + list_for_each_entry(blk, list, submit_head) { + iocb = &bdat->iocbs[nr]; + bdat->iocbps[nr] = iocb; + + memset(iocb, 0, sizeof(struct iocb)); + + iocb->aio_data = (intptr_t)blk; + iocb->aio_lio_opcode = blk_is_dirty(blk) ? IOCB_CMD_PWRITE : IOCB_CMD_PREAD; + iocb->aio_fildes = bdat->meta_fd; + iocb->aio_buf = (intptr_t)blk->buf; + iocb->aio_nbytes = blk->size; + iocb->aio_offset = blk->blkno * blk->size; + + nr++; + + debug_blk(blk, "submit"); + + if ((nr < bdat->nr_events) && blk->submit_head.next != list) + continue; + + qsort(bdat->iocbps, nr, sizeof(bdat->iocbps[0]), compar_iocbp); + + ret = syscall(__NR_io_submit, bdat->ctx, nr, bdat->iocbps); + if (ret != nr) { + if (ret >= 0) + errno = EIO; + ret = -errno; + fprintf(stderr, "fatal system error submitting async IO: "ENO_FMT"\n", + ENO_ARG(-ret)); + goto out; + } + + ret = syscall(__NR_io_getevents, bdat->ctx, nr, nr, bdat->events, NULL); + if (ret != nr) { + if (ret >= 0) + errno = EIO; + ret = -errno; + fprintf(stderr, "fatal system error getting IO events: "ENO_FMT"\n", + ENO_ARG(-ret)); + goto out; + } + + ret = 0; + for (i = 0; i < nr; i++) { + event = &bdat->events[i]; + iocb = (struct iocb *)(intptr_t)event->obj; + blk = (struct block *)(intptr_t)event->data; + + debug_blk(blk, "complete res %lld", (long long)event->res); + + if (event->res >= 0 && event->res != blk->size) + event->res = -EIO; + + /* io errors are fatal */ + if (event->res < 0) { + ret = event->res; + goto out; + } + + if (iocb->aio_lio_opcode == IOCB_CMD_PREAD) { + blk->uptodate = 1; + } else { + list_del_init(&blk->dirty_head); + bdat->nr_dirty--; + } + } + nr = 0; + } + + ret = 0; +out: + return ret ?: err; +} + +static void inc_refcount(struct block *blk) +{ + blk->refcount++; +} + +void block_put(struct block **blkp) +{ + struct block_data *bdat = &global_bdat; + struct block *blk = *blkp; + + if (blk) { + blk->refcount--; + *blkp = NULL; + + rebalance_cache(bdat); + } +} + +static struct list_head *hash_bucket(struct block_data *bdat, u64 blkno) +{ + u32 hash = scoutfs_hash32(&blkno, sizeof(blkno)); + + return &bdat->hash_lists[hash % bdat->hash_nr]; +} + +static struct block *get_or_alloc(struct block_data *bdat, u64 blkno, int bf) +{ + struct list_head *bucket = hash_bucket(bdat, blkno); + struct block *search; + struct block *blk; + size_t size; + + size = (bf & BF_SM) ? SCOUTFS_BLOCK_SM_SIZE : SCOUTFS_BLOCK_LG_SIZE; + + blk = NULL; + list_for_each_entry(search, bucket, hash_head) { + if (search->blkno == blkno && search->size == size) { + blk = search; + break; + } + } + + if (!blk) { + blk = alloc_block(bdat, blkno, size); + if (blk) { + list_add(&blk->hash_head, bucket); + list_add(&blk->lru_head, &bdat->inactive_head); + bdat->nr_inactive++; + } + } + if (blk) + inc_refcount(blk); + + return blk; +} + +/* + * Get a block. + * + * The caller holds a refcount to the block while it's in use that + * prevents it from being removed from the cache. It must be dropped + * with block_put(); + */ +int block_get(struct block **blk_ret, u64 blkno, int bf) +{ + struct block_data *bdat = &global_bdat; + struct block *blk; + LIST_HEAD(list); + int ret; + + blk = get_or_alloc(bdat, blkno, bf); + if (!blk) { + ret = -ENOMEM; + goto out; + } + + if ((bf & BF_ZERO)) { + memset(blk->buf, 0, blk->size); + blk->uptodate = 1; + } + + if (bf & BF_OVERWRITE) + blk->uptodate = 1; + + if (!blk->uptodate) { + list_add(&blk->submit_head, &list); + ret = submit_and_wait(bdat, &list); + list_del_init(&blk->submit_head); + if (ret < 0) + goto out; + } + + if ((bf & BF_DIRTY) && !blk_is_dirty(blk)) { + list_add_tail(&bdat->dirty_list, &blk->dirty_head); + bdat->nr_dirty++; + } + + make_active(bdat, blk); + + rebalance_cache(bdat); + ret = 0; +out: + if (ret < 0) + block_put(&blk); + *blk_ret = blk; + return ret; +} + +void *block_buf(struct block *blk) +{ + return blk->buf; +} + +size_t block_size(struct block *blk) +{ + return blk->size; +} + +/* + * Drop the block from the cache, regardless of if it was free or not. + * This is used to avoid writing blocks which were dirtied but then + * later freed. + * + * The block is immediately freed and can't be referenced after this + * returns. + */ +void block_drop(struct block **blkp) +{ + struct block_data *bdat = &global_bdat; + + free_block(bdat, *blkp); + *blkp = NULL; + rebalance_cache(bdat); +} + +/* + * This doesn't quite work for mixing large and small blocks, but that's + * fine, we never do that. + */ +static int compar_u64(const void *A, const void *B) +{ + u64 a = *((u64 *)A); + u64 b = *((u64 *)B); + + return scoutfs_cmp(a, b); +} + +/* + * This read-ahead is synchronous and errors are ignored. If any of the + * blknos aren't present in the cache then we issue concurrent reads for + * them and wait. Any existing cached blocks will be left as is. + * + * We might be trying to read a lot more than the number of events so we + * sort the caller's blknos before iterating over them rather than + * relying on submission sorting the blocks in each submitted set. + */ +void block_readahead(u64 *blknos, size_t nr) +{ + struct block_data *bdat = &global_bdat; + struct block *blk; + struct block *blk_; + LIST_HEAD(list); + size_t i; + + if (nr == 0) + return; + + qsort(blknos, nr, sizeof(blknos[0]), compar_u64); + + for (i = 0; i < nr; i++) { + blk = get_or_alloc(bdat, blknos[i], 0); + if (blk) { + if (!blk->uptodate) + list_add_tail(&blk->submit_head, &list); + else + block_put(&blk); + } + } + + (void)submit_and_wait(bdat, &list); + + list_for_each_entry_safe(blk, blk_, &list, submit_head) { + list_del_init(&blk->submit_head); + block_put(&blk); + } + + rebalance_cache(bdat); +} + +/* + * The caller's block changes form a consistent transaction. If the amount of dirty + * blocks is large enough we issue a write. + */ +int block_try_commit(bool force) +{ + struct block_data *bdat = &global_bdat; + struct block *blk; + struct block *blk_; + LIST_HEAD(list); + int ret; + + if (!force && bdat->nr_dirty < bdat->nr_events) + return 0; + + list_for_each_entry(blk, &bdat->dirty_list, dirty_head) { + list_add_tail(&blk->submit_head, &list); + inc_refcount(blk); + } + + ret = submit_and_wait(bdat, &list); + + list_for_each_entry_safe(blk, blk_, &list, submit_head) { + list_del_init(&blk->submit_head); + block_put(&blk); + } + + if (ret < 0) { + fprintf(stderr, "error writing dirty transaction blocks\n"); + goto out; + } + + ret = block_get(&blk, SCOUTFS_SUPER_BLKNO, BF_SM | BF_OVERWRITE | BF_DIRTY); + if (ret == 0) { + list_add(&blk->submit_head, &list); + ret = submit_and_wait(bdat, &list); + list_del_init(&blk->submit_head); + block_put(&blk); + } else { + ret = -ENOMEM; + } + if (ret < 0) + fprintf(stderr, "error writing super block to commit transaction\n"); + +out: + rebalance_cache(bdat); + return ret; +} + +int block_setup(int meta_fd, size_t max_cached_bytes, size_t max_dirty_bytes) +{ + struct block_data *bdat = &global_bdat; + size_t i; + int ret; + + bdat->max_cached = DIV_ROUND_UP(max_cached_bytes, SCOUTFS_BLOCK_LG_SIZE); + bdat->hash_nr = bdat->max_cached / 4; + bdat->nr_events = DIV_ROUND_UP(max_dirty_bytes, SCOUTFS_BLOCK_LG_SIZE); + + bdat->iocbs = calloc(bdat->nr_events, sizeof(bdat->iocbs[0])); + bdat->iocbps = calloc(bdat->nr_events, sizeof(bdat->iocbps[0])); + bdat->events = calloc(bdat->nr_events, sizeof(bdat->events[0])); + bdat->hash_lists = calloc(bdat->hash_nr, sizeof(bdat->hash_lists[0])); + if (!bdat->iocbs || !bdat->iocbps || !bdat->events || !bdat->hash_lists) { + ret = -ENOMEM; + goto out; + } + + INIT_LIST_HEAD(&bdat->active_head); + INIT_LIST_HEAD(&bdat->inactive_head); + INIT_LIST_HEAD(&bdat->dirty_list); + bdat->meta_fd = meta_fd; + list_add(&bdat->inactive_head, &bdat->active_head); + + for (i = 0; i < bdat->hash_nr; i++) + INIT_LIST_HEAD(&bdat->hash_lists[i]); + + ret = syscall(__NR_io_setup, bdat->nr_events, &bdat->ctx); + +out: + if (ret < 0) { + free(bdat->iocbs); + free(bdat->iocbps); + free(bdat->events); + free(bdat->hash_lists); + } + + return ret; +} + +void block_shutdown(void) +{ + struct block_data *bdat = &global_bdat; + + syscall(SYS_io_destroy, bdat->ctx); + + free(bdat->iocbs); + free(bdat->iocbps); + free(bdat->events); + free(bdat->hash_lists); +} diff --git a/utils/src/check/block.h b/utils/src/check/block.h new file mode 100644 index 00000000..ad7195ce --- /dev/null +++ b/utils/src/check/block.h @@ -0,0 +1,32 @@ +#ifndef _SCOUTFS_UTILS_CHECK_BLOCK_H_ +#define _SCOUTFS_UTILS_CHECK_BLOCK_H_ + +#include +#include + +struct block; + +#include "sparse.h" + +/* block flags passed to block_get() */ +enum { + BF_ZERO = (1 << 0), /* zero contents buf as block is returned */ + BF_DIRTY = (1 << 1), /* block will be written with transaction */ + BF_SM = (1 << 2), /* small 4k block instead of large 64k block */ + BF_OVERWRITE = (1 << 3), /* caller will overwrite contents, don't read */ +}; + +int block_get(struct block **blk_ret, u64 blkno, int bf); +void block_put(struct block **blkp); + +void *block_buf(struct block *blk); +size_t block_size(struct block *blk); +void block_drop(struct block **blkp); + +void block_readahead(u64 *blknos, size_t nr); +int block_try_commit(bool force); + +int block_setup(int meta_fd, size_t max_cached_bytes, size_t max_dirty_bytes); +void block_shutdown(void); + +#endif diff --git a/utils/src/check/btree.c b/utils/src/check/btree.c new file mode 100644 index 00000000..50bd1fa2 --- /dev/null +++ b/utils/src/check/btree.c @@ -0,0 +1,209 @@ +#include +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" +#include "key.h" +#include "avl.h" + +#include "block.h" +#include "btree.h" +#include "extent.h" +#include "iter.h" +#include "sns.h" +#include "meta.h" +#include "problem.h" + +static inline void *item_val(struct scoutfs_btree_block *bt, struct scoutfs_btree_item *item) +{ + return (void *)bt + le16_to_cpu(item->val_off); +} + +static void readahead_refs(struct scoutfs_btree_block *bt) +{ + struct scoutfs_btree_item *item; + struct scoutfs_avl_node *node; + struct scoutfs_block_ref *ref; + u64 *blknos; + u64 blkno; + u16 valid = 0; + u16 nr = le16_to_cpu(bt->nr_items); + int i; + + blknos = calloc(nr, sizeof(blknos[0])); + if (!blknos) + return; + + node = avl_first(&bt->item_root); + + for (i = 0; i < nr; i++) { + item = container_of(node, struct scoutfs_btree_item, node); + ref = item_val(bt, item); + blkno = le64_to_cpu(ref->blkno); + + if (valid_meta_blkno(blkno)) + blknos[valid++] = blkno; + + node = avl_next(&bt->item_root, &item->node); + } + + if (valid > 0) + block_readahead(blknos, valid); + free(blknos); +} + +/* + * Call the callback on the referenced block. Then if the block + * contains referneces read it and recurse into all its references. + */ +static int btree_ref_meta_iter(struct scoutfs_block_ref *ref, unsigned level, extent_cb_t cb, + void *cb_arg) +{ + struct scoutfs_btree_item *item; + struct scoutfs_btree_block *bt; + struct scoutfs_avl_node *node; + struct block *blk = NULL; + u64 blkno; + int ret; + int i; + + blkno = le64_to_cpu(ref->blkno); + if (!blkno) + return 0; + + ret = cb(blkno, 1, cb_arg); + if (ret < 0) { + ret = xlate_iter_errno(ret); + return 0; + } + + if (level == 0) + return 0; + + ret = block_get(&blk, blkno, 0); + if (ret < 0) + return ret; + + sns_push("btree_parent", blkno, 0); + + bt = block_buf(blk); + + /* XXX integrate verification with block cache */ + if (bt->level != level) { + problem(PB_BTREE_BLOCK_BAD_LEVEL, "expected %u level %u", level, bt->level); + ret = -EINVAL; + goto out; + } + + /* read-ahead last level of parents */ + if (level == 2) + readahead_refs(bt); + + node = avl_first(&bt->item_root); + + for (i = 0; i < le16_to_cpu(bt->nr_items); i++) { + item = container_of(node, struct scoutfs_btree_item, node); + ref = item_val(bt, item); + + ret = btree_ref_meta_iter(ref, level - 1, cb, cb_arg); + if (ret < 0) + goto out; + + node = avl_next(&bt->item_root, &item->node); + } + + ret = 0; +out: + block_put(&blk); + sns_pop(); + + return ret; +} + +int btree_meta_iter(struct scoutfs_btree_root *root, extent_cb_t cb, void *cb_arg) +{ + /* XXX check root */ + if (root->height == 0) + return 0; + + return btree_ref_meta_iter(&root->ref, root->height - 1, cb, cb_arg); +} + +static int btree_ref_item_iter(struct scoutfs_block_ref *ref, unsigned level, + btree_item_cb_t cb, void *cb_arg) +{ + struct scoutfs_btree_item *item; + struct scoutfs_btree_block *bt; + struct scoutfs_avl_node *node; + struct block *blk = NULL; + u64 blkno; + int ret; + int i; + + blkno = le64_to_cpu(ref->blkno); + if (!blkno) + return 0; + + ret = block_get(&blk, blkno, 0); + if (ret < 0) + return ret; + + if (level) + sns_push("btree_parent", blkno, 0); + else + sns_push("btree_leaf", blkno, 0); + + bt = block_buf(blk); + + /* XXX integrate verification with block cache */ + if (bt->level != level) { + problem(PB_BTREE_BLOCK_BAD_LEVEL, "expected %u level %u", level, bt->level); + ret = -EINVAL; + goto out; + } + + /* read-ahead leaves that contain items */ + if (level == 1) + readahead_refs(bt); + + node = avl_first(&bt->item_root); + + for (i = 0; i < le16_to_cpu(bt->nr_items); i++) { + item = container_of(node, struct scoutfs_btree_item, node); + + if (level) { + ref = item_val(bt, item); + ret = btree_ref_item_iter(ref, level - 1, cb, cb_arg); + } else { + ret = cb(&item->key, item_val(bt, item), + le16_to_cpu(item->val_len), cb_arg); + debug("free item key "SK_FMT" ret %d", SK_ARG(&item->key), ret); + } + if (ret < 0) { + ret = xlate_iter_errno(ret); + goto out; + } + + node = avl_next(&bt->item_root, &item->node); + } + + ret = 0; +out: + block_put(&blk); + sns_pop(); + + return ret; +} + +int btree_item_iter(struct scoutfs_btree_root *root, btree_item_cb_t cb, void *cb_arg) +{ + /* XXX check root */ + if (root->height == 0) + return 0; + + return btree_ref_item_iter(&root->ref, root->height - 1, cb, cb_arg); +} diff --git a/utils/src/check/btree.h b/utils/src/check/btree.h new file mode 100644 index 00000000..dc0b3bf9 --- /dev/null +++ b/utils/src/check/btree.h @@ -0,0 +1,14 @@ +#ifndef _SCOUTFS_UTILS_CHECK_BTREE_H_ +#define _SCOUTFS_UTILS_CHECK_BTREE_H_ + +#include "util.h" +#include "format.h" + +#include "extent.h" + +typedef int (*btree_item_cb_t)(struct scoutfs_key *key, void *val, u16 val_len, void *cb_arg); + +int btree_meta_iter(struct scoutfs_btree_root *root, extent_cb_t cb, void *cb_arg); +int btree_item_iter(struct scoutfs_btree_root *root, btree_item_cb_t cb, void *cb_arg); + +#endif diff --git a/utils/src/check/check.c b/utils/src/check/check.c new file mode 100644 index 00000000..b74b5c52 --- /dev/null +++ b/utils/src/check/check.c @@ -0,0 +1,152 @@ +#define _GNU_SOURCE /* O_DIRECT */ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "sparse.h" +#include "parse.h" +#include "util.h" +#include "format.h" +#include "ioctl.h" +#include "cmd.h" +#include "dev.h" + +#include "alloc.h" +#include "block.h" +#include "debug.h" +#include "meta.h" +#include "super.h" + +struct check_args { + char *meta_device; + char *data_device; + char *debug_path; +}; + +static int do_check(struct check_args *args) +{ + int debug_fd = -1; + int meta_fd = -1; + int data_fd = -1; + int ret; + + if (args->debug_path) { + if (strcmp(args->debug_path, "-") == 0) + debug_fd = dup(STDERR_FILENO); + else + debug_fd = open(args->debug_path, O_WRONLY | O_CREAT | O_TRUNC, 0644); + if (debug_fd < 0) { + ret = -errno; + fprintf(stderr, "error opening debug output file '%s': %s (%d)\n", + args->debug_path, strerror(errno), errno); + goto out; + } + + debug_enable(debug_fd); + } + + meta_fd = open(args->meta_device, O_DIRECT | O_RDWR | O_EXCL); + if (meta_fd < 0) { + ret = -errno; + fprintf(stderr, "failed to open meta device '%s': %s (%d)\n", + args->meta_device, strerror(errno), errno); + goto out; + } + + data_fd = open(args->data_device, O_DIRECT | O_RDWR | O_EXCL); + if (data_fd < 0) { + ret = -errno; + fprintf(stderr, "failed to open data device '%s': %s (%d)\n", + args->data_device, strerror(errno), errno); + goto out; + } + + ret = block_setup(meta_fd, 128 * 1024 * 1024, 32 * 1024 * 1024); + if (ret < 0) + goto out; + + ret = check_supers() ?: + check_meta_refs(); +out: + /* and tear it all down */ + block_shutdown(); + super_shutdown(); + debug_disable(); + + if (meta_fd >= 0) + close(meta_fd); + if (data_fd >= 0) + close(data_fd); + if (debug_fd >= 0) + close(debug_fd); + + return ret; +} + +static int parse_opt(int key, char *arg, struct argp_state *state) +{ + struct check_args *args = state->input; + + switch (key) { + case 'd': + args->debug_path = strdup_or_error(state, arg); + break; + case 'e': + case ARGP_KEY_ARG: + if (!args->meta_device) + args->meta_device = strdup_or_error(state, arg); + else if (!args->data_device) + args->data_device = strdup_or_error(state, arg); + else + argp_error(state, "more than two device arguments given"); + break; + case ARGP_KEY_FINI: + if (!args->meta_device) + argp_error(state, "no metadata device argument given"); + if (!args->data_device) + argp_error(state, "no data device argument given"); + break; + default: + break; + } + + return 0; +} + +static struct argp_option options[] = { + { "debug", 'd', "FILE_PATH", 0, "Path to debug output file, will be created or truncated"}, + { NULL } +}; + +static struct argp argp = { + options, + parse_opt, + "META-DEVICE DATA-DEVICE", + "Check filesystem consistency" +}; + +static int check_cmd(int argc, char **argv) +{ + struct check_args check_args = {NULL}; + int ret; + + ret = argp_parse(&argp, argc, argv, 0, NULL, &check_args); + if (ret) + return ret; + + return do_check(&check_args); +} + +static void __attribute__((constructor)) check_ctor(void) +{ + cmd_register_argp("check", &argp, GROUP_CORE, check_cmd); +} diff --git a/utils/src/check/debug.c b/utils/src/check/debug.c new file mode 100644 index 00000000..0017c1aa --- /dev/null +++ b/utils/src/check/debug.c @@ -0,0 +1,16 @@ +#include + +#include "debug.h" + +int debug_fd = -1; + +void debug_enable(int fd) +{ + debug_fd = fd; +} + +void debug_disable(void) +{ + if (debug_fd >= 0) + debug_fd = -1; +} diff --git a/utils/src/check/debug.h b/utils/src/check/debug.h new file mode 100644 index 00000000..a5103494 --- /dev/null +++ b/utils/src/check/debug.h @@ -0,0 +1,17 @@ +#ifndef _SCOUTFS_UTILS_CHECK_DEBUG_H_ +#define _SCOUTFS_UTILS_CHECK_DEBUG_H_ + +#include + +#define debug(fmt, args...) \ +do { \ + if (debug_fd >= 0) \ + dprintf(debug_fd, fmt"\n", ##args); \ +} while (0) + +extern int debug_fd; + +void debug_enable(int fd); +void debug_disable(void); + +#endif diff --git a/utils/src/check/eno.h b/utils/src/check/eno.h new file mode 100644 index 00000000..14579fce --- /dev/null +++ b/utils/src/check/eno.h @@ -0,0 +1,9 @@ +#ifndef _SCOUTFS_UTILS_CHECK_ENO_H_ +#define _SCOUTFS_UTILS_CHECK_ENO_H_ + +#include + +#define ENO_FMT "%d (%s)" +#define ENO_ARG(eno) eno, strerror(eno) + +#endif diff --git a/utils/src/check/extent.c b/utils/src/check/extent.c new file mode 100644 index 00000000..bbbcc887 --- /dev/null +++ b/utils/src/check/extent.c @@ -0,0 +1,313 @@ +#include +#include +#include +#include +#include + +#include "util.h" +#include "lk_rbtree_wrapper.h" + +#include "debug.h" +#include "extent.h" + +/* + * In-memory extent management in rbtree nodes. + */ + +bool extents_overlap(u64 a_start, u64 a_len, u64 b_start, u64 b_len) +{ + u64 a_end = a_start + a_len; + u64 b_end = b_start + b_len; + + return !((a_end <= b_start) || (b_end <= a_start)); +} + +static int ext_contains(struct extent_node *ext, u64 start, u64 len) +{ + return ext->start <= start && ext->start + ext->len >= start + len; +} + +/* + * True if the given extent is bisected by the given range; there's + * leftover containing extents on both the left and right sides of the + * range in the extent. + */ +static int ext_bisected(struct extent_node *ext, u64 start, u64 len) +{ + return ext->start < start && ext->start + ext->len > start + len; +} + +static struct extent_node *ext_from_rbnode(struct rb_node *rbnode) +{ + return rbnode ? container_of(rbnode, struct extent_node, rbnode) : NULL; +} + +static struct extent_node *next_ext(struct extent_node *ext) +{ + return ext ? ext_from_rbnode(rb_next(&ext->rbnode)) : NULL; +} + +static struct extent_node *prev_ext(struct extent_node *ext) +{ + return ext ? ext_from_rbnode(rb_prev(&ext->rbnode)) : NULL; +} + +struct walk_results { + unsigned bisect_to_leaf:1; + struct extent_node *found; + struct extent_node *next; + struct rb_node *parent; + struct rb_node **node; +}; + +static void walk_extents(struct extent_root *root, u64 start, u64 len, struct walk_results *wlk) +{ + struct rb_node **node = &root->rbroot.rb_node; + struct extent_node *ext; + u64 end = start + len; + int cmp; + + wlk->found = NULL; + wlk->next = NULL; + wlk->parent = NULL; + + while (*node) { + wlk->parent = *node; + ext = ext_from_rbnode(*node); + cmp = end <= ext->start ? -1 : + start >= ext->start + ext->len ? 1 : 0; + + if (cmp < 0) { + node = &ext->rbnode.rb_left; + wlk->next = ext; + } else if (cmp > 0) { + node = &ext->rbnode.rb_right; + } else { + wlk->found = ext; + if (!(wlk->bisect_to_leaf && ext_bisected(ext, start, len))) + break; + /* walk right so we can insert greater right from bisection */ + node = &ext->rbnode.rb_right; + } + } + + wlk->node = node; +} + +/* + * Return an extent that overlaps with the given range. + */ +int extent_lookup(struct extent_root *root, u64 start, u64 len, struct extent_node *found) +{ + struct walk_results wlk = { 0, }; + int ret; + + walk_extents(root, start, len, &wlk); + if (wlk.found) { + memset(found, 0, sizeof(struct extent_node)); + found->start = wlk.found->start; + found->len = wlk.found->len; + ret = 0; + } else { + ret = -ENOENT; + } + + return ret; +} + +/* + * Callers can iterate through direct node references and are entirely + * responsible for consistency when doing so. + */ +struct extent_node *extent_first(struct extent_root *root) +{ + struct walk_results wlk = { 0, }; + + walk_extents(root, 0, 1, &wlk); + + return wlk.found ?: wlk.next; +} + +struct extent_node *extent_next(struct extent_node *ext) +{ + return next_ext(ext); +} + +struct extent_node *extent_prev(struct extent_node *ext) +{ + return prev_ext(ext); +} + +/* + * Insert a new extent into the tree. We can extend existing nodes, + * merge with neighbours, or remove existing extents entirely if we + * insert a range that fully spans existing nodes. + */ +static int walk_insert(struct extent_root *root, u64 start, u64 len, int found_err) +{ + struct walk_results wlk = { 0, }; + struct extent_node *ext; + struct extent_node *nei; + int ret; + + walk_extents(root, start, len, &wlk); + + ext = wlk.found; + if (ext && found_err) { + ret = found_err; + goto out; + } + + if (!ext) { + ext = malloc(sizeof(struct extent_node)); + if (!ext) { + ret = -ENOMEM; + goto out; + } + + ext->start = start; + ext->len = len; + + rb_link_node(&ext->rbnode, wlk.parent, wlk.node); + rb_insert_color(&ext->rbnode, &root->rbroot); + } + + /* start by expanding an existing extent if our range is larger */ + if (start < ext->start) { + ext->len += ext->start - start; + ext->start = start; + } + if (ext->start + ext->len < start + len) + ext->len += (start + len) - (ext->start + ext->len); + + /* drop any fully spanned neighbors, possibly merging with a final adjacent one */ + + while ((nei = prev_ext(ext))) { + if (nei->start + nei->len < ext->start) + break; + + if (nei->start < ext->start) { + ext->len += ext->start - nei->start; + ext->start = nei->start; + } + + rb_erase(&nei->rbnode, &root->rbroot); + free(nei); + } + + while ((nei = next_ext(ext))) { + if (ext->start + ext->len < nei->start) + break; + + if (ext->start + ext->len < nei->start + nei->len) + ext->len += (nei->start + nei->len) - (ext->start + ext->len); + + rb_erase(&nei->rbnode, &root->rbroot); + free(nei); + } + + ret = 0; +out: + if (ret < 0) + debug("start %llu len %llu ret %d", start, len, ret); + return ret; +} + +/* + * Insert a new extent. The specified extent must not overlap with any + * existing extents or -EEXIST is returned. + */ +int extent_insert_new(struct extent_root *root, u64 start, u64 len) +{ + return walk_insert(root, start, len, true); +} + +/* + * Insert an extent, extending any existing extents that may overlap. + */ +int extent_insert_extend(struct extent_root *root, u64 start, u64 len) +{ + return walk_insert(root, start, len, false); +} + +/* + * Remove the specified extent from an existing node. The given extent must be fully + * contained in a single node or -ENOENT is returned. + */ +int extent_remove(struct extent_root *root, u64 start, u64 len) +{ + struct extent_node *ext; + struct extent_node *ins; + struct walk_results wlk = { + .bisect_to_leaf = 1, + }; + int ret; + + walk_extents(root, start, len, &wlk); + + if (!(ext = wlk.found) || !ext_contains(ext, start, len)) { + ret = -ENOENT; + goto out; + } + + if (ext_bisected(ext, start, len)) { + debug("found bisected start %llu len %llu", ext->start, ext->len); + ins = malloc(sizeof(struct extent_node)); + if (!ins) { + ret = -ENOMEM; + goto out; + } + + ins->start = start + len; + ins->len = (ext->start + ext->len) - ins->start; + + rb_link_node(&ins->rbnode, wlk.parent, wlk.node); + rb_insert_color(&ins->rbnode, &root->rbroot); + } + + if (start > ext->start) { + ext->len = start - ext->start; + } else if (len < ext->len) { + ext->start += len; + ext->len -= len; + } else { + rb_erase(&ext->rbnode, &root->rbroot); + } + + ret = 0; +out: + debug("start %llu len %llu ret %d", start, len, ret); + + return ret; +} + +void extent_root_init(struct extent_root *root) +{ + root->rbroot = RB_ROOT; + root->total = 0; +} + +void extent_root_free(struct extent_root *root) +{ + struct extent_node *ext; + struct rb_node *node; + struct rb_node *tmp; + + for (node = rb_first(&root->rbroot); node && ((tmp = rb_next(node)), 1); node = tmp) { + ext = rb_entry(node, struct extent_node, rbnode); + rb_erase(&ext->rbnode, &root->rbroot); + free(ext); + } +} + +void extent_root_print(struct extent_root *root) +{ + struct extent_node *ext; + struct rb_node *node; + struct rb_node *tmp; + + for (node = rb_first(&root->rbroot); node && ((tmp = rb_next(node)), 1); node = tmp) { + ext = rb_entry(node, struct extent_node, rbnode); + debug(" start %llu len %llu", ext->start, ext->len); + } +} diff --git a/utils/src/check/extent.h b/utils/src/check/extent.h new file mode 100644 index 00000000..2a38f765 --- /dev/null +++ b/utils/src/check/extent.h @@ -0,0 +1,38 @@ +#ifndef _SCOUTFS_UTILS_CHECK_EXTENT_H_ +#define _SCOUTFS_UTILS_CHECK_EXTENT_H_ + +#include "lk_rbtree_wrapper.h" + +struct extent_root { + struct rb_root rbroot; + u64 total; +}; + +struct extent_node { + struct rb_node rbnode; + u64 start; + u64 len; +}; + +typedef int (*extent_cb_t)(u64 start, u64 len, void *arg); + +struct extent_cb_arg_t { + extent_cb_t cb; + void *cb_arg; +}; + +bool extents_overlap(u64 a_start, u64 a_len, u64 b_start, u64 b_len); + +int extent_lookup(struct extent_root *root, u64 start, u64 len, struct extent_node *found); +struct extent_node *extent_first(struct extent_root *root); +struct extent_node *extent_next(struct extent_node *ext); +struct extent_node *extent_prev(struct extent_node *ext); +int extent_insert_new(struct extent_root *root, u64 start, u64 len); +int extent_insert_extend(struct extent_root *root, u64 start, u64 len); +int extent_remove(struct extent_root *root, u64 start, u64 len); + +void extent_root_init(struct extent_root *root); +void extent_root_free(struct extent_root *root); +void extent_root_print(struct extent_root *root); + +#endif diff --git a/utils/src/check/iter.h b/utils/src/check/iter.h new file mode 100644 index 00000000..54c5d13b --- /dev/null +++ b/utils/src/check/iter.h @@ -0,0 +1,15 @@ +#ifndef _SCOUTFS_UTILS_CHECK_ITER_H_ +#define _SCOUTFS_UTILS_CHECK_ITER_H_ + +/* + * Callbacks can return a weird -errno that we'll never use to indicate + * that iteration can stop and return 0 for success. + */ +#define ECHECK_ITER_DONE EL2HLT + +static inline int xlate_iter_errno(int ret) +{ + return ret == -ECHECK_ITER_DONE ? 0 : ret; +} + +#endif diff --git a/utils/src/check/log_trees.c b/utils/src/check/log_trees.c new file mode 100644 index 00000000..627052c7 --- /dev/null +++ b/utils/src/check/log_trees.c @@ -0,0 +1,98 @@ +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" +#include "key.h" + +#include "alloc.h" +#include "btree.h" +#include "debug.h" +#include "extent.h" +#include "iter.h" +#include "sns.h" +#include "log_trees.h" +#include "super.h" + +struct iter_args { + extent_cb_t cb; + void *cb_arg; +}; + +static int lt_meta_iter(struct scoutfs_key *key, void *val, u16 val_len, void *cb_arg) +{ + struct iter_args *ia = cb_arg; + struct scoutfs_log_trees *lt; + int ret; + + if (val_len != sizeof(struct scoutfs_log_trees)) + ; /* XXX */ + + lt = val; + + sns_push("log_trees", le64_to_cpu(lt->rid), le64_to_cpu(lt->nr)); + + debug("lt rid 0x%16llx nr %llu", le64_to_cpu(lt->rid), le64_to_cpu(lt->nr)); + + sns_push("meta_avail", 0, 0); + ret = alloc_list_meta_iter(<->meta_avail, ia->cb, ia->cb_arg); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("meta_freed", 0, 0); + ret = alloc_list_meta_iter(<->meta_freed, ia->cb, ia->cb_arg); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("item_root", 0, 0); + ret = btree_meta_iter(<->item_root, ia->cb, ia->cb_arg); + sns_pop(); + if (ret < 0) + goto out; + + if (lt->bloom_ref.blkno) { + sns_push("bloom_ref", 0, 0); + ret = ia->cb(le64_to_cpu(lt->bloom_ref.blkno), 1, ia->cb_arg); + sns_pop(); + if (ret < 0) { + ret = xlate_iter_errno(ret); + goto out; + } + } + + sns_push("data_avail", 0, 0); + ret = alloc_root_meta_iter(<->data_avail, ia->cb, ia->cb_arg); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("data_freed", 0, 0); + ret = alloc_root_meta_iter(<->data_freed, ia->cb, ia->cb_arg); + sns_pop(); + if (ret < 0) + goto out; + + ret = 0; +out: + sns_pop(); + + return ret; +} + +/* + * Call the callers callback with the extent of all the metadata block references contained + * in log btrees. We walk the logs_root btree items and walk all the metadata structures + * they reference. + */ +int log_trees_meta_iter(extent_cb_t cb, void *cb_arg) +{ + struct scoutfs_super_block *super = global_super; + struct iter_args ia = { .cb = cb, .cb_arg = cb_arg }; + + return btree_item_iter(&super->logs_root, lt_meta_iter, &ia); +} diff --git a/utils/src/check/log_trees.h b/utils/src/check/log_trees.h new file mode 100644 index 00000000..7a7150b1 --- /dev/null +++ b/utils/src/check/log_trees.h @@ -0,0 +1,8 @@ +#ifndef _SCOUTFS_UTILS_CHECK_LOG_TREES_H_ +#define _SCOUTFS_UTILS_CHECK_LOG_TREES_H_ + +#include "extent.h" + +int log_trees_meta_iter(extent_cb_t cb, void *cb_arg); + +#endif diff --git a/utils/src/check/meta.c b/utils/src/check/meta.c new file mode 100644 index 00000000..40a2e5a5 --- /dev/null +++ b/utils/src/check/meta.c @@ -0,0 +1,367 @@ +#include +#include +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" +#include "bitmap.h" +#include "key.h" + +#include "alloc.h" +#include "btree.h" +#include "debug.h" +#include "extent.h" +#include "sns.h" +#include "log_trees.h" +#include "meta.h" +#include "problem.h" +#include "super.h" + +static struct meta_data { + struct extent_root meta_refed; + struct extent_root meta_free; + struct { + u64 ref_blocks; + u64 free_extents; + u64 free_blocks; + } stats; +} global_mdat; + +bool valid_meta_blkno(u64 blkno) +{ + u64 tot = le64_to_cpu(global_super->total_meta_blocks); + + return blkno >= SCOUTFS_META_DEV_START_BLKNO && blkno < tot; +} + +static bool valid_meta_extent(u64 start, u64 len) +{ + u64 tot = le64_to_cpu(global_super->total_meta_blocks); + bool valid; + + valid = len > 0 && + start >= SCOUTFS_META_DEV_START_BLKNO && + start < tot && + len <= tot && + ((start + len) <= tot) && + ((start + len) > start); + + debug("start %llu len %llu valid %u", start, len, !!valid); + + if (!valid) + problem(PB_META_EXTENT_INVALID, "start %llu len %llu", start, len); + + return valid; +} + +/* + * Track references to individual metadata blocks. This uses the extent + * callback type but is only ever called for single block references. + * Any reference to a block that has already been referenced is + * considered invalid and is ignored. Later repair will resolve + * duplicate references. + */ +static int insert_meta_ref(u64 start, u64 len, void *arg) +{ + struct meta_data *mdat = &global_mdat; + struct extent_root *root = arg; + int ret = 0; + + /* this is tracking single metadata block references */ + if (len != 1) { + ret = -EINVAL; + goto out; + } + + if (valid_meta_blkno(start)) { + ret = extent_insert_new(root, start, len); + if (ret == 0) + mdat->stats.ref_blocks++; + else if (ret == -EEXIST) + problem(PB_META_REF_OVERLAPS_EXISTING, "blkno %llu", start); + } + +out: + return ret; +} + +static int insert_meta_free(u64 start, u64 len, void *arg) +{ + struct meta_data *mdat = &global_mdat; + struct extent_root *root = arg; + int ret = 0; + + if (valid_meta_extent(start, len)) { + ret = extent_insert_new(root, start, len); + if (ret == 0) { + mdat->stats.free_extents++; + mdat->stats.free_blocks++; + + } else if (ret == -EEXIST) { + problem(PB_META_FREE_OVERLAPS_EXISTING, + "start %llu llen %llu", start, len); + } + + } + + return ret; +} + +/* + * Walk all metadata references in the system. This walk doesn't need + * to read metadata that doesn't contain any metadata references so it + * can skip the bulk of metadata blocks. This gives us the set of + * referenced metadata blocks which we can then use to repair metadata + * allocator structures. + */ +static int get_meta_refs(void) +{ + struct meta_data *mdat = &global_mdat; + struct scoutfs_super_block *super = global_super; + int ret; + + extent_root_init(&mdat->meta_refed); + + /* XXX record reserved blocks around super as referenced */ + + sns_push("meta_alloc", 0, 0); + ret = alloc_root_meta_iter(&super->meta_alloc[0], insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("meta_alloc", 1, 0); + ret = alloc_root_meta_iter(&super->meta_alloc[1], insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("data_alloc", 1, 0); + ret = alloc_root_meta_iter(&super->data_alloc, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_avail", 0, 0); + ret = alloc_list_meta_iter(&super->server_meta_avail[0], + insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_avail", 1, 0); + ret = alloc_list_meta_iter(&super->server_meta_avail[1], + insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_freed", 0, 0); + ret = alloc_list_meta_iter(&super->server_meta_freed[0], + insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_freed", 1, 0); + ret = alloc_list_meta_iter(&super->server_meta_freed[1], + insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("fs_root", 0, 0); + ret = btree_meta_iter(&super->fs_root, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("logs_root", 0, 0); + ret = btree_meta_iter(&super->logs_root, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("log_merge", 0, 0); + ret = btree_meta_iter(&super->log_merge, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("mounted_clients", 0, 0); + ret = btree_meta_iter(&super->mounted_clients, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("srch_root", 0, 0); + ret = btree_meta_iter(&super->srch_root, insert_meta_ref, &mdat->meta_refed); + sns_pop(); + if (ret < 0) + goto out; + + ret = log_trees_meta_iter(insert_meta_ref, &mdat->meta_refed); + if (ret < 0) + goto out; + + debug("found %llu referenced metadata blocks", mdat->stats.ref_blocks); + ret = 0; +out: + return ret; +} + +static int get_meta_free(void) +{ + struct meta_data *mdat = &global_mdat; + struct scoutfs_super_block *super = global_super; + int ret; + + extent_root_init(&mdat->meta_free); + + sns_push("meta_alloc", 0, 0); + ret = alloc_root_extent_iter(&super->meta_alloc[0], insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("meta_alloc", 1, 0); + ret = alloc_root_extent_iter(&super->meta_alloc[1], insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_avail", 0, 0); + ret = alloc_list_extent_iter(&super->server_meta_avail[0], + insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_avail", 1, 0); + ret = alloc_list_extent_iter(&super->server_meta_avail[1], + insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_freed", 0, 0); + ret = alloc_list_extent_iter(&super->server_meta_freed[0], + insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + sns_push("server_meta_freed", 1, 0); + ret = alloc_list_extent_iter(&super->server_meta_freed[1], + insert_meta_free, &mdat->meta_free); + sns_pop(); + if (ret < 0) + goto out; + + debug("found %llu free metadata blocks in %llu extents", + mdat->stats.free_blocks, mdat->stats.free_extents); + ret = 0; +out: + return ret; +} + +/* + * All the space between referenced blocks must be recorded in the free + * extents. The free extent walk didn't check that the extents + * overlapped with references, we do that here. Remember that metadata + * block references were merged into extents here, the refed extents + * aren't necessarily all a single block. + */ +static int compare_refs_and_free(void) +{ + struct meta_data *mdat = &global_mdat; + struct extent_node *ref; + struct extent_node *free; + struct extent_node *next; + struct extent_node *prev; + u64 expect; + u64 start; + u64 end; + + expect = 0; + ref = extent_first(&mdat->meta_refed); + free = extent_first(&mdat->meta_free); + while (ref || free) { + + debug("exp %llu ref %llu.%llu free %llu.%llu", + expect, ref ? ref->start : 0, ref ? ref->len : 0, + free ? free->start : 0, free ? free->len : 0); + + /* referenced marked free, remove ref from free and continue from same point */ + if (ref && free && extents_overlap(ref->start, ref->len, free->start, free->len)) { + debug("ref extent %llu.%llu overlaps free %llu %llu", + ref->start, ref->len, free->start, free->len); + + start = max(ref->start, free->start); + end = min(ref->start + ref->len, free->start + free->len); + + prev = extent_prev(free); + + extent_remove(&mdat->meta_free, start, end - start); + + if (prev) + free = extent_next(prev); + else + free = extent_first(&mdat->meta_free); + continue; + } + + /* see which extent starts earlier */ + if (!free || (ref && ref->start <= free->start)) + next = ref; + else + next = free; + + /* untracked region before next extent */ + if (expect < next->start) { + debug("missing free extent %llu.%llu", expect, next->start - expect); + expect = next->start; + continue; + } + + + /* didn't overlap, advance past next extent */ + expect = next->start + next->len; + if (next == ref) + ref = extent_next(ref); + else + free = extent_next(free); + } + + return 0; +} + +/* + * Check the metadata allocators by comparing the set of referenced + * blocks with the set of free blocks that are stored in free btree + * items and alloc list blocks. + */ +int check_meta_alloc(void) +{ + int ret; + + ret = get_meta_refs(); + if (ret < 0) + goto out; + + ret = get_meta_free(); + if (ret < 0) + goto out; + + ret = compare_refs_and_free(); + if (ret < 0) + goto out; + + ret = 0; +out: + return ret; +} diff --git a/utils/src/check/meta.h b/utils/src/check/meta.h new file mode 100644 index 00000000..80c97a03 --- /dev/null +++ b/utils/src/check/meta.h @@ -0,0 +1,9 @@ +#ifndef _SCOUTFS_UTILS_CHECK_META_H_ +#define _SCOUTFS_UTILS_CHECK_META_H_ + +bool valid_meta_blkno(u64 blkno); + +int check_meta_alloc(void); + +#endif + diff --git a/utils/src/check/padding.c b/utils/src/check/padding.c new file mode 100644 index 00000000..81e12c33 --- /dev/null +++ b/utils/src/check/padding.c @@ -0,0 +1,23 @@ +#include +#include + +#include "util.h" +#include "padding.h" + +bool padding_is_zeros(const void *data, size_t sz) +{ + static char zeros[32] = {0,}; + const size_t batch = array_size(zeros); + + while (sz >= batch) { + if (memcmp(data, zeros, batch)) + return false; + data += batch; + sz -= batch; + } + + if (sz > 0 && memcmp(data, zeros, sz)) + return false; + + return true; +} diff --git a/utils/src/check/padding.h b/utils/src/check/padding.h new file mode 100644 index 00000000..9bf03a81 --- /dev/null +++ b/utils/src/check/padding.h @@ -0,0 +1,6 @@ +#ifndef _SCOUTFS_UTILS_CHECK_PADDING_H_ +#define _SCOUTFS_UTILS_CHECK_PADDING_H_ + +bool padding_is_zeros(const void *data, size_t sz); + +#endif diff --git a/utils/src/check/problem.c b/utils/src/check/problem.c new file mode 100644 index 00000000..2191726f --- /dev/null +++ b/utils/src/check/problem.c @@ -0,0 +1,23 @@ +#include +#include + +#include "problem.h" + +#if 0 +#define PROB_STR(pb) [pb] = #pb +static char *prob_strs[] = { + PROB_STR(PB_META_EXTENT_INVALID), + PROB_STR(PB_META_EXTENT_OVERLAPS_EXISTING), +}; +#endif + +static struct problem_data { + uint64_t counts[PB__NR]; +} global_pdat; + +void problem_record(prob_t pb) +{ + struct problem_data *pdat = &global_pdat; + + pdat->counts[pb]++; +} diff --git a/utils/src/check/problem.h b/utils/src/check/problem.h new file mode 100644 index 00000000..ce7b7fde --- /dev/null +++ b/utils/src/check/problem.h @@ -0,0 +1,23 @@ +#ifndef _SCOUTFS_UTILS_CHECK_PROBLEM_H_ +#define _SCOUTFS_UTILS_CHECK_PROBLEM_H_ + +#include "debug.h" +#include "sns.h" + +typedef enum { + PB_META_EXTENT_INVALID, + PB_META_REF_OVERLAPS_EXISTING, + PB_META_FREE_OVERLAPS_EXISTING, + PB_BTREE_BLOCK_BAD_LEVEL, + PB__NR, +} prob_t; + +#define problem(pb, fmt, ...) \ +do { \ + debug("problem found: "#pb": %s: "fmt, sns_str(), __VA_ARGS__); \ + problem_record(pb); \ +} while (0) + +void problem_record(prob_t pb); + +#endif diff --git a/utils/src/check/sns.c b/utils/src/check/sns.c new file mode 100644 index 00000000..45f45453 --- /dev/null +++ b/utils/src/check/sns.c @@ -0,0 +1,118 @@ +#include +#include + +#include "sns.h" + +/* + * This "str num stack" is used to describe our location in metadata at + * any given time. + * + * As we descend into structures we pop a string on decribing them, + * perhaps with associated numbers. Pushing and popping is very cheap + * and only rarely do we format the stack into a string, as an arbitrary + * example: + * super.fs_root.btree_parent:1231.btree_leaf:3231" + */ + +#define SNS_MAX_DEPTH 1000 +#define SNS_STR_SIZE (SNS_MAX_DEPTH * (SNS_MAX_STR_LEN + 1 + 16 + 1)) + +static struct sns_data { + unsigned int depth; + + struct sns_entry { + char *str; + size_t len; + u64 a; + u64 b; + } ents[SNS_MAX_DEPTH]; + + char str[SNS_STR_SIZE]; + +} global_lsdat; + +void _sns_push(char *str, size_t len, u64 a, u64 b) +{ + struct sns_data *lsdat = &global_lsdat; + + if (lsdat->depth < SNS_MAX_DEPTH) { + lsdat->ents[lsdat->depth++] = (struct sns_entry) { + .str = str, + .len = len, + .a = a, + .b = b, + }; + } +} + +void sns_pop(void) +{ + struct sns_data *lsdat = &global_lsdat; + + if (lsdat->depth > 0) + lsdat->depth--; +} + +static char *append_str(char *pos, char *str, size_t len) +{ + memcpy(pos, str, len); + return pos + len; +} + +/* + * This is not called for x = 0 so we don't need to emit an initial 0. + * We could by using do {} while instead of while {}. + */ +static char *append_u64x(char *pos, u64 x) +{ + static char hex[] = "0123456789abcdef"; + + while (x) { + *pos++ = hex[x & 0xf]; + x >>= 4; + } + + return pos; +} + +static char *append_char(char *pos, char c) +{ + *(pos++) = c; + return pos; +} + +/* + * Return a pointer to a null terminated string that describes the + * current location stack. The string buffer is global. + */ +char *sns_str(void) +{ + struct sns_data *lsdat = &global_lsdat; + struct sns_entry *ent; + char *pos; + int i; + + pos = lsdat->str; + for (i = 0; i < lsdat->depth; i++) { + ent = &lsdat->ents[i]; + + if (i) + pos = append_char(pos, '.'); + + pos = append_str(pos, ent->str, ent->len); + + if (ent->a) { + pos = append_char(pos, ':'); + pos = append_u64x(pos, ent->a); + } + + if (ent->b) { + pos = append_char(pos, ':'); + pos = append_u64x(pos, ent->b); + } + } + + *pos = '\0'; + + return lsdat->str; +} diff --git a/utils/src/check/sns.h b/utils/src/check/sns.h new file mode 100644 index 00000000..34c1a2be --- /dev/null +++ b/utils/src/check/sns.h @@ -0,0 +1,20 @@ +#ifndef _SCOUTFS_UTILS_CHECK_SNS_H_ +#define _SCOUTFS_UTILS_CHECK_SNS_H_ + +#include + +#include "sparse.h" + +#define SNS_MAX_STR_LEN 20 + +#define sns_push(str, a, b) \ +do { \ + build_assert(sizeof(str) - 1 <= SNS_MAX_STR_LEN); \ + _sns_push((str), sizeof(str) - 1, a, b); \ +} while (0) + +void _sns_push(char *str, size_t len, u64 a, u64 b); +void sns_pop(void); +char *sns_str(void); + +#endif diff --git a/utils/src/check/super.c b/utils/src/check/super.c new file mode 100644 index 00000000..9c2f078d --- /dev/null +++ b/utils/src/check/super.c @@ -0,0 +1,57 @@ +#include +#include +#include +#include + +#include "sparse.h" +#include "util.h" +#include "format.h" + +#include "block.h" +#include "super.h" + +/* + * After we check the super blocks we provide a global buffer to track + * the current super block. It is referenced to get static information + * about the system and is also modified and written as part of + * transactions. + */ +struct scoutfs_super_block *global_super; + +/* + * After checking the supers we save a copy of it in a global buffer that's used by + * other modules to track the current super. It can be modified and written during commits. + */ +int check_supers(void) +{ + struct scoutfs_super_block *super = NULL; + struct block *blk = NULL; + int ret; + + global_super = malloc(sizeof(struct scoutfs_super_block)); + if (!global_super) { + fprintf(stderr, "error allocating super block buffer\n"); + ret = -ENOMEM; + goto out; + } + + ret = block_get(&blk, SCOUTFS_SUPER_BLKNO, BF_SM); + if (ret < 0) { + fprintf(stderr, "error reading super block\n"); + goto out; + } + + super = block_buf(blk); + + memcpy(global_super, super, sizeof(struct scoutfs_super_block)); + ret = 0; +out: + block_put(&blk); + + return ret; +} + +void super_shutdown(void) +{ + free(global_super); +} diff --git a/utils/src/check/super.h b/utils/src/check/super.h new file mode 100644 index 00000000..7c75ad2d --- /dev/null +++ b/utils/src/check/super.h @@ -0,0 +1,9 @@ +#ifndef _SCOUTFS_UTILS_CHECK_SUPER_H_ +#define _SCOUTFS_UTILS_CHECK_SUPER_H_ + +extern struct scoutfs_super_block *global_super; + +int check_supers(void); +void super_shutdown(void); + +#endif