From 50b9e27f75172dd5c663071465c2f7d17ca45c38 Mon Sep 17 00:00:00 2001 From: William Gill Date: Wed, 22 Apr 2026 10:21:21 -0500 Subject: [PATCH] Initial scoutfs-notify patch series for v1.29 Two-patch series that layers an observer-only file-access notification stream onto scoutfs. See README.md for the design summary and rebase workflow. Base: scoutfs v1.29 --- .gitattributes | 3 + README.md | 72 ++ apply.sh | 26 + base.txt | 1 + ...e-access-notification-infrastructure.patch | 838 ++++++++++++++++++ ...002-notify-file-open-read-hook-sites.patch | 137 +++ 6 files changed, 1077 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 apply.sh create mode 100644 base.txt create mode 100644 patches/0001-notify-core-file-access-notification-infrastructure.patch create mode 100644 patches/0002-notify-file-open-read-hook-sites.patch diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..ad50943 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,3 @@ +* text=auto eol=lf +*.patch binary +apply.sh text eol=lf diff --git a/README.md b/README.md new file mode 100644 index 0000000..f40be18 --- /dev/null +++ b/README.md @@ -0,0 +1,72 @@ +# scoutfs-notify + +Observer-only file access notifications for [ScoutFS](https://github.com/versity/scoutfs). +Maintained as a rebasable `git format-patch` series so it can be layered onto +each upstream release with minimal maintenance. + +## What it adds + +* New mount option `notify_events=0|1` (default `0`) that turns on a per-mount + ring buffer of file **open** and **read** events. +* New mount option `notify_ring_kb=N` (4..4096, default 64) sizing that ring. +* New ioctl `SCOUTFS_IOC_READ_NOTIFY` (nr 25) that lets a single privileged + userspace reader drain the ring. Reader is expected to relay events over a + Unix socket to a watcher daemon. +* Three percpu counters: `notify_emitted`, `notify_dropped_ring_full`, + `notify_reader_attached`. + +The notification path is strictly observer-only: if the ring fills, records +are dropped but the monotonic `seq` field still advances so readers see a +gap. `notify_events=0` reduces the hook to a single predicted-false branch. + +Nothing in the data-waiter state machine is touched by this series; a future +patch series may add data-waiter observation events (reserved type values 3+). + +## Base + +Currently rebased against: + + scoutfs v1.29 + +See [base.txt](./base.txt) for the exact tag the patches target. + +## Applying + +On a git checkout of scoutfs at the base tag: + +```sh +./apply.sh /path/to/scoutfs +``` + +The script runs `git am --3way` against each `patches/*.patch`. Requires the +target to be a git working tree. + +For a non-git tarball, apply with: + +```sh +cd /path/to/scoutfs +for p in /path/to/scoutfs-notify/patches/*.patch; do + patch -p1 < "$p" +done +``` + +## Rebasing onto a new upstream release + +```sh +# In your scoutfs working clone +git fetch --tags +git checkout -B notify v1.30 # new upstream tag +git am --3way patches/*.patch # from this repo +# ... resolve any conflicts, git am --continue ... +git format-patch v1.30..notify -o patches/ +# Update base.txt and commit the refreshed patches +``` + +## Tagging + +Tag releases of the patch set with both the upstream version and a patch +revision so consumers can pin precisely: + + v1.29-notify-1 + v1.29-notify-2 + v1.30-notify-1 diff --git a/apply.sh b/apply.sh new file mode 100644 index 0000000..9f54f2e --- /dev/null +++ b/apply.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +# +# Apply the scoutfs-notify patch series to a scoutfs working tree. +# +# Usage: apply.sh /path/to/scoutfs +# +set -euo pipefail + +HERE=$(cd "$(dirname "$0")" && pwd) +TREE="${1:?usage: $0 /path/to/scoutfs}" + +if [ ! -d "$TREE/.git" ]; then + echo "error: $TREE is not a git working tree" >&2 + echo "hint: use 'patch -p1' directly for tarball trees (see README)" >&2 + exit 2 +fi + +BASE="$(head -n1 "$HERE/base.txt" | awk '{print $1}')" +if [ -n "$BASE" ]; then + if ! git -C "$TREE" rev-parse --verify "$BASE" >/dev/null 2>&1; then + echo "warning: base '$BASE' not found in target tree" >&2 + fi +fi + +cd "$TREE" +git am --3way --keep-cr "$HERE"/patches/*.patch diff --git a/base.txt b/base.txt new file mode 100644 index 0000000..acb4095 --- /dev/null +++ b/base.txt @@ -0,0 +1 @@ +v1.29 diff --git a/patches/0001-notify-core-file-access-notification-infrastructure.patch b/patches/0001-notify-core-file-access-notification-infrastructure.patch new file mode 100644 index 0000000..df0d5cb --- /dev/null +++ b/patches/0001-notify-core-file-access-notification-infrastructure.patch @@ -0,0 +1,838 @@ +From b57f3ec7e3c3d971c8f4b6825b7d8837f6668efb Mon Sep 17 00:00:00 2001 +From: William Gill +Date: Wed, 22 Apr 2026 10:17:24 -0500 +Subject: [PATCH 1/2] notify: core file-access notification infrastructure + +Adds an optional, observer-only notification stream to the scoutfs +kernel module. File open/read events are recorded in a per-mount +ring buffer and drained by a single privileged userspace reader +through a new ioctl. + +Behavior: + +- Disabled by default. A new mount option, notify_events=0|1, + turns the stream on; notify_ring_kb=N sizes the ring (4..4096 KiB, + default 64 KiB). +- New ioctl SCOUTFS_IOC_READ_NOTIFY (nr 25) fills the caller's + array of scoutfs_ioctl_notify_event records; supports a + timeout_ms wait policy. Requires CAP_SYS_ADMIN. +- Ring is lossy: on overflow records are dropped but the monotonic + seq field still advances, so a reader detects gaps. Three + percpu counters (notify_emitted, notify_dropped_ring_full, + notify_reader_attached) surface operational state via sysfs. +- Single-reader: concurrent readers receive -EBUSY. + +This patch only adds the infrastructure (new files, mount options, +ioctl, lifecycle, counters). No scoutfs code path calls +scoutfs_notify_emit() yet, so behavior is bit-identical to stock +with notify_events=0 or notify_events=1. + +Event type values 1..2 are defined (OPEN, READ); 3+ are reserved +for future data-waiter observation events. +--- + kmod/src/Makefile | 1 + + kmod/src/counters.h | 3 + + kmod/src/ioctl.c | 3 + + kmod/src/ioctl.h | 78 +++++++++ + kmod/src/notify.c | 389 ++++++++++++++++++++++++++++++++++++++++++++ + kmod/src/notify.h | 39 +++++ + kmod/src/options.c | 91 +++++++++++ + kmod/src/options.h | 2 + + kmod/src/super.c | 3 + + kmod/src/super.h | 5 + + 10 files changed, 614 insertions(+) + create mode 100644 kmod/src/notify.c + create mode 100644 kmod/src/notify.h + +diff --git a/kmod/src/Makefile b/kmod/src/Makefile +index fa632aa..de14c74 100644 +--- a/kmod/src/Makefile ++++ b/kmod/src/Makefile +@@ -31,6 +31,7 @@ scoutfs-y += \ + lock_server.o \ + msg.o \ + net.o \ ++ notify.o \ + omap.o \ + options.o \ + per_task.o \ +diff --git a/kmod/src/counters.h b/kmod/src/counters.h +index 9088496..94e4f44 100644 +--- a/kmod/src/counters.h ++++ b/kmod/src/counters.h +@@ -156,6 +156,9 @@ + EXPAND_COUNTER(net_recv_invalid_message) \ + EXPAND_COUNTER(net_recv_messages) \ + EXPAND_COUNTER(net_unknown_request) \ ++ EXPAND_COUNTER(notify_dropped_ring_full) \ ++ EXPAND_COUNTER(notify_emitted) \ ++ EXPAND_COUNTER(notify_reader_attached) \ + EXPAND_COUNTER(orphan_scan) \ + EXPAND_COUNTER(orphan_scan_attempts) \ + EXPAND_COUNTER(orphan_scan_cached) \ +diff --git a/kmod/src/ioctl.c b/kmod/src/ioctl.c +index 0a5fc4c..b3a4d9a 100644 +--- a/kmod/src/ioctl.c ++++ b/kmod/src/ioctl.c +@@ -47,6 +47,7 @@ + #include "totl.h" + #include "wkic.h" + #include "quota.h" ++#include "notify.h" + #include "scoutfs_trace.h" + #include "util.h" + +@@ -1790,6 +1791,8 @@ long scoutfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg) + return scoutfs_ioc_read_xattr_index(file, arg); + case SCOUTFS_IOC_PUNCH_OFFLINE: + return scoutfs_ioc_punch_offline(file, arg); ++ case SCOUTFS_IOC_READ_NOTIFY: ++ return scoutfs_ioc_read_notify(file, arg); + } + + return -ENOTTY; +diff --git a/kmod/src/ioctl.h b/kmod/src/ioctl.h +index c0d2285..fc58f8c 100644 +--- a/kmod/src/ioctl.h ++++ b/kmod/src/ioctl.h +@@ -876,4 +876,82 @@ struct scoutfs_ioctl_punch_offline { + #define SCOUTFS_IOC_PUNCH_OFFLINE \ + _IOW(SCOUTFS_IOCTL_MAGIC, 24, struct scoutfs_ioctl_punch_offline) + ++/* ++ * File access notification stream (observer-only). ++ * ++ * When the notify_events mount option is enabled, the kernel records ++ * file open and read events in a per-mount ring buffer. A single ++ * privileged reader drains the ring via this ioctl and forwards events ++ * to a userspace watcher (typically via a Unix socket). ++ * ++ * The notification stream does not alter scoutfs behavior. Events are ++ * best-effort: if the ring fills, records are dropped but the monotonic ++ * @seq field still advances, so a reader can detect gaps. ++ * ++ * Reader exclusivity: only one caller may be blocked in this ioctl at a ++ * time. A second concurrent caller receives -EBUSY. ++ * ++ * Required capability: CAP_SYS_ADMIN. ++ * ++ * Event record (64 bytes, wire-stable): ++ * ++ * @seq: monotonic sequence number assigned at emit time. Gaps ++ * indicate ring-full drops between adjacent records. ++ * @ino: scoutfs inode number of the file. ++ * @offset: byte offset for READ events; 0 for OPEN. ++ * @length: byte length for READ events; 0 for OPEN. ++ * @time_ns: CLOCK_REALTIME time of the event in nanoseconds. ++ * @pid: thread group id of the task that caused the event. ++ * @uid: effective uid of the task that caused the event, as ++ * observed from the init user namespace. ++ * @type: one of the SCOUTFS_NOTIFY_TYPE_* values. ++ * @flags: per-event flag bits. ++ */ ++#define SCOUTFS_NOTIFY_TYPE_OPEN 1 ++#define SCOUTFS_NOTIFY_TYPE_READ 2 ++/* type values 3+ reserved for future data-waiter events */ ++ ++#define SCOUTFS_NOTIFY_F_WRITE_OPEN (1 << 0) ++ ++struct scoutfs_ioctl_notify_event { ++ __u64 seq; ++ __u64 ino; ++ __u64 offset; ++ __u64 length; ++ __u64 time_ns; ++ __u32 pid; ++ __u32 uid; ++ __u8 type; ++ __u8 flags; ++ __u8 _pad[6]; ++}; ++ ++/* ++ * Request for the notification read ioctl. ++ * ++ * @events_ptr: user-space address of an array of ++ * struct scoutfs_ioctl_notify_event. Must be 8-byte ++ * aligned. ++ * @events_nr: number of entries the array can hold. ++ * @timeout_ms: wait policy when the ring is empty. 0 returns ++ * immediately with 0 events (non-blocking), U32_MAX waits ++ * without a timeout, any other value waits up to that ++ * many milliseconds. ++ * @flags: reserved, must be 0. ++ * ++ * On success the number of events written to the user array is ++ * returned (may be 0 for non-blocking empty). -ETIMEDOUT indicates ++ * the timeout elapsed with no events available. -EBUSY indicates ++ * another reader is currently attached. ++ */ ++struct scoutfs_ioctl_read_notify { ++ __u64 events_ptr; ++ __u32 events_nr; ++ __u32 timeout_ms; ++ __u64 flags; ++}; ++ ++#define SCOUTFS_IOC_READ_NOTIFY \ ++ _IOWR(SCOUTFS_IOCTL_MAGIC, 25, struct scoutfs_ioctl_read_notify) ++ + #endif +diff --git a/kmod/src/notify.c b/kmod/src/notify.c +new file mode 100644 +index 0000000..94fab17 +--- /dev/null ++++ b/kmod/src/notify.c +@@ -0,0 +1,389 @@ ++/* ++ * Copyright (C) 2026 Versity Software, Inc. All rights reserved. ++ * ++ * This program is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU General Public ++ * License v2 as published by the Free Software Foundation. ++ * ++ * This program is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * General Public License for more details. ++ */ ++ ++/* ++ * Observer-only file access notifications. ++ * ++ * scoutfs_notify_emit() is called from file-op fast paths and must ++ * never block scoutfs, take sleeping locks, allocate memory, or ++ * propagate errors. It acquires one leaf spinlock, stamps the next ++ * sequence number, copies (or drops) one fixed-size record, wakes the ++ * reader, and returns. The sequence counter advances on drops so the ++ * reader sees a gap rather than a seamless stream. ++ * ++ * Userspace drains the ring via SCOUTFS_IOC_READ_NOTIFY. Only one ++ * reader may be attached at a time. ++ */ ++ ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++#include ++ ++#include "super.h" ++#include "options.h" ++#include "counters.h" ++#include "ioctl.h" ++#include "notify.h" ++#include "msg.h" ++ ++/* ++ * Power-of-two ring of fixed-size records. head advances on write, ++ * tail advances on read, both unsigned indices masked to the ring ++ * capacity. Full when (head - tail) == capacity. ++ */ ++struct notify_info { ++ spinlock_t lock; ++ wait_queue_head_t reader_wq; ++ atomic_t reader_attached; ++ ++ /* all fields below this point are protected by ->lock */ ++ struct scoutfs_ioctl_notify_event *ring; ++ u32 capacity; /* power of two */ ++ u32 mask; ++ u32 head; ++ u32 tail; ++ u64 next_seq; ++}; ++ ++#define NOTIFY_MIN_KB 4 ++#define NOTIFY_MAX_KB 4096 ++ ++static u32 ring_len_locked(struct notify_info *ni) ++{ ++ return ni->head - ni->tail; ++} ++ ++static bool ring_empty_locked(struct notify_info *ni) ++{ ++ return ni->head == ni->tail; ++} ++ ++static bool ring_full_locked(struct notify_info *ni) ++{ ++ return ring_len_locked(ni) == ni->capacity; ++} ++ ++/* ++ * Allocate the ring buffer. Uses kvmalloc to accept large user-chosen ++ * sizes without failing under fragmentation. ++ */ ++static int alloc_ring(struct notify_info *ni, u32 capacity) ++{ ++ size_t bytes; ++ ++ if (!is_power_of_2(capacity)) ++ return -EINVAL; ++ ++ if (check_mul_overflow((size_t)capacity, ++ sizeof(struct scoutfs_ioctl_notify_event), ++ &bytes)) ++ return -EINVAL; ++ ++ ni->ring = kvzalloc(bytes, GFP_KERNEL); ++ if (!ni->ring) ++ return -ENOMEM; ++ ++ ni->capacity = capacity; ++ ni->mask = capacity - 1; ++ ni->head = 0; ++ ni->tail = 0; ++ return 0; ++} ++ ++/* ++ * Compute ring capacity in records from the options-configured size in ++ * KiB. Rounds down to a power of two so the mask-based index math is ++ * valid. Returns a record count, not a byte count. ++ */ ++static u32 capacity_from_kb(u32 kb) ++{ ++ u64 bytes; ++ u32 records; ++ ++ if (kb < NOTIFY_MIN_KB) ++ kb = NOTIFY_MIN_KB; ++ if (kb > NOTIFY_MAX_KB) ++ kb = NOTIFY_MAX_KB; ++ ++ bytes = (u64)kb << 10; ++ records = (u32)(bytes / sizeof(struct scoutfs_ioctl_notify_event)); ++ if (records < 2) ++ records = 2; ++ ++ /* round down to power of two */ ++ return 1U << (fls(records) - 1); ++} ++ ++int scoutfs_notify_setup(struct super_block *sb) ++{ ++ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb); ++ struct scoutfs_mount_options opts; ++ struct notify_info *ni; ++ u32 capacity; ++ int ret; ++ ++ ni = kzalloc(sizeof(*ni), GFP_KERNEL); ++ if (!ni) ++ return -ENOMEM; ++ ++ spin_lock_init(&ni->lock); ++ init_waitqueue_head(&ni->reader_wq); ++ atomic_set(&ni->reader_attached, 0); ++ ni->next_seq = 1; ++ ++ scoutfs_options_read(sb, &opts); ++ capacity = capacity_from_kb(opts.notify_ring_kb); ++ ++ ret = alloc_ring(ni, capacity); ++ if (ret < 0) { ++ kfree(ni); ++ return ret; ++ } ++ ++ sbi->notify_info = ni; ++ ++ /* ++ * Publish the enable flag last so emit callers observe a fully ++ * initialized ring. ++ */ ++ smp_wmb(); ++ WRITE_ONCE(sbi->notify_enabled, opts.notify_events ? true : false); ++ ++ return 0; ++} ++ ++void scoutfs_notify_destroy(struct super_block *sb) ++{ ++ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb); ++ struct notify_info *ni = sbi->notify_info; ++ ++ if (!ni) ++ return; ++ ++ /* ++ * Silence the emit fast path first, then wake any reader so it ++ * can exit its wait loop and stop touching the ring. ++ */ ++ WRITE_ONCE(sbi->notify_enabled, false); ++ wake_up_all(&ni->reader_wq); ++ ++ sbi->notify_info = NULL; ++ ++ kvfree(ni->ring); ++ kfree(ni); ++} ++ ++/* ++ * Best-effort observer-only event emit. ++ * ++ * - Callers must gate this on sbi->notify_enabled; we re-check under ++ * the spinlock so shutdown is race-free. ++ * - On ring-full we drop the record but still consume a sequence ++ * number so the reader sees exactly the number of drops. ++ * - Never returns an error and never blocks scoutfs. ++ */ ++void scoutfs_notify_emit(struct super_block *sb, __u8 type, __u64 ino, ++ __u64 offset, __u64 length, __u8 flags) ++{ ++ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb); ++ struct notify_info *ni = sbi->notify_info; ++ struct scoutfs_ioctl_notify_event *slot; ++ unsigned long irqflags; ++ u64 time_ns; ++ u32 pid; ++ u32 uid; ++ bool dropped = false; ++ bool woke = false; ++ ++ if (!ni) ++ return; ++ ++ time_ns = ktime_get_real_ns(); ++ pid = (u32)task_tgid_nr(current); ++ uid = from_kuid(&init_user_ns, current_uid()); ++ ++ spin_lock_irqsave(&ni->lock, irqflags); ++ ++ if (ring_full_locked(ni)) { ++ dropped = true; ++ ni->next_seq++; ++ } else { ++ slot = &ni->ring[ni->head & ni->mask]; ++ slot->seq = ni->next_seq++; ++ slot->ino = ino; ++ slot->offset = offset; ++ slot->length = length; ++ slot->time_ns = time_ns; ++ slot->pid = pid; ++ slot->uid = uid; ++ slot->type = type; ++ slot->flags = flags; ++ memset(slot->_pad, 0, sizeof(slot->_pad)); ++ ni->head++; ++ woke = (ring_len_locked(ni) == 1); ++ } ++ ++ spin_unlock_irqrestore(&ni->lock, irqflags); ++ ++ if (dropped) ++ scoutfs_inc_counter(sb, notify_dropped_ring_full); ++ else ++ scoutfs_inc_counter(sb, notify_emitted); ++ ++ if (woke) ++ wake_up_interruptible(&ni->reader_wq); ++} ++ ++static bool reader_should_wake(struct notify_info *ni, ++ struct scoutfs_sb_info *sbi) ++{ ++ unsigned long irqflags; ++ bool empty; ++ ++ spin_lock_irqsave(&ni->lock, irqflags); ++ empty = ring_empty_locked(ni); ++ spin_unlock_irqrestore(&ni->lock, irqflags); ++ ++ return !empty || !READ_ONCE(sbi->notify_enabled); ++} ++ ++/* ++ * Copy events out of the ring under the spinlock into a small ++ * on-stack batch, release the lock, then copy_to_user. This keeps ++ * the emit path's lock hold time bounded by batch size independent ++ * of user buffer size. ++ */ ++#define NOTIFY_DRAIN_BATCH 16 ++ ++static int drain_ring(struct notify_info *ni, ++ struct scoutfs_ioctl_notify_event __user *udst, ++ u32 max_events) ++{ ++ struct scoutfs_ioctl_notify_event batch[NOTIFY_DRAIN_BATCH]; ++ unsigned long irqflags; ++ u32 total = 0; ++ u32 i, n; ++ ++ while (total < max_events) { ++ n = min_t(u32, max_events - total, NOTIFY_DRAIN_BATCH); ++ ++ spin_lock_irqsave(&ni->lock, irqflags); ++ n = min(n, ring_len_locked(ni)); ++ for (i = 0; i < n; i++) { ++ batch[i] = ni->ring[ni->tail & ni->mask]; ++ ni->tail++; ++ } ++ spin_unlock_irqrestore(&ni->lock, irqflags); ++ ++ if (n == 0) ++ break; ++ ++ if (copy_to_user(udst + total, batch, ++ n * sizeof(batch[0]))) ++ return total ? (int)total : -EFAULT; ++ ++ total += n; ++ ++ if (n < NOTIFY_DRAIN_BATCH) ++ break; ++ } ++ ++ return (int)total; ++} ++ ++long scoutfs_ioc_read_notify(struct file *file, unsigned long arg) ++{ ++ struct super_block *sb = file_inode(file)->i_sb; ++ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb); ++ struct notify_info *ni = sbi->notify_info; ++ struct scoutfs_ioctl_read_notify __user *uargs = (void __user *)arg; ++ struct scoutfs_ioctl_read_notify args; ++ struct scoutfs_ioctl_notify_event __user *uevents; ++ unsigned long irqflags; ++ long jiffies_timeout; ++ bool attached = false; ++ bool empty; ++ int ret; ++ ++ if (!capable(CAP_SYS_ADMIN)) ++ return -EPERM; ++ ++ if (!ni) ++ return -ENODEV; ++ ++ if (copy_from_user(&args, uargs, sizeof(args))) ++ return -EFAULT; ++ ++ if (args.flags != 0) ++ return -EINVAL; ++ if (args.events_nr == 0) ++ return 0; ++ if (args.events_ptr & 0x7) ++ return -EINVAL; ++ ++ uevents = (struct scoutfs_ioctl_notify_event __user *) ++ (unsigned long)args.events_ptr; ++ ++ if (atomic_cmpxchg(&ni->reader_attached, 0, 1) != 0) ++ return -EBUSY; ++ attached = true; ++ scoutfs_inc_counter(sb, notify_reader_attached); ++ ++ spin_lock_irqsave(&ni->lock, irqflags); ++ empty = ring_empty_locked(ni); ++ spin_unlock_irqrestore(&ni->lock, irqflags); ++ ++ if (empty) { ++ if (args.timeout_ms == 0) { ++ ret = 0; ++ goto out; ++ } ++ ++ if (args.timeout_ms == U32_MAX) { ++ ret = wait_event_interruptible(ni->reader_wq, ++ reader_should_wake(ni, sbi)); ++ if (ret < 0) ++ goto out; ++ } else { ++ jiffies_timeout = msecs_to_jiffies(args.timeout_ms); ++ ret = wait_event_interruptible_timeout(ni->reader_wq, ++ reader_should_wake(ni, sbi), ++ jiffies_timeout); ++ if (ret == 0) { ++ ret = -ETIMEDOUT; ++ goto out; ++ } ++ if (ret < 0) ++ goto out; ++ } ++ } ++ ++ ret = drain_ring(ni, uevents, args.events_nr); ++ ++out: ++ if (attached) ++ atomic_set(&ni->reader_attached, 0); ++ return ret; ++} +diff --git a/kmod/src/notify.h b/kmod/src/notify.h +new file mode 100644 +index 0000000..fe6a2d9 +--- /dev/null ++++ b/kmod/src/notify.h +@@ -0,0 +1,39 @@ ++#ifndef _SCOUTFS_NOTIFY_H_ ++#define _SCOUTFS_NOTIFY_H_ ++ ++/* ++ * File access notification subsystem. ++ * ++ * Emits best-effort, observer-only events (file open, read) to a ++ * per-super in-kernel ring. A single privileged userspace reader ++ * drains the ring via SCOUTFS_IOC_READ_NOTIFY. Events are dropped on ++ * ring overflow and the sequence number advances regardless so the ++ * reader can detect gaps. ++ * ++ * The emit path is non-blocking and takes only a leaf spinlock. It ++ * does not allocate, does not sleep, and ignores any downstream state. ++ * Callers must never check its outcome. ++ * ++ * Event-type values (SCOUTFS_NOTIFY_TYPE_*) and per-event flag bits ++ * (SCOUTFS_NOTIFY_F_*) are the wire-stable constants defined in ++ * ioctl.h; they are shared between kernel emit paths and user-visible ++ * records. ++ */ ++ ++#include ++ ++#include "ioctl.h" ++ ++struct super_block; ++struct file; ++struct notify_info; ++ ++int scoutfs_notify_setup(struct super_block *sb); ++void scoutfs_notify_destroy(struct super_block *sb); ++ ++void scoutfs_notify_emit(struct super_block *sb, __u8 type, __u64 ino, ++ __u64 offset, __u64 length, __u8 flags); ++ ++long scoutfs_ioc_read_notify(struct file *file, unsigned long arg); ++ ++#endif /* _SCOUTFS_NOTIFY_H_ */ +diff --git a/kmod/src/options.c b/kmod/src/options.c +index b7565d7..90e7713 100644 +--- a/kmod/src/options.c ++++ b/kmod/src/options.c +@@ -38,6 +38,8 @@ enum { + Opt_log_merge_wait_timeout_ms, + Opt_metadev_path, + Opt_noacl, ++ Opt_notify_events, ++ Opt_notify_ring_kb, + Opt_orphan_scan_delay_ms, + Opt_quorum_heartbeat_timeout_ms, + Opt_quorum_slot_nr, +@@ -54,6 +56,8 @@ static const match_table_t tokens = { + {Opt_log_merge_wait_timeout_ms, "log_merge_wait_timeout_ms=%s"}, + {Opt_metadev_path, "metadev_path=%s"}, + {Opt_noacl, "noacl"}, ++ {Opt_notify_events, "notify_events=%s"}, ++ {Opt_notify_ring_kb, "notify_ring_kb=%s"}, + {Opt_orphan_scan_delay_ms, "orphan_scan_delay_ms=%s"}, + {Opt_quorum_heartbeat_timeout_ms, "quorum_heartbeat_timeout_ms=%s"}, + {Opt_quorum_slot_nr, "quorum_slot_nr=%s"}, +@@ -138,6 +142,10 @@ static void free_options(struct scoutfs_mount_options *opts) + + #define DEFAULT_TCP_KEEPALIVE_TIMEOUT_MS (60 * MSEC_PER_SEC) + ++#define MIN_NOTIFY_RING_KB 4U ++#define DEFAULT_NOTIFY_RING_KB 64U ++#define MAX_NOTIFY_RING_KB 4096U ++ + static void init_default_options(struct scoutfs_mount_options *opts) + { + memset(opts, 0, sizeof(*opts)); +@@ -147,6 +155,8 @@ static void init_default_options(struct scoutfs_mount_options *opts) + opts->ino_alloc_per_lock = SCOUTFS_LOCK_INODE_GROUP_NR; + opts->lock_idle_count = DEFAULT_LOCK_IDLE_COUNT; + opts->log_merge_wait_timeout_ms = DEFAULT_LOG_MERGE_WAIT_TIMEOUT_MS; ++ opts->notify_events = false; ++ opts->notify_ring_kb = DEFAULT_NOTIFY_RING_KB; + opts->orphan_scan_delay_ms = -1; + opts->quorum_heartbeat_timeout_ms = SCOUTFS_QUORUM_DEF_HB_TIMEO_MS; + opts->quorum_slot_nr = -1; +@@ -309,6 +319,29 @@ static int parse_options(struct super_block *sb, char *options, struct scoutfs_m + sb->s_flags &= ~SB_POSIXACL; + break; + ++ case Opt_notify_events: ++ ret = match_int(args, &nr); ++ if (ret < 0 || nr < 0 || nr > 1) { ++ scoutfs_err(sb, "invalid notify_events option, bool must only be 0 or 1"); ++ if (ret == 0) ++ ret = -EINVAL; ++ return ret; ++ } ++ opts->notify_events = nr; ++ break; ++ ++ case Opt_notify_ring_kb: ++ ret = match_int(args, &nr); ++ if (ret < 0 || nr < MIN_NOTIFY_RING_KB || nr > MAX_NOTIFY_RING_KB) { ++ scoutfs_err(sb, "invalid notify_ring_kb option, must be between %u and %u", ++ MIN_NOTIFY_RING_KB, MAX_NOTIFY_RING_KB); ++ if (ret == 0) ++ ret = -EINVAL; ++ return ret; ++ } ++ opts->notify_ring_kb = nr; ++ break; ++ + case Opt_orphan_scan_delay_ms: + if (opts->orphan_scan_delay_ms != -1) { + scoutfs_err(sb, "multiple orphan_scan_delay_ms options provided, only provide one."); +@@ -442,6 +475,8 @@ int scoutfs_options_show(struct seq_file *seq, struct dentry *root) + seq_printf(seq, ",metadev_path=%s", opts.metadev_path); + if (!is_acl) + seq_puts(seq, ",noacl"); ++ seq_printf(seq, ",notify_events=%u", opts.notify_events); ++ seq_printf(seq, ",notify_ring_kb=%u", opts.notify_ring_kb); + seq_printf(seq, ",orphan_scan_delay_ms=%u", opts.orphan_scan_delay_ms); + if (opts.quorum_slot_nr >= 0) + seq_printf(seq, ",quorum_slot_nr=%d", opts.quorum_slot_nr); +@@ -651,6 +686,60 @@ static ssize_t metadev_path_show(struct kobject *kobj, struct kobj_attribute *at + } + SCOUTFS_ATTR_RO(metadev_path); + ++static ssize_t notify_events_show(struct kobject *kobj, struct kobj_attribute *attr, ++ char *buf) ++{ ++ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj); ++ struct scoutfs_mount_options opts; ++ ++ scoutfs_options_read(sb, &opts); ++ ++ return snprintf(buf, PAGE_SIZE, "%u", opts.notify_events); ++} ++static ssize_t notify_events_store(struct kobject *kobj, struct kobj_attribute *attr, ++ const char *buf, size_t count) ++{ ++ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj); ++ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb); ++ DECLARE_OPTIONS_INFO(sb, optinf); ++ char nullterm[20]; ++ long val; ++ int len; ++ int ret; ++ ++ len = min(count, sizeof(nullterm) - 1); ++ memcpy(nullterm, buf, len); ++ nullterm[len] = '\0'; ++ ++ ret = kstrtol(nullterm, 0, &val); ++ if (ret < 0 || val < 0 || val > 1) { ++ scoutfs_err(sb, "invalid notify_events option, bool must be 0 or 1"); ++ return -EINVAL; ++ } ++ ++ write_seqlock(&optinf->seqlock); ++ optinf->opts.notify_events = val; ++ write_sequnlock(&optinf->seqlock); ++ ++ /* mirror into sbi so emit fast path can check with one load */ ++ WRITE_ONCE(sbi->notify_enabled, val ? true : false); ++ ++ return count; ++} ++SCOUTFS_ATTR_RW(notify_events); ++ ++static ssize_t notify_ring_kb_show(struct kobject *kobj, struct kobj_attribute *attr, ++ char *buf) ++{ ++ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj); ++ struct scoutfs_mount_options opts; ++ ++ scoutfs_options_read(sb, &opts); ++ ++ return snprintf(buf, PAGE_SIZE, "%u", opts.notify_ring_kb); ++} ++SCOUTFS_ATTR_RO(notify_ring_kb); ++ + static ssize_t orphan_scan_delay_ms_show(struct kobject *kobj, struct kobj_attribute *attr, + char *buf) + { +@@ -747,6 +836,8 @@ static struct attribute *options_attrs[] = { + SCOUTFS_ATTR_PTR(lock_idle_count), + SCOUTFS_ATTR_PTR(log_merge_wait_timeout_ms), + SCOUTFS_ATTR_PTR(metadev_path), ++ SCOUTFS_ATTR_PTR(notify_events), ++ SCOUTFS_ATTR_PTR(notify_ring_kb), + SCOUTFS_ATTR_PTR(orphan_scan_delay_ms), + SCOUTFS_ATTR_PTR(quorum_heartbeat_timeout_ms), + SCOUTFS_ATTR_PTR(quorum_slot_nr), +diff --git a/kmod/src/options.h b/kmod/src/options.h +index b37bbd7..12b976e 100644 +--- a/kmod/src/options.h ++++ b/kmod/src/options.h +@@ -12,6 +12,8 @@ struct scoutfs_mount_options { + int lock_idle_count; + unsigned int log_merge_wait_timeout_ms; + char *metadev_path; ++ bool notify_events; ++ unsigned int notify_ring_kb; + unsigned int orphan_scan_delay_ms; + int quorum_slot_nr; + u64 quorum_heartbeat_timeout_ms; +diff --git a/kmod/src/super.c b/kmod/src/super.c +index 3c83716..3c028b0 100644 +--- a/kmod/src/super.c ++++ b/kmod/src/super.c +@@ -51,6 +51,7 @@ + #include "xattr.h" + #include "wkic.h" + #include "quota.h" ++#include "notify.h" + #include "scoutfs_trace.h" + + static struct dentry *scoutfs_debugfs_root; +@@ -223,6 +224,7 @@ static void scoutfs_put_super(struct super_block *sb) + scoutfs_block_destroy(sb); + scoutfs_destroy_triggers(sb); + scoutfs_fence_destroy(sb); ++ scoutfs_notify_destroy(sb); + scoutfs_options_destroy(sb); + debugfs_remove(sbi->debug_root); + scoutfs_destroy_counters(sb); +@@ -579,6 +581,7 @@ static int scoutfs_fill_super(struct super_block *sb, void *data, int silent) + scoutfs_setup_sysfs(sb) ?: + scoutfs_setup_counters(sb) ?: + scoutfs_options_setup(sb) ?: ++ scoutfs_notify_setup(sb) ?: + scoutfs_setup_triggers(sb) ?: + scoutfs_fence_setup(sb) ?: + scoutfs_block_setup(sb) ?: +diff --git a/kmod/src/super.h b/kmod/src/super.h +index 3bb10dd..d6353c9 100644 +--- a/kmod/src/super.h ++++ b/kmod/src/super.h +@@ -32,6 +32,7 @@ struct volopt_info; + struct fence_info; + struct wkic_info; + struct squota_info; ++struct notify_info; + + struct scoutfs_sb_info { + struct super_block *sb; +@@ -81,6 +82,10 @@ struct scoutfs_sb_info { + struct scoutfs_counters *counters; + struct scoutfs_triggers *triggers; + ++ /* file access notifications (optional, gated by notify_events opt) */ ++ struct notify_info *notify_info; ++ bool notify_enabled; ++ + struct dentry *debug_root; + + bool forced_unmount; +-- +2.49.0.windows.1 + diff --git a/patches/0002-notify-file-open-read-hook-sites.patch b/patches/0002-notify-file-open-read-hook-sites.patch new file mode 100644 index 0000000..dc578c9 --- /dev/null +++ b/patches/0002-notify-file-open-read-hook-sites.patch @@ -0,0 +1,137 @@ +From dcf118e874fa4af97144c8e0c960ac42c7e17245 Mon Sep 17 00:00:00 2001 +From: William Gill +Date: Wed, 22 Apr 2026 10:18:11 -0500 +Subject: [PATCH 2/2] notify: file open/read hook sites + +Wires the notification emit path into the two scoutfs file +operations that carry user-visible activity we want to observe. + +OPEN (data.c): + - A new scoutfs_file_open() wrapper is installed as ->open in + scoutfs_file_fops. The wrapper calls generic_file_open() to + preserve the existing VFS default semantics for regular files, + then, only on success and only when notifications are enabled, + emits a SCOUTFS_NOTIFY_TYPE_OPEN record. The file mode is + inspected for FMODE_WRITE to set SCOUTFS_NOTIFY_F_WRITE_OPEN. + +READ (file.c): + - Both the aio_read (KC_LINUX_HAVE_FOP_AIO_READ) and read_iter + paths get an emit placed past the existing data-waiter retry + check, guarded on (ret > 0) so only successful reads are + reported and retries never double-count. start_pos is captured + at entry of read_iter before generic_file_read_iter advances + iocb->ki_pos. + +Every hook is behind unlikely(READ_ONCE(sbi->notify_enabled)), so +with the default notify_events=0 the path is a single +predicted-false branch. No scoutfs state is mutated, no error is +propagated, and no existing control flow is altered. + +Nothing in the data-waiter state machine (scoutfs_data_wait_check, +scoutfs_data_wait, scoutfs_data_wait_changed, the waiter rbtree, or +SCOUTFS_IOC_DATA_WAITING / DATA_WAIT_ERR) is touched by this patch. +That work is deferred to a future series. +--- + kmod/src/data.c | 28 ++++++++++++++++++++++++++++ + kmod/src/file.c | 11 +++++++++++ + 2 files changed, 39 insertions(+) + +diff --git a/kmod/src/data.c b/kmod/src/data.c +index 0abb48c..33a61af 100644 +--- a/kmod/src/data.c ++++ b/kmod/src/data.c +@@ -41,6 +41,7 @@ + #include "msg.h" + #include "ext.h" + #include "util.h" ++#include "notify.h" + + /* + * We want to amortize work done after dirtying the shared transaction +@@ -2304,6 +2305,32 @@ const struct address_space_operations scoutfs_file_aops = { + .write_end = scoutfs_write_end, + }; + ++/* ++ * Thin observer wrapper around the VFS default open for regular files. ++ * ++ * generic_file_open() is exactly what the VFS uses when a file_operations ++ * table leaves ->open NULL, so installing this wrapper does not change ++ * open semantics. When notifications are enabled we emit an OPEN event ++ * after the generic open has succeeded; a failed open emits nothing. ++ */ ++static int scoutfs_file_open(struct inode *inode, struct file *file) ++{ ++ struct super_block *sb = inode->i_sb; ++ int ret; ++ ++ ret = generic_file_open(inode, file); ++ if (ret == 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled))) { ++ u8 flags = 0; ++ ++ if (file->f_mode & FMODE_WRITE) ++ flags |= SCOUTFS_NOTIFY_F_WRITE_OPEN; ++ ++ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_OPEN, ++ scoutfs_ino(inode), 0, 0, flags); ++ } ++ return ret; ++} ++ + const struct file_operations scoutfs_file_fops = { + #ifdef KC_LINUX_HAVE_FOP_AIO_READ + .read = do_sync_read, +@@ -2316,6 +2343,7 @@ const struct file_operations scoutfs_file_fops = { + .splice_read = generic_file_splice_read, + .splice_write = iter_file_splice_write, + #endif ++ .open = scoutfs_file_open, + .mmap = scoutfs_file_mmap, + .unlocked_ioctl = scoutfs_ioctl, + .fsync = scoutfs_file_fsync, +diff --git a/kmod/src/file.c b/kmod/src/file.c +index 15158a2..44f1bc4 100644 +--- a/kmod/src/file.c ++++ b/kmod/src/file.c +@@ -29,6 +29,7 @@ + #include "per_task.h" + #include "omap.h" + #include "quota.h" ++#include "notify.h" + + #ifdef KC_LINUX_HAVE_FOP_AIO_READ + /* +@@ -84,6 +85,10 @@ out: + goto retry; + } + ++ if (ret > 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled))) ++ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_READ, ++ scoutfs_ino(inode), pos, ret, 0); ++ + return ret; + } + +@@ -166,6 +171,7 @@ ssize_t scoutfs_file_read_iter(struct kiocb *iocb, struct iov_iter *to) + struct scoutfs_lock *scoutfs_inode_lock = NULL; + SCOUTFS_DECLARE_PER_TASK_ENTRY(pt_ent); + DECLARE_DATA_WAIT(dw); ++ loff_t start_pos = iocb->ki_pos; + int ret; + + retry: +@@ -200,6 +206,11 @@ out: + if (ret == 0) + goto retry; + } ++ ++ if (ret > 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled))) ++ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_READ, ++ scoutfs_ino(inode), start_pos, ret, 0); ++ + return ret; + } + +-- +2.49.0.windows.1 +