Initial scoutfs-notify patch series for v1.29

Two-patch series that layers an observer-only file-access
notification stream onto scoutfs.  See README.md for the design
summary and rebase workflow.

Base: scoutfs v1.29
This commit is contained in:
William Gill
2026-04-22 10:21:21 -05:00
commit 50b9e27f75
6 changed files with 1077 additions and 0 deletions
+3
View File
@@ -0,0 +1,3 @@
* text=auto eol=lf
*.patch binary
apply.sh text eol=lf
+72
View File
@@ -0,0 +1,72 @@
# scoutfs-notify
Observer-only file access notifications for [ScoutFS](https://github.com/versity/scoutfs).
Maintained as a rebasable `git format-patch` series so it can be layered onto
each upstream release with minimal maintenance.
## What it adds
* New mount option `notify_events=0|1` (default `0`) that turns on a per-mount
ring buffer of file **open** and **read** events.
* New mount option `notify_ring_kb=N` (4..4096, default 64) sizing that ring.
* New ioctl `SCOUTFS_IOC_READ_NOTIFY` (nr 25) that lets a single privileged
userspace reader drain the ring. Reader is expected to relay events over a
Unix socket to a watcher daemon.
* Three percpu counters: `notify_emitted`, `notify_dropped_ring_full`,
`notify_reader_attached`.
The notification path is strictly observer-only: if the ring fills, records
are dropped but the monotonic `seq` field still advances so readers see a
gap. `notify_events=0` reduces the hook to a single predicted-false branch.
Nothing in the data-waiter state machine is touched by this series; a future
patch series may add data-waiter observation events (reserved type values 3+).
## Base
Currently rebased against:
scoutfs v1.29
See [base.txt](./base.txt) for the exact tag the patches target.
## Applying
On a git checkout of scoutfs at the base tag:
```sh
./apply.sh /path/to/scoutfs
```
The script runs `git am --3way` against each `patches/*.patch`. Requires the
target to be a git working tree.
For a non-git tarball, apply with:
```sh
cd /path/to/scoutfs
for p in /path/to/scoutfs-notify/patches/*.patch; do
patch -p1 < "$p"
done
```
## Rebasing onto a new upstream release
```sh
# In your scoutfs working clone
git fetch --tags
git checkout -B notify v1.30 # new upstream tag
git am --3way patches/*.patch # from this repo
# ... resolve any conflicts, git am --continue ...
git format-patch v1.30..notify -o patches/
# Update base.txt and commit the refreshed patches
```
## Tagging
Tag releases of the patch set with both the upstream version and a patch
revision so consumers can pin precisely:
v1.29-notify-1
v1.29-notify-2
v1.30-notify-1
+26
View File
@@ -0,0 +1,26 @@
#!/usr/bin/env bash
#
# Apply the scoutfs-notify patch series to a scoutfs working tree.
#
# Usage: apply.sh /path/to/scoutfs
#
set -euo pipefail
HERE=$(cd "$(dirname "$0")" && pwd)
TREE="${1:?usage: $0 /path/to/scoutfs}"
if [ ! -d "$TREE/.git" ]; then
echo "error: $TREE is not a git working tree" >&2
echo "hint: use 'patch -p1' directly for tarball trees (see README)" >&2
exit 2
fi
BASE="$(head -n1 "$HERE/base.txt" | awk '{print $1}')"
if [ -n "$BASE" ]; then
if ! git -C "$TREE" rev-parse --verify "$BASE" >/dev/null 2>&1; then
echo "warning: base '$BASE' not found in target tree" >&2
fi
fi
cd "$TREE"
git am --3way --keep-cr "$HERE"/patches/*.patch
+1
View File
@@ -0,0 +1 @@
v1.29
@@ -0,0 +1,838 @@
From b57f3ec7e3c3d971c8f4b6825b7d8837f6668efb Mon Sep 17 00:00:00 2001
From: William Gill <claude@williamgill.net>
Date: Wed, 22 Apr 2026 10:17:24 -0500
Subject: [PATCH 1/2] notify: core file-access notification infrastructure
Adds an optional, observer-only notification stream to the scoutfs
kernel module. File open/read events are recorded in a per-mount
ring buffer and drained by a single privileged userspace reader
through a new ioctl.
Behavior:
- Disabled by default. A new mount option, notify_events=0|1,
turns the stream on; notify_ring_kb=N sizes the ring (4..4096 KiB,
default 64 KiB).
- New ioctl SCOUTFS_IOC_READ_NOTIFY (nr 25) fills the caller's
array of scoutfs_ioctl_notify_event records; supports a
timeout_ms wait policy. Requires CAP_SYS_ADMIN.
- Ring is lossy: on overflow records are dropped but the monotonic
seq field still advances, so a reader detects gaps. Three
percpu counters (notify_emitted, notify_dropped_ring_full,
notify_reader_attached) surface operational state via sysfs.
- Single-reader: concurrent readers receive -EBUSY.
This patch only adds the infrastructure (new files, mount options,
ioctl, lifecycle, counters). No scoutfs code path calls
scoutfs_notify_emit() yet, so behavior is bit-identical to stock
with notify_events=0 or notify_events=1.
Event type values 1..2 are defined (OPEN, READ); 3+ are reserved
for future data-waiter observation events.
---
kmod/src/Makefile | 1 +
kmod/src/counters.h | 3 +
kmod/src/ioctl.c | 3 +
kmod/src/ioctl.h | 78 +++++++++
kmod/src/notify.c | 389 ++++++++++++++++++++++++++++++++++++++++++++
kmod/src/notify.h | 39 +++++
kmod/src/options.c | 91 +++++++++++
kmod/src/options.h | 2 +
kmod/src/super.c | 3 +
kmod/src/super.h | 5 +
10 files changed, 614 insertions(+)
create mode 100644 kmod/src/notify.c
create mode 100644 kmod/src/notify.h
diff --git a/kmod/src/Makefile b/kmod/src/Makefile
index fa632aa..de14c74 100644
--- a/kmod/src/Makefile
+++ b/kmod/src/Makefile
@@ -31,6 +31,7 @@ scoutfs-y += \
lock_server.o \
msg.o \
net.o \
+ notify.o \
omap.o \
options.o \
per_task.o \
diff --git a/kmod/src/counters.h b/kmod/src/counters.h
index 9088496..94e4f44 100644
--- a/kmod/src/counters.h
+++ b/kmod/src/counters.h
@@ -156,6 +156,9 @@
EXPAND_COUNTER(net_recv_invalid_message) \
EXPAND_COUNTER(net_recv_messages) \
EXPAND_COUNTER(net_unknown_request) \
+ EXPAND_COUNTER(notify_dropped_ring_full) \
+ EXPAND_COUNTER(notify_emitted) \
+ EXPAND_COUNTER(notify_reader_attached) \
EXPAND_COUNTER(orphan_scan) \
EXPAND_COUNTER(orphan_scan_attempts) \
EXPAND_COUNTER(orphan_scan_cached) \
diff --git a/kmod/src/ioctl.c b/kmod/src/ioctl.c
index 0a5fc4c..b3a4d9a 100644
--- a/kmod/src/ioctl.c
+++ b/kmod/src/ioctl.c
@@ -47,6 +47,7 @@
#include "totl.h"
#include "wkic.h"
#include "quota.h"
+#include "notify.h"
#include "scoutfs_trace.h"
#include "util.h"
@@ -1790,6 +1791,8 @@ long scoutfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
return scoutfs_ioc_read_xattr_index(file, arg);
case SCOUTFS_IOC_PUNCH_OFFLINE:
return scoutfs_ioc_punch_offline(file, arg);
+ case SCOUTFS_IOC_READ_NOTIFY:
+ return scoutfs_ioc_read_notify(file, arg);
}
return -ENOTTY;
diff --git a/kmod/src/ioctl.h b/kmod/src/ioctl.h
index c0d2285..fc58f8c 100644
--- a/kmod/src/ioctl.h
+++ b/kmod/src/ioctl.h
@@ -876,4 +876,82 @@ struct scoutfs_ioctl_punch_offline {
#define SCOUTFS_IOC_PUNCH_OFFLINE \
_IOW(SCOUTFS_IOCTL_MAGIC, 24, struct scoutfs_ioctl_punch_offline)
+/*
+ * File access notification stream (observer-only).
+ *
+ * When the notify_events mount option is enabled, the kernel records
+ * file open and read events in a per-mount ring buffer. A single
+ * privileged reader drains the ring via this ioctl and forwards events
+ * to a userspace watcher (typically via a Unix socket).
+ *
+ * The notification stream does not alter scoutfs behavior. Events are
+ * best-effort: if the ring fills, records are dropped but the monotonic
+ * @seq field still advances, so a reader can detect gaps.
+ *
+ * Reader exclusivity: only one caller may be blocked in this ioctl at a
+ * time. A second concurrent caller receives -EBUSY.
+ *
+ * Required capability: CAP_SYS_ADMIN.
+ *
+ * Event record (64 bytes, wire-stable):
+ *
+ * @seq: monotonic sequence number assigned at emit time. Gaps
+ * indicate ring-full drops between adjacent records.
+ * @ino: scoutfs inode number of the file.
+ * @offset: byte offset for READ events; 0 for OPEN.
+ * @length: byte length for READ events; 0 for OPEN.
+ * @time_ns: CLOCK_REALTIME time of the event in nanoseconds.
+ * @pid: thread group id of the task that caused the event.
+ * @uid: effective uid of the task that caused the event, as
+ * observed from the init user namespace.
+ * @type: one of the SCOUTFS_NOTIFY_TYPE_* values.
+ * @flags: per-event flag bits.
+ */
+#define SCOUTFS_NOTIFY_TYPE_OPEN 1
+#define SCOUTFS_NOTIFY_TYPE_READ 2
+/* type values 3+ reserved for future data-waiter events */
+
+#define SCOUTFS_NOTIFY_F_WRITE_OPEN (1 << 0)
+
+struct scoutfs_ioctl_notify_event {
+ __u64 seq;
+ __u64 ino;
+ __u64 offset;
+ __u64 length;
+ __u64 time_ns;
+ __u32 pid;
+ __u32 uid;
+ __u8 type;
+ __u8 flags;
+ __u8 _pad[6];
+};
+
+/*
+ * Request for the notification read ioctl.
+ *
+ * @events_ptr: user-space address of an array of
+ * struct scoutfs_ioctl_notify_event. Must be 8-byte
+ * aligned.
+ * @events_nr: number of entries the array can hold.
+ * @timeout_ms: wait policy when the ring is empty. 0 returns
+ * immediately with 0 events (non-blocking), U32_MAX waits
+ * without a timeout, any other value waits up to that
+ * many milliseconds.
+ * @flags: reserved, must be 0.
+ *
+ * On success the number of events written to the user array is
+ * returned (may be 0 for non-blocking empty). -ETIMEDOUT indicates
+ * the timeout elapsed with no events available. -EBUSY indicates
+ * another reader is currently attached.
+ */
+struct scoutfs_ioctl_read_notify {
+ __u64 events_ptr;
+ __u32 events_nr;
+ __u32 timeout_ms;
+ __u64 flags;
+};
+
+#define SCOUTFS_IOC_READ_NOTIFY \
+ _IOWR(SCOUTFS_IOCTL_MAGIC, 25, struct scoutfs_ioctl_read_notify)
+
#endif
diff --git a/kmod/src/notify.c b/kmod/src/notify.c
new file mode 100644
index 0000000..94fab17
--- /dev/null
+++ b/kmod/src/notify.c
@@ -0,0 +1,389 @@
+/*
+ * Copyright (C) 2026 Versity Software, Inc. All rights reserved.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public
+ * License v2 as published by the Free Software Foundation.
+ *
+ * This program is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * General Public License for more details.
+ */
+
+/*
+ * Observer-only file access notifications.
+ *
+ * scoutfs_notify_emit() is called from file-op fast paths and must
+ * never block scoutfs, take sleeping locks, allocate memory, or
+ * propagate errors. It acquires one leaf spinlock, stamps the next
+ * sequence number, copies (or drops) one fixed-size record, wakes the
+ * reader, and returns. The sequence counter advances on drops so the
+ * reader sees a gap rather than a seamless stream.
+ *
+ * Userspace drains the ring via SCOUTFS_IOC_READ_NOTIFY. Only one
+ * reader may be attached at a time.
+ */
+
+#include <linux/kernel.h>
+#include <linux/fs.h>
+#include <linux/slab.h>
+#include <linux/sched.h>
+#include <linux/uaccess.h>
+#include <linux/wait.h>
+#include <linux/spinlock.h>
+#include <linux/ktime.h>
+#include <linux/cred.h>
+#include <linux/capability.h>
+#include <linux/mm.h>
+#include <linux/vmalloc.h>
+#include <linux/overflow.h>
+#include <linux/log2.h>
+#include <linux/bitops.h>
+
+#include "super.h"
+#include "options.h"
+#include "counters.h"
+#include "ioctl.h"
+#include "notify.h"
+#include "msg.h"
+
+/*
+ * Power-of-two ring of fixed-size records. head advances on write,
+ * tail advances on read, both unsigned indices masked to the ring
+ * capacity. Full when (head - tail) == capacity.
+ */
+struct notify_info {
+ spinlock_t lock;
+ wait_queue_head_t reader_wq;
+ atomic_t reader_attached;
+
+ /* all fields below this point are protected by ->lock */
+ struct scoutfs_ioctl_notify_event *ring;
+ u32 capacity; /* power of two */
+ u32 mask;
+ u32 head;
+ u32 tail;
+ u64 next_seq;
+};
+
+#define NOTIFY_MIN_KB 4
+#define NOTIFY_MAX_KB 4096
+
+static u32 ring_len_locked(struct notify_info *ni)
+{
+ return ni->head - ni->tail;
+}
+
+static bool ring_empty_locked(struct notify_info *ni)
+{
+ return ni->head == ni->tail;
+}
+
+static bool ring_full_locked(struct notify_info *ni)
+{
+ return ring_len_locked(ni) == ni->capacity;
+}
+
+/*
+ * Allocate the ring buffer. Uses kvmalloc to accept large user-chosen
+ * sizes without failing under fragmentation.
+ */
+static int alloc_ring(struct notify_info *ni, u32 capacity)
+{
+ size_t bytes;
+
+ if (!is_power_of_2(capacity))
+ return -EINVAL;
+
+ if (check_mul_overflow((size_t)capacity,
+ sizeof(struct scoutfs_ioctl_notify_event),
+ &bytes))
+ return -EINVAL;
+
+ ni->ring = kvzalloc(bytes, GFP_KERNEL);
+ if (!ni->ring)
+ return -ENOMEM;
+
+ ni->capacity = capacity;
+ ni->mask = capacity - 1;
+ ni->head = 0;
+ ni->tail = 0;
+ return 0;
+}
+
+/*
+ * Compute ring capacity in records from the options-configured size in
+ * KiB. Rounds down to a power of two so the mask-based index math is
+ * valid. Returns a record count, not a byte count.
+ */
+static u32 capacity_from_kb(u32 kb)
+{
+ u64 bytes;
+ u32 records;
+
+ if (kb < NOTIFY_MIN_KB)
+ kb = NOTIFY_MIN_KB;
+ if (kb > NOTIFY_MAX_KB)
+ kb = NOTIFY_MAX_KB;
+
+ bytes = (u64)kb << 10;
+ records = (u32)(bytes / sizeof(struct scoutfs_ioctl_notify_event));
+ if (records < 2)
+ records = 2;
+
+ /* round down to power of two */
+ return 1U << (fls(records) - 1);
+}
+
+int scoutfs_notify_setup(struct super_block *sb)
+{
+ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb);
+ struct scoutfs_mount_options opts;
+ struct notify_info *ni;
+ u32 capacity;
+ int ret;
+
+ ni = kzalloc(sizeof(*ni), GFP_KERNEL);
+ if (!ni)
+ return -ENOMEM;
+
+ spin_lock_init(&ni->lock);
+ init_waitqueue_head(&ni->reader_wq);
+ atomic_set(&ni->reader_attached, 0);
+ ni->next_seq = 1;
+
+ scoutfs_options_read(sb, &opts);
+ capacity = capacity_from_kb(opts.notify_ring_kb);
+
+ ret = alloc_ring(ni, capacity);
+ if (ret < 0) {
+ kfree(ni);
+ return ret;
+ }
+
+ sbi->notify_info = ni;
+
+ /*
+ * Publish the enable flag last so emit callers observe a fully
+ * initialized ring.
+ */
+ smp_wmb();
+ WRITE_ONCE(sbi->notify_enabled, opts.notify_events ? true : false);
+
+ return 0;
+}
+
+void scoutfs_notify_destroy(struct super_block *sb)
+{
+ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb);
+ struct notify_info *ni = sbi->notify_info;
+
+ if (!ni)
+ return;
+
+ /*
+ * Silence the emit fast path first, then wake any reader so it
+ * can exit its wait loop and stop touching the ring.
+ */
+ WRITE_ONCE(sbi->notify_enabled, false);
+ wake_up_all(&ni->reader_wq);
+
+ sbi->notify_info = NULL;
+
+ kvfree(ni->ring);
+ kfree(ni);
+}
+
+/*
+ * Best-effort observer-only event emit.
+ *
+ * - Callers must gate this on sbi->notify_enabled; we re-check under
+ * the spinlock so shutdown is race-free.
+ * - On ring-full we drop the record but still consume a sequence
+ * number so the reader sees exactly the number of drops.
+ * - Never returns an error and never blocks scoutfs.
+ */
+void scoutfs_notify_emit(struct super_block *sb, __u8 type, __u64 ino,
+ __u64 offset, __u64 length, __u8 flags)
+{
+ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb);
+ struct notify_info *ni = sbi->notify_info;
+ struct scoutfs_ioctl_notify_event *slot;
+ unsigned long irqflags;
+ u64 time_ns;
+ u32 pid;
+ u32 uid;
+ bool dropped = false;
+ bool woke = false;
+
+ if (!ni)
+ return;
+
+ time_ns = ktime_get_real_ns();
+ pid = (u32)task_tgid_nr(current);
+ uid = from_kuid(&init_user_ns, current_uid());
+
+ spin_lock_irqsave(&ni->lock, irqflags);
+
+ if (ring_full_locked(ni)) {
+ dropped = true;
+ ni->next_seq++;
+ } else {
+ slot = &ni->ring[ni->head & ni->mask];
+ slot->seq = ni->next_seq++;
+ slot->ino = ino;
+ slot->offset = offset;
+ slot->length = length;
+ slot->time_ns = time_ns;
+ slot->pid = pid;
+ slot->uid = uid;
+ slot->type = type;
+ slot->flags = flags;
+ memset(slot->_pad, 0, sizeof(slot->_pad));
+ ni->head++;
+ woke = (ring_len_locked(ni) == 1);
+ }
+
+ spin_unlock_irqrestore(&ni->lock, irqflags);
+
+ if (dropped)
+ scoutfs_inc_counter(sb, notify_dropped_ring_full);
+ else
+ scoutfs_inc_counter(sb, notify_emitted);
+
+ if (woke)
+ wake_up_interruptible(&ni->reader_wq);
+}
+
+static bool reader_should_wake(struct notify_info *ni,
+ struct scoutfs_sb_info *sbi)
+{
+ unsigned long irqflags;
+ bool empty;
+
+ spin_lock_irqsave(&ni->lock, irqflags);
+ empty = ring_empty_locked(ni);
+ spin_unlock_irqrestore(&ni->lock, irqflags);
+
+ return !empty || !READ_ONCE(sbi->notify_enabled);
+}
+
+/*
+ * Copy events out of the ring under the spinlock into a small
+ * on-stack batch, release the lock, then copy_to_user. This keeps
+ * the emit path's lock hold time bounded by batch size independent
+ * of user buffer size.
+ */
+#define NOTIFY_DRAIN_BATCH 16
+
+static int drain_ring(struct notify_info *ni,
+ struct scoutfs_ioctl_notify_event __user *udst,
+ u32 max_events)
+{
+ struct scoutfs_ioctl_notify_event batch[NOTIFY_DRAIN_BATCH];
+ unsigned long irqflags;
+ u32 total = 0;
+ u32 i, n;
+
+ while (total < max_events) {
+ n = min_t(u32, max_events - total, NOTIFY_DRAIN_BATCH);
+
+ spin_lock_irqsave(&ni->lock, irqflags);
+ n = min(n, ring_len_locked(ni));
+ for (i = 0; i < n; i++) {
+ batch[i] = ni->ring[ni->tail & ni->mask];
+ ni->tail++;
+ }
+ spin_unlock_irqrestore(&ni->lock, irqflags);
+
+ if (n == 0)
+ break;
+
+ if (copy_to_user(udst + total, batch,
+ n * sizeof(batch[0])))
+ return total ? (int)total : -EFAULT;
+
+ total += n;
+
+ if (n < NOTIFY_DRAIN_BATCH)
+ break;
+ }
+
+ return (int)total;
+}
+
+long scoutfs_ioc_read_notify(struct file *file, unsigned long arg)
+{
+ struct super_block *sb = file_inode(file)->i_sb;
+ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb);
+ struct notify_info *ni = sbi->notify_info;
+ struct scoutfs_ioctl_read_notify __user *uargs = (void __user *)arg;
+ struct scoutfs_ioctl_read_notify args;
+ struct scoutfs_ioctl_notify_event __user *uevents;
+ unsigned long irqflags;
+ long jiffies_timeout;
+ bool attached = false;
+ bool empty;
+ int ret;
+
+ if (!capable(CAP_SYS_ADMIN))
+ return -EPERM;
+
+ if (!ni)
+ return -ENODEV;
+
+ if (copy_from_user(&args, uargs, sizeof(args)))
+ return -EFAULT;
+
+ if (args.flags != 0)
+ return -EINVAL;
+ if (args.events_nr == 0)
+ return 0;
+ if (args.events_ptr & 0x7)
+ return -EINVAL;
+
+ uevents = (struct scoutfs_ioctl_notify_event __user *)
+ (unsigned long)args.events_ptr;
+
+ if (atomic_cmpxchg(&ni->reader_attached, 0, 1) != 0)
+ return -EBUSY;
+ attached = true;
+ scoutfs_inc_counter(sb, notify_reader_attached);
+
+ spin_lock_irqsave(&ni->lock, irqflags);
+ empty = ring_empty_locked(ni);
+ spin_unlock_irqrestore(&ni->lock, irqflags);
+
+ if (empty) {
+ if (args.timeout_ms == 0) {
+ ret = 0;
+ goto out;
+ }
+
+ if (args.timeout_ms == U32_MAX) {
+ ret = wait_event_interruptible(ni->reader_wq,
+ reader_should_wake(ni, sbi));
+ if (ret < 0)
+ goto out;
+ } else {
+ jiffies_timeout = msecs_to_jiffies(args.timeout_ms);
+ ret = wait_event_interruptible_timeout(ni->reader_wq,
+ reader_should_wake(ni, sbi),
+ jiffies_timeout);
+ if (ret == 0) {
+ ret = -ETIMEDOUT;
+ goto out;
+ }
+ if (ret < 0)
+ goto out;
+ }
+ }
+
+ ret = drain_ring(ni, uevents, args.events_nr);
+
+out:
+ if (attached)
+ atomic_set(&ni->reader_attached, 0);
+ return ret;
+}
diff --git a/kmod/src/notify.h b/kmod/src/notify.h
new file mode 100644
index 0000000..fe6a2d9
--- /dev/null
+++ b/kmod/src/notify.h
@@ -0,0 +1,39 @@
+#ifndef _SCOUTFS_NOTIFY_H_
+#define _SCOUTFS_NOTIFY_H_
+
+/*
+ * File access notification subsystem.
+ *
+ * Emits best-effort, observer-only events (file open, read) to a
+ * per-super in-kernel ring. A single privileged userspace reader
+ * drains the ring via SCOUTFS_IOC_READ_NOTIFY. Events are dropped on
+ * ring overflow and the sequence number advances regardless so the
+ * reader can detect gaps.
+ *
+ * The emit path is non-blocking and takes only a leaf spinlock. It
+ * does not allocate, does not sleep, and ignores any downstream state.
+ * Callers must never check its outcome.
+ *
+ * Event-type values (SCOUTFS_NOTIFY_TYPE_*) and per-event flag bits
+ * (SCOUTFS_NOTIFY_F_*) are the wire-stable constants defined in
+ * ioctl.h; they are shared between kernel emit paths and user-visible
+ * records.
+ */
+
+#include <linux/types.h>
+
+#include "ioctl.h"
+
+struct super_block;
+struct file;
+struct notify_info;
+
+int scoutfs_notify_setup(struct super_block *sb);
+void scoutfs_notify_destroy(struct super_block *sb);
+
+void scoutfs_notify_emit(struct super_block *sb, __u8 type, __u64 ino,
+ __u64 offset, __u64 length, __u8 flags);
+
+long scoutfs_ioc_read_notify(struct file *file, unsigned long arg);
+
+#endif /* _SCOUTFS_NOTIFY_H_ */
diff --git a/kmod/src/options.c b/kmod/src/options.c
index b7565d7..90e7713 100644
--- a/kmod/src/options.c
+++ b/kmod/src/options.c
@@ -38,6 +38,8 @@ enum {
Opt_log_merge_wait_timeout_ms,
Opt_metadev_path,
Opt_noacl,
+ Opt_notify_events,
+ Opt_notify_ring_kb,
Opt_orphan_scan_delay_ms,
Opt_quorum_heartbeat_timeout_ms,
Opt_quorum_slot_nr,
@@ -54,6 +56,8 @@ static const match_table_t tokens = {
{Opt_log_merge_wait_timeout_ms, "log_merge_wait_timeout_ms=%s"},
{Opt_metadev_path, "metadev_path=%s"},
{Opt_noacl, "noacl"},
+ {Opt_notify_events, "notify_events=%s"},
+ {Opt_notify_ring_kb, "notify_ring_kb=%s"},
{Opt_orphan_scan_delay_ms, "orphan_scan_delay_ms=%s"},
{Opt_quorum_heartbeat_timeout_ms, "quorum_heartbeat_timeout_ms=%s"},
{Opt_quorum_slot_nr, "quorum_slot_nr=%s"},
@@ -138,6 +142,10 @@ static void free_options(struct scoutfs_mount_options *opts)
#define DEFAULT_TCP_KEEPALIVE_TIMEOUT_MS (60 * MSEC_PER_SEC)
+#define MIN_NOTIFY_RING_KB 4U
+#define DEFAULT_NOTIFY_RING_KB 64U
+#define MAX_NOTIFY_RING_KB 4096U
+
static void init_default_options(struct scoutfs_mount_options *opts)
{
memset(opts, 0, sizeof(*opts));
@@ -147,6 +155,8 @@ static void init_default_options(struct scoutfs_mount_options *opts)
opts->ino_alloc_per_lock = SCOUTFS_LOCK_INODE_GROUP_NR;
opts->lock_idle_count = DEFAULT_LOCK_IDLE_COUNT;
opts->log_merge_wait_timeout_ms = DEFAULT_LOG_MERGE_WAIT_TIMEOUT_MS;
+ opts->notify_events = false;
+ opts->notify_ring_kb = DEFAULT_NOTIFY_RING_KB;
opts->orphan_scan_delay_ms = -1;
opts->quorum_heartbeat_timeout_ms = SCOUTFS_QUORUM_DEF_HB_TIMEO_MS;
opts->quorum_slot_nr = -1;
@@ -309,6 +319,29 @@ static int parse_options(struct super_block *sb, char *options, struct scoutfs_m
sb->s_flags &= ~SB_POSIXACL;
break;
+ case Opt_notify_events:
+ ret = match_int(args, &nr);
+ if (ret < 0 || nr < 0 || nr > 1) {
+ scoutfs_err(sb, "invalid notify_events option, bool must only be 0 or 1");
+ if (ret == 0)
+ ret = -EINVAL;
+ return ret;
+ }
+ opts->notify_events = nr;
+ break;
+
+ case Opt_notify_ring_kb:
+ ret = match_int(args, &nr);
+ if (ret < 0 || nr < MIN_NOTIFY_RING_KB || nr > MAX_NOTIFY_RING_KB) {
+ scoutfs_err(sb, "invalid notify_ring_kb option, must be between %u and %u",
+ MIN_NOTIFY_RING_KB, MAX_NOTIFY_RING_KB);
+ if (ret == 0)
+ ret = -EINVAL;
+ return ret;
+ }
+ opts->notify_ring_kb = nr;
+ break;
+
case Opt_orphan_scan_delay_ms:
if (opts->orphan_scan_delay_ms != -1) {
scoutfs_err(sb, "multiple orphan_scan_delay_ms options provided, only provide one.");
@@ -442,6 +475,8 @@ int scoutfs_options_show(struct seq_file *seq, struct dentry *root)
seq_printf(seq, ",metadev_path=%s", opts.metadev_path);
if (!is_acl)
seq_puts(seq, ",noacl");
+ seq_printf(seq, ",notify_events=%u", opts.notify_events);
+ seq_printf(seq, ",notify_ring_kb=%u", opts.notify_ring_kb);
seq_printf(seq, ",orphan_scan_delay_ms=%u", opts.orphan_scan_delay_ms);
if (opts.quorum_slot_nr >= 0)
seq_printf(seq, ",quorum_slot_nr=%d", opts.quorum_slot_nr);
@@ -651,6 +686,60 @@ static ssize_t metadev_path_show(struct kobject *kobj, struct kobj_attribute *at
}
SCOUTFS_ATTR_RO(metadev_path);
+static ssize_t notify_events_show(struct kobject *kobj, struct kobj_attribute *attr,
+ char *buf)
+{
+ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj);
+ struct scoutfs_mount_options opts;
+
+ scoutfs_options_read(sb, &opts);
+
+ return snprintf(buf, PAGE_SIZE, "%u", opts.notify_events);
+}
+static ssize_t notify_events_store(struct kobject *kobj, struct kobj_attribute *attr,
+ const char *buf, size_t count)
+{
+ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj);
+ struct scoutfs_sb_info *sbi = SCOUTFS_SB(sb);
+ DECLARE_OPTIONS_INFO(sb, optinf);
+ char nullterm[20];
+ long val;
+ int len;
+ int ret;
+
+ len = min(count, sizeof(nullterm) - 1);
+ memcpy(nullterm, buf, len);
+ nullterm[len] = '\0';
+
+ ret = kstrtol(nullterm, 0, &val);
+ if (ret < 0 || val < 0 || val > 1) {
+ scoutfs_err(sb, "invalid notify_events option, bool must be 0 or 1");
+ return -EINVAL;
+ }
+
+ write_seqlock(&optinf->seqlock);
+ optinf->opts.notify_events = val;
+ write_sequnlock(&optinf->seqlock);
+
+ /* mirror into sbi so emit fast path can check with one load */
+ WRITE_ONCE(sbi->notify_enabled, val ? true : false);
+
+ return count;
+}
+SCOUTFS_ATTR_RW(notify_events);
+
+static ssize_t notify_ring_kb_show(struct kobject *kobj, struct kobj_attribute *attr,
+ char *buf)
+{
+ struct super_block *sb = SCOUTFS_SYSFS_ATTRS_SB(kobj);
+ struct scoutfs_mount_options opts;
+
+ scoutfs_options_read(sb, &opts);
+
+ return snprintf(buf, PAGE_SIZE, "%u", opts.notify_ring_kb);
+}
+SCOUTFS_ATTR_RO(notify_ring_kb);
+
static ssize_t orphan_scan_delay_ms_show(struct kobject *kobj, struct kobj_attribute *attr,
char *buf)
{
@@ -747,6 +836,8 @@ static struct attribute *options_attrs[] = {
SCOUTFS_ATTR_PTR(lock_idle_count),
SCOUTFS_ATTR_PTR(log_merge_wait_timeout_ms),
SCOUTFS_ATTR_PTR(metadev_path),
+ SCOUTFS_ATTR_PTR(notify_events),
+ SCOUTFS_ATTR_PTR(notify_ring_kb),
SCOUTFS_ATTR_PTR(orphan_scan_delay_ms),
SCOUTFS_ATTR_PTR(quorum_heartbeat_timeout_ms),
SCOUTFS_ATTR_PTR(quorum_slot_nr),
diff --git a/kmod/src/options.h b/kmod/src/options.h
index b37bbd7..12b976e 100644
--- a/kmod/src/options.h
+++ b/kmod/src/options.h
@@ -12,6 +12,8 @@ struct scoutfs_mount_options {
int lock_idle_count;
unsigned int log_merge_wait_timeout_ms;
char *metadev_path;
+ bool notify_events;
+ unsigned int notify_ring_kb;
unsigned int orphan_scan_delay_ms;
int quorum_slot_nr;
u64 quorum_heartbeat_timeout_ms;
diff --git a/kmod/src/super.c b/kmod/src/super.c
index 3c83716..3c028b0 100644
--- a/kmod/src/super.c
+++ b/kmod/src/super.c
@@ -51,6 +51,7 @@
#include "xattr.h"
#include "wkic.h"
#include "quota.h"
+#include "notify.h"
#include "scoutfs_trace.h"
static struct dentry *scoutfs_debugfs_root;
@@ -223,6 +224,7 @@ static void scoutfs_put_super(struct super_block *sb)
scoutfs_block_destroy(sb);
scoutfs_destroy_triggers(sb);
scoutfs_fence_destroy(sb);
+ scoutfs_notify_destroy(sb);
scoutfs_options_destroy(sb);
debugfs_remove(sbi->debug_root);
scoutfs_destroy_counters(sb);
@@ -579,6 +581,7 @@ static int scoutfs_fill_super(struct super_block *sb, void *data, int silent)
scoutfs_setup_sysfs(sb) ?:
scoutfs_setup_counters(sb) ?:
scoutfs_options_setup(sb) ?:
+ scoutfs_notify_setup(sb) ?:
scoutfs_setup_triggers(sb) ?:
scoutfs_fence_setup(sb) ?:
scoutfs_block_setup(sb) ?:
diff --git a/kmod/src/super.h b/kmod/src/super.h
index 3bb10dd..d6353c9 100644
--- a/kmod/src/super.h
+++ b/kmod/src/super.h
@@ -32,6 +32,7 @@ struct volopt_info;
struct fence_info;
struct wkic_info;
struct squota_info;
+struct notify_info;
struct scoutfs_sb_info {
struct super_block *sb;
@@ -81,6 +82,10 @@ struct scoutfs_sb_info {
struct scoutfs_counters *counters;
struct scoutfs_triggers *triggers;
+ /* file access notifications (optional, gated by notify_events opt) */
+ struct notify_info *notify_info;
+ bool notify_enabled;
+
struct dentry *debug_root;
bool forced_unmount;
--
2.49.0.windows.1
@@ -0,0 +1,137 @@
From dcf118e874fa4af97144c8e0c960ac42c7e17245 Mon Sep 17 00:00:00 2001
From: William Gill <claude@williamgill.net>
Date: Wed, 22 Apr 2026 10:18:11 -0500
Subject: [PATCH 2/2] notify: file open/read hook sites
Wires the notification emit path into the two scoutfs file
operations that carry user-visible activity we want to observe.
OPEN (data.c):
- A new scoutfs_file_open() wrapper is installed as ->open in
scoutfs_file_fops. The wrapper calls generic_file_open() to
preserve the existing VFS default semantics for regular files,
then, only on success and only when notifications are enabled,
emits a SCOUTFS_NOTIFY_TYPE_OPEN record. The file mode is
inspected for FMODE_WRITE to set SCOUTFS_NOTIFY_F_WRITE_OPEN.
READ (file.c):
- Both the aio_read (KC_LINUX_HAVE_FOP_AIO_READ) and read_iter
paths get an emit placed past the existing data-waiter retry
check, guarded on (ret > 0) so only successful reads are
reported and retries never double-count. start_pos is captured
at entry of read_iter before generic_file_read_iter advances
iocb->ki_pos.
Every hook is behind unlikely(READ_ONCE(sbi->notify_enabled)), so
with the default notify_events=0 the path is a single
predicted-false branch. No scoutfs state is mutated, no error is
propagated, and no existing control flow is altered.
Nothing in the data-waiter state machine (scoutfs_data_wait_check,
scoutfs_data_wait, scoutfs_data_wait_changed, the waiter rbtree, or
SCOUTFS_IOC_DATA_WAITING / DATA_WAIT_ERR) is touched by this patch.
That work is deferred to a future series.
---
kmod/src/data.c | 28 ++++++++++++++++++++++++++++
kmod/src/file.c | 11 +++++++++++
2 files changed, 39 insertions(+)
diff --git a/kmod/src/data.c b/kmod/src/data.c
index 0abb48c..33a61af 100644
--- a/kmod/src/data.c
+++ b/kmod/src/data.c
@@ -41,6 +41,7 @@
#include "msg.h"
#include "ext.h"
#include "util.h"
+#include "notify.h"
/*
* We want to amortize work done after dirtying the shared transaction
@@ -2304,6 +2305,32 @@ const struct address_space_operations scoutfs_file_aops = {
.write_end = scoutfs_write_end,
};
+/*
+ * Thin observer wrapper around the VFS default open for regular files.
+ *
+ * generic_file_open() is exactly what the VFS uses when a file_operations
+ * table leaves ->open NULL, so installing this wrapper does not change
+ * open semantics. When notifications are enabled we emit an OPEN event
+ * after the generic open has succeeded; a failed open emits nothing.
+ */
+static int scoutfs_file_open(struct inode *inode, struct file *file)
+{
+ struct super_block *sb = inode->i_sb;
+ int ret;
+
+ ret = generic_file_open(inode, file);
+ if (ret == 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled))) {
+ u8 flags = 0;
+
+ if (file->f_mode & FMODE_WRITE)
+ flags |= SCOUTFS_NOTIFY_F_WRITE_OPEN;
+
+ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_OPEN,
+ scoutfs_ino(inode), 0, 0, flags);
+ }
+ return ret;
+}
+
const struct file_operations scoutfs_file_fops = {
#ifdef KC_LINUX_HAVE_FOP_AIO_READ
.read = do_sync_read,
@@ -2316,6 +2343,7 @@ const struct file_operations scoutfs_file_fops = {
.splice_read = generic_file_splice_read,
.splice_write = iter_file_splice_write,
#endif
+ .open = scoutfs_file_open,
.mmap = scoutfs_file_mmap,
.unlocked_ioctl = scoutfs_ioctl,
.fsync = scoutfs_file_fsync,
diff --git a/kmod/src/file.c b/kmod/src/file.c
index 15158a2..44f1bc4 100644
--- a/kmod/src/file.c
+++ b/kmod/src/file.c
@@ -29,6 +29,7 @@
#include "per_task.h"
#include "omap.h"
#include "quota.h"
+#include "notify.h"
#ifdef KC_LINUX_HAVE_FOP_AIO_READ
/*
@@ -84,6 +85,10 @@ out:
goto retry;
}
+ if (ret > 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled)))
+ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_READ,
+ scoutfs_ino(inode), pos, ret, 0);
+
return ret;
}
@@ -166,6 +171,7 @@ ssize_t scoutfs_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
struct scoutfs_lock *scoutfs_inode_lock = NULL;
SCOUTFS_DECLARE_PER_TASK_ENTRY(pt_ent);
DECLARE_DATA_WAIT(dw);
+ loff_t start_pos = iocb->ki_pos;
int ret;
retry:
@@ -200,6 +206,11 @@ out:
if (ret == 0)
goto retry;
}
+
+ if (ret > 0 && unlikely(READ_ONCE(SCOUTFS_SB(sb)->notify_enabled)))
+ scoutfs_notify_emit(sb, SCOUTFS_NOTIFY_TYPE_READ,
+ scoutfs_ino(inode), start_pos, ret, 0);
+
return ret;
}
--
2.49.0.windows.1