mirror of
https://github.com/SCST-project/scst.git
synced 2026-07-28 19:12:55 +00:00
svn+ssh://vlnb@svn.code.sf.net/p/scst/svn/trunk ........ r5955 | bvassche | 2014-12-22 05:10:41 -0800 (Mon, 22 Dec 2014) | 1 line Update for kernel 3.18 ........ r5959 | bvassche | 2015-01-06 05:25:28 -0800 (Tue, 06 Jan 2015) | 1 line scst_calc_block_shift: Log block shift and sector size upon mismatch ........ r5960 | bvassche | 2015-01-07 01:20:06 -0800 (Wed, 07 Jan 2015) | 4 lines scst_local: Fix unique per session sas address Signed-off-by: Sebastian Herbszt <herbszt@gmx.de> ........ r5961 | bvassche | 2015-01-09 04:23:25 -0800 (Fri, 09 Jan 2015) | 4 lines scst_sysfs: return EINVAL on too big LUN Signed-off-by: Sebastian Herbszt <herbszt@gmx.de> ........ r5965 | bvassche | 2015-01-13 00:55:46 -0800 (Tue, 13 Jan 2015) | 68 lines qla2x00t: Copy entire SCST sense buffer to q2x ctio There seems to be a bug in passing sense information to QLA HBAs, where the last 2 bytes of the sense data (ASC, ASCQ) are not copied to the low level sense buffer. We encountered this in ESX, which relies on these 2 bytes to parse the MISCOMPARE sense code (0xE1, 0x1D, 0x00). Bellow is a simple test to recreate this issue, but during vMotion operations (where VMs are moved from one host to another), this may cause the operation to fail leaving the VM in an inconsistent state. The test I ran to verify that we are indeed missing the bytes is the following: 1. Create a SCST based device 2. Expose the device to 2 ESX hosts 3. Format the device as VMFS5, create a test directory 4. From both hosts, I start writing to this directory (no VMs involved, just write normal files) At this stage, both ESX hosts try to take access to the directory. The VMFS filesystem contains a per-directory lock which is managed by COMPARE AND WRITE command. Each ESX will attempt to change the VMFS lock location from unlocked to locked to create the new file. Obviously there are bound to be failures (which are equivalent to programming locking conflicts), these are reported by the MISCOMPARE sense code. Upon these MISCOMPARE errors, the host will re-try taking the lock until it succeeds, and will then proceed to perform the write operation on the directory. Due to the bug in copying the sense buffer from the SCST core to the QLA ctio, instead of the full sense code, only the key (0xE) is sent, and ESX does not know how to handle it resulting in IO error. Here are the errors as they appear on the command line: /vmfs/volumes/54a297c4-ca5af1cc-7f94-002219d20f28/ats_test # ./open_close_test-esx2.sh ./open_close_test-esx2.sh: line 8: can't create ats_fileoptest-esx2_1.txt: Input/output error ./open_close_test-esx2.sh: line 8: can't create ats_fileoptest-esx2_21.txt: Input/output error ./open_close_test-esx2.sh: line 8: can't create ats_fileoptest-esx2_110.txt: Input/output error ./open_close_test-esx2.sh: line 8: can't create ats_fileoptest-esx2_111.txt: Input/output error In the /var/log/vmkernel.log, we can see that the sense information is missing (0xE, 0x0, 0x0) instead of (0xE, 0x1D, 0x0). 2014-12-30T12:13:20.714Z cpu6:33519)ScsiDeviceIO: 2338: Cmd(0x412e84f957c0) 0x89, CmdSN 0x234d from world 519051 to dev "eui.0024f400d5020007" failed H:0x0 D:0x2 P:0x0 Valid sense data: 0xe 0x0 0x0. 2014-12-30T12:13:20.766Z cpu6:33519)ScsiDeviceIO: 2338: Cmd(0x412e84f91d00) 0x89, CmdSN 0x2350 from world 519051 to dev "eui.0024f400d5020007" failed H:0x0 D:0x2 P:0x0 Valid sense data: 0xe 0x0 0x0. 2014-12-30T12:13:20.766Z cpu6:33519)ScsiDeviceIO: 2338: Cmd(0x412e80449fc0) 0x89, CmdSN 0x234f from world 519051 to dev "eui.0024f400d5020007" failed H:0x0 D:0x2 P:0x0 Valid sense data: 0xe 0x0 0x0. This patch fixes this issue, the test will run without a problem with the fix (no IO errors, all the files are properly written to the directory). Signed-off-by: Shahar Salzman <shahar.salzman@kaminario.com> Reviewed-by: Eran Mann <eran.mann@kaminario.com> [bvanassche: simplified implementation] Signed-off-by: Bart Van Assche <bvanassche@acm.org> ........ git-svn-id: http://svn.code.sf.net/p/scst/svn/branches/3.0.x@6110 d57e44dd-8a1f-0410-8b47-8ef2f437770f
388 lines
12 KiB
Diff
388 lines
12 KiB
Diff
Subject: [PATCH] put_page_callback
|
|
|
|
---
|
|
drivers/block/drbd/drbd_receiver.c | 2 +-
|
|
include/linux/mm_types.h | 11 +++++++++
|
|
include/linux/net.h | 40 ++++++++++++++++++++++++++++++
|
|
include/linux/skbuff.h | 4 +--
|
|
net/Kconfig | 12 +++++++++
|
|
net/ceph/pagevec.c | 2 +-
|
|
net/core/skbuff.c | 14 +++++------
|
|
net/core/sock.c | 4 +--
|
|
net/ipv4/Makefile | 1 +
|
|
net/ipv4/ip_output.c | 4 +--
|
|
net/ipv4/tcp.c | 4 +--
|
|
net/ipv4/tcp_zero_copy.c | 50 ++++++++++++++++++++++++++++++++++++++
|
|
net/ipv6/ip6_output.c | 2 +-
|
|
13 files changed, 132 insertions(+), 18 deletions(-)
|
|
create mode 100644 net/ipv4/tcp_zero_copy.c
|
|
|
|
diff --git a/drivers/block/drbd/drbd_receiver.c b/drivers/block/drbd/drbd_receiver.c
|
|
index 6960fb0..8fa4016 100644
|
|
--- a/drivers/block/drbd/drbd_receiver.c
|
|
+++ b/drivers/block/drbd/drbd_receiver.c
|
|
@@ -132,7 +132,7 @@ static int page_chain_free(struct page *page)
|
|
struct page *tmp;
|
|
int i = 0;
|
|
page_chain_for_each_safe(page, tmp) {
|
|
- put_page(page);
|
|
+ net_put_page(page);
|
|
++i;
|
|
}
|
|
return i;
|
|
diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h
|
|
index 6e0b286..5706a4d 100644
|
|
--- a/include/linux/mm_types.h
|
|
+++ b/include/linux/mm_types.h
|
|
@@ -196,6 +196,17 @@ struct page {
|
|
#ifdef LAST_CPUPID_NOT_IN_PAGE_FLAGS
|
|
int _last_cpupid;
|
|
#endif
|
|
+
|
|
+#if defined(CONFIG_TCP_ZERO_COPY_TRANSFER_COMPLETION_NOTIFICATION)
|
|
+ /*
|
|
+ * Used to implement support for notification on zero-copy TCP transfer
|
|
+ * completion. It might look as not good to have this field here and
|
|
+ * it's better to have it in struct sk_buff, but it would make the code
|
|
+ * much more complicated and fragile, since all skb then would have to
|
|
+ * contain only pages with the same value in this field.
|
|
+ */
|
|
+ void *net_priv;
|
|
+#endif
|
|
}
|
|
/*
|
|
* The struct page can be forced to be double word aligned so that atomic ops
|
|
diff --git a/include/linux/net.h b/include/linux/net.h
|
|
index 17d8339..f784384 100644
|
|
--- a/include/linux/net.h
|
|
+++ b/include/linux/net.h
|
|
@@ -19,6 +19,7 @@
|
|
#define _LINUX_NET_H
|
|
|
|
#include <linux/stringify.h>
|
|
+#include <linux/mm.h>
|
|
#include <linux/random.h>
|
|
#include <linux/wait.h>
|
|
#include <linux/fcntl.h> /* For O_CLOEXEC and O_NONBLOCK */
|
|
@@ -285,6 +286,45 @@ int kernel_sendpage(struct socket *sock, struct page *page, int offset,
|
|
int kernel_sock_ioctl(struct socket *sock, int cmd, unsigned long arg);
|
|
int kernel_sock_shutdown(struct socket *sock, enum sock_shutdown_cmd how);
|
|
|
|
+#if defined(CONFIG_TCP_ZERO_COPY_TRANSFER_COMPLETION_NOTIFICATION)
|
|
+/* Support for notification on zero-copy TCP transfer completion */
|
|
+typedef void (*net_get_page_callback_t)(struct page *page);
|
|
+typedef void (*net_put_page_callback_t)(struct page *page);
|
|
+
|
|
+extern net_get_page_callback_t net_get_page_callback;
|
|
+extern net_put_page_callback_t net_put_page_callback;
|
|
+
|
|
+extern int net_set_get_put_page_callbacks(
|
|
+ net_get_page_callback_t get_callback,
|
|
+ net_put_page_callback_t put_callback);
|
|
+
|
|
+/*
|
|
+ * See comment for net_set_get_put_page_callbacks() why those functions
|
|
+ * don't need any protection.
|
|
+ */
|
|
+static inline void net_get_page(struct page *page)
|
|
+{
|
|
+ if (page->net_priv != 0)
|
|
+ net_get_page_callback(page);
|
|
+ get_page(page);
|
|
+}
|
|
+static inline void net_put_page(struct page *page)
|
|
+{
|
|
+ if (page->net_priv != 0)
|
|
+ net_put_page_callback(page);
|
|
+ put_page(page);
|
|
+}
|
|
+#else
|
|
+static inline void net_get_page(struct page *page)
|
|
+{
|
|
+ get_page(page);
|
|
+}
|
|
+static inline void net_put_page(struct page *page)
|
|
+{
|
|
+ put_page(page);
|
|
+}
|
|
+#endif /* CONFIG_TCP_ZERO_COPY_TRANSFER_COMPLETION_NOTIFICATION */
|
|
+
|
|
#define MODULE_ALIAS_NETPROTO(proto) \
|
|
MODULE_ALIAS("net-pf-" __stringify(proto))
|
|
|
|
diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h
|
|
index 6c8b6f6..edf6195 100644
|
|
--- a/include/linux/skbuff.h
|
|
+++ b/include/linux/skbuff.h
|
|
@@ -2250,7 +2250,7 @@ static inline struct page *skb_frag_page(const skb_frag_t *frag)
|
|
*/
|
|
static inline void __skb_frag_ref(skb_frag_t *frag)
|
|
{
|
|
- get_page(skb_frag_page(frag));
|
|
+ net_get_page(skb_frag_page(frag));
|
|
}
|
|
|
|
/**
|
|
@@ -2273,7 +2273,7 @@ static inline void skb_frag_ref(struct sk_buff *skb, int f)
|
|
*/
|
|
static inline void __skb_frag_unref(skb_frag_t *frag)
|
|
{
|
|
- put_page(skb_frag_page(frag));
|
|
+ net_put_page(skb_frag_page(frag));
|
|
}
|
|
|
|
/**
|
|
diff --git a/net/Kconfig b/net/Kconfig
|
|
index 99815b5..ac45213 100644
|
|
--- a/net/Kconfig
|
|
+++ b/net/Kconfig
|
|
@@ -76,6 +76,18 @@ config INET
|
|
|
|
Short answer: say Y.
|
|
|
|
+config TCP_ZERO_COPY_TRANSFER_COMPLETION_NOTIFICATION
|
|
+ bool "TCP/IP zero-copy transfer completion notification"
|
|
+ depends on INET
|
|
+ default SCST_ISCSI
|
|
+ ---help---
|
|
+ Adds support for sending a notification upon completion of a
|
|
+ zero-copy TCP/IP transfer. This can speed up certain TCP/IP
|
|
+ software. Currently this is only used by the iSCSI target driver
|
|
+ iSCSI-SCST.
|
|
+
|
|
+ If unsure, say N.
|
|
+
|
|
if INET
|
|
source "net/ipv4/Kconfig"
|
|
source "net/ipv6/Kconfig"
|
|
diff --git a/net/ceph/pagevec.c b/net/ceph/pagevec.c
|
|
index 5550130..993f710 100644
|
|
--- a/net/ceph/pagevec.c
|
|
+++ b/net/ceph/pagevec.c
|
|
@@ -51,7 +51,7 @@ void ceph_put_page_vector(struct page **pages, int num_pages, bool dirty)
|
|
for (i = 0; i < num_pages; i++) {
|
|
if (dirty)
|
|
set_page_dirty_lock(pages[i]);
|
|
- put_page(pages[i]);
|
|
+ net_put_page(pages[i]);
|
|
}
|
|
if (is_vmalloc_addr(pages))
|
|
vfree(pages);
|
|
diff --git a/net/core/skbuff.c b/net/core/skbuff.c
|
|
index 32e31c2..6eb3a9e 100644
|
|
--- a/net/core/skbuff.c
|
|
+++ b/net/core/skbuff.c
|
|
@@ -437,7 +437,7 @@ struct sk_buff *__netdev_alloc_skb(struct net_device *dev,
|
|
if (likely(data)) {
|
|
skb = build_skb(data, fragsz);
|
|
if (unlikely(!skb))
|
|
- put_page(virt_to_head_page(data));
|
|
+ net_put_page(virt_to_head_page(data));
|
|
}
|
|
} else {
|
|
skb = __alloc_skb(length + NET_SKB_PAD, gfp_mask,
|
|
@@ -495,7 +495,7 @@ static void skb_clone_fraglist(struct sk_buff *skb)
|
|
static void skb_free_head(struct sk_buff *skb)
|
|
{
|
|
if (skb->head_frag)
|
|
- put_page(virt_to_head_page(skb->head));
|
|
+ net_put_page(virt_to_head_page(skb->head));
|
|
else
|
|
kfree(skb->head);
|
|
}
|
|
@@ -822,7 +822,7 @@ int skb_copy_ubufs(struct sk_buff *skb, gfp_t gfp_mask)
|
|
if (!page) {
|
|
while (head) {
|
|
struct page *next = (struct page *)page_private(head);
|
|
- put_page(head);
|
|
+ net_put_page(head);
|
|
head = next;
|
|
}
|
|
return -ENOMEM;
|
|
@@ -1669,7 +1669,7 @@ EXPORT_SYMBOL(skb_copy_bits);
|
|
*/
|
|
static void sock_spd_release(struct splice_pipe_desc *spd, unsigned int i)
|
|
{
|
|
- put_page(spd->pages[i]);
|
|
+ net_put_page(spd->pages[i]);
|
|
}
|
|
|
|
static struct page *linear_to_page(struct page *page, unsigned int *len,
|
|
@@ -1722,7 +1722,7 @@ static bool spd_fill_page(struct splice_pipe_desc *spd,
|
|
spd->partial[spd->nr_pages - 1].len += *len;
|
|
return false;
|
|
}
|
|
- get_page(page);
|
|
+ net_get_page(page);
|
|
spd->pages[spd->nr_pages] = page;
|
|
spd->partial[spd->nr_pages].len = *len;
|
|
spd->partial[spd->nr_pages].offset = offset;
|
|
@@ -2181,7 +2181,7 @@ skb_zerocopy(struct sk_buff *to, struct sk_buff *from, int len, int hlen)
|
|
page = virt_to_head_page(from->head);
|
|
offset = from->data - (unsigned char *)page_address(page);
|
|
__skb_fill_page_desc(to, 0, page, offset, plen);
|
|
- get_page(page);
|
|
+ net_get_page(page);
|
|
j = 1;
|
|
len -= plen;
|
|
}
|
|
@@ -2835,7 +2835,7 @@ int skb_append_datato_frags(struct sock *sk, struct sk_buff *skb,
|
|
copy);
|
|
frg_cnt++;
|
|
pfrag->offset += copy;
|
|
- get_page(pfrag->page);
|
|
+ net_get_page(pfrag->page);
|
|
|
|
skb->truesize += copy;
|
|
atomic_add(copy, &sk->sk_wmem_alloc);
|
|
diff --git a/net/core/sock.c b/net/core/sock.c
|
|
index 15e0c67..e8ea0df 100644
|
|
--- a/net/core/sock.c
|
|
+++ b/net/core/sock.c
|
|
@@ -1830,7 +1830,7 @@ bool skb_page_frag_refill(unsigned int sz, struct page_frag *pfrag, gfp_t gfp)
|
|
}
|
|
if (pfrag->offset + sz <= pfrag->size)
|
|
return true;
|
|
- put_page(pfrag->page);
|
|
+ net_put_page(pfrag->page);
|
|
}
|
|
|
|
pfrag->offset = 0;
|
|
@@ -2581,7 +2581,7 @@ void sk_common_release(struct sock *sk)
|
|
sk_refcnt_debug_release(sk);
|
|
|
|
if (sk->sk_frag.page) {
|
|
- put_page(sk->sk_frag.page);
|
|
+ net_put_page(sk->sk_frag.page);
|
|
sk->sk_frag.page = NULL;
|
|
}
|
|
|
|
diff --git a/net/ipv4/Makefile b/net/ipv4/Makefile
|
|
index 518c04e..4072a87 100644
|
|
--- a/net/ipv4/Makefile
|
|
+++ b/net/ipv4/Makefile
|
|
@@ -57,6 +57,7 @@ obj-$(CONFIG_TCP_CONG_ILLINOIS) += tcp_illinois.o
|
|
obj-$(CONFIG_MEMCG_KMEM) += tcp_memcontrol.o
|
|
obj-$(CONFIG_NETLABEL) += cipso_ipv4.o
|
|
obj-$(CONFIG_GENEVE) += geneve.o
|
|
+obj-$(CONFIG_TCP_ZERO_COPY_TRANSFER_COMPLETION_NOTIFICATION) += tcp_zero_copy.o
|
|
|
|
obj-$(CONFIG_XFRM) += xfrm4_policy.o xfrm4_state.o xfrm4_input.o \
|
|
xfrm4_output.o xfrm4_protocol.o
|
|
diff --git a/net/ipv4/ip_output.c b/net/ipv4/ip_output.c
|
|
index bc6471d..ab9e262 100644
|
|
--- a/net/ipv4/ip_output.c
|
|
+++ b/net/ipv4/ip_output.c
|
|
@@ -1051,7 +1051,7 @@ alloc_new_skb:
|
|
__skb_fill_page_desc(skb, i, pfrag->page,
|
|
pfrag->offset, 0);
|
|
skb_shinfo(skb)->nr_frags = ++i;
|
|
- get_page(pfrag->page);
|
|
+ net_get_page(pfrag->page);
|
|
}
|
|
copy = min_t(int, copy, pfrag->size - pfrag->offset);
|
|
if (getfrag(from,
|
|
@@ -1276,7 +1276,7 @@ ssize_t ip_append_page(struct sock *sk, struct flowi4 *fl4, struct page *page,
|
|
if (skb_can_coalesce(skb, i, page, offset)) {
|
|
skb_frag_size_add(&skb_shinfo(skb)->frags[i-1], len);
|
|
} else if (i < MAX_SKB_FRAGS) {
|
|
- get_page(page);
|
|
+ net_get_page(page);
|
|
skb_fill_page_desc(skb, i, page, offset, len);
|
|
} else {
|
|
err = -EMSGSIZE;
|
|
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
|
|
index 38c2bcb..f089a7a 100644
|
|
--- a/net/ipv4/tcp.c
|
|
+++ b/net/ipv4/tcp.c
|
|
@@ -949,7 +949,7 @@ new_segment:
|
|
if (can_coalesce) {
|
|
skb_frag_size_add(&skb_shinfo(skb)->frags[i - 1], copy);
|
|
} else {
|
|
- get_page(page);
|
|
+ net_get_page(page);
|
|
skb_fill_page_desc(skb, i, page, offset, copy);
|
|
}
|
|
skb_shinfo(skb)->tx_flags |= SKBTX_SHARED_FRAG;
|
|
@@ -1250,7 +1250,7 @@ new_segment:
|
|
} else {
|
|
skb_fill_page_desc(skb, i, pfrag->page,
|
|
pfrag->offset, copy);
|
|
- get_page(pfrag->page);
|
|
+ net_get_page(pfrag->page);
|
|
}
|
|
pfrag->offset += copy;
|
|
}
|
|
diff --git a/net/ipv4/tcp_zero_copy.c b/net/ipv4/tcp_zero_copy.c
|
|
new file mode 100644
|
|
index 0000000..430147e
|
|
--- /dev/null
|
|
+++ b/net/ipv4/tcp_zero_copy.c
|
|
@@ -0,0 +1,50 @@
|
|
+/*
|
|
+ * Support routines for TCP zero copy transmit
|
|
+ *
|
|
+ * Created by Vladislav Bolkhovitin
|
|
+ *
|
|
+ * This program is free software; you can redistribute it and/or
|
|
+ * modify it under the terms of the GNU General Public License
|
|
+ * version 2 as published by the Free Software Foundation.
|
|
+ */
|
|
+
|
|
+#include <linux/export.h>
|
|
+#include <linux/skbuff.h>
|
|
+
|
|
+net_get_page_callback_t net_get_page_callback __read_mostly;
|
|
+EXPORT_SYMBOL_GPL(net_get_page_callback);
|
|
+
|
|
+net_put_page_callback_t net_put_page_callback __read_mostly;
|
|
+EXPORT_SYMBOL_GPL(net_put_page_callback);
|
|
+
|
|
+/*
|
|
+ * Caller of this function must ensure that at the moment when it's called
|
|
+ * there are no pages in the system with net_priv field set to non-zero
|
|
+ * value. Hence, this function, as well as net_get_page() and net_put_page(),
|
|
+ * don't need any protection.
|
|
+ */
|
|
+int net_set_get_put_page_callbacks(
|
|
+ net_get_page_callback_t get_callback,
|
|
+ net_put_page_callback_t put_callback)
|
|
+{
|
|
+ int res = 0;
|
|
+
|
|
+ if ((net_get_page_callback != NULL) && (get_callback != NULL) &&
|
|
+ (net_get_page_callback != get_callback)) {
|
|
+ res = -EBUSY;
|
|
+ goto out;
|
|
+ }
|
|
+
|
|
+ if ((net_put_page_callback != NULL) && (put_callback != NULL) &&
|
|
+ (net_put_page_callback != put_callback)) {
|
|
+ res = -EBUSY;
|
|
+ goto out;
|
|
+ }
|
|
+
|
|
+ net_get_page_callback = get_callback;
|
|
+ net_put_page_callback = put_callback;
|
|
+
|
|
+out:
|
|
+ return res;
|
|
+}
|
|
+EXPORT_SYMBOL_GPL(net_set_get_put_page_callbacks);
|
|
diff --git a/net/ipv6/ip6_output.c b/net/ipv6/ip6_output.c
|
|
index 8e950c2..8cb4760 100644
|
|
--- a/net/ipv6/ip6_output.c
|
|
+++ b/net/ipv6/ip6_output.c
|
|
@@ -1472,7 +1472,7 @@ alloc_new_skb:
|
|
__skb_fill_page_desc(skb, i, pfrag->page,
|
|
pfrag->offset, 0);
|
|
skb_shinfo(skb)->nr_frags = ++i;
|
|
- get_page(pfrag->page);
|
|
+ net_get_page(pfrag->page);
|
|
}
|
|
copy = min_t(int, copy, pfrag->size - pfrag->offset);
|
|
if (getfrag(from,
|
|
--
|
|
2.1.2
|
|
|