git: 741cd29526e0 - main - nfscl: Yet more fixes for the NFS over RDMA client glue

From: Rick Macklem <rmacklem_at_FreeBSD.org>
Date: Sat, 12 Sep 2026 22:49:03 UTC
The branch main has been updated by rmacklem:

URL: https://cgit.FreeBSD.org/src/commit/?id=741cd29526e0033c5b8282f79cae828bf2ad12a8

commit 741cd29526e0033c5b8282f79cae828bf2ad12a8
Author:     Rick Macklem <rmacklem@FreeBSD.org>
AuthorDate: 2026-09-12 22:47:20 +0000
Commit:     Rick Macklem <rmacklem@FreeBSD.org>
CommitDate: 2026-09-12 22:47:20 +0000

    nfscl: Yet more fixes for the NFS over RDMA client glue
    
    This should be it for a while, but there will be another cycle
    of "glue" updates.  I just found out that I'll need to create
    an alternate code path that uses a contigmalloc() blob instead
    of scatter/gather of pages, since some NICs cannot do the
    scatter/gather of pages well.
    
    This commit should not affect non-RDMA behaviour.
    
    MFC after:      3 months
    Fixes:  884ee8d6c9b4 ("nfscl: Add some glue for client side NFS over RDMA")
---
 sys/fs/nfs/nfs_commonkrpc.c     |  2 ++
 sys/fs/nfsclient/nfs_clrpcops.c |  6 ++++--
 sys/rpc/clnt_rc.c               |  2 +-
 sys/rpc/clntrdma.h              | 22 ++++++++++++++++++++--
 sys/rpc/rpc_generic.c           | 21 +++++++++++++++++++++
 5 files changed, 48 insertions(+), 5 deletions(-)

diff --git a/sys/fs/nfs/nfs_commonkrpc.c b/sys/fs/nfs/nfs_commonkrpc.c
index 813005ed0cc4..415bba961716 100644
--- a/sys/fs/nfs/nfs_commonkrpc.c
+++ b/sys/fs/nfs/nfs_commonkrpc.c
@@ -59,6 +59,7 @@
 
 #include <rpc/rpc.h>
 #include <rpc/krpc.h>
+#include <rpc/clntrdma.h>
 
 #include <kgssapi/krb5/kcrypto.h>
 
@@ -1535,6 +1536,7 @@ out:
 	}
 #endif
 
+	rpc_remove_mreduce(nd->nd_mreq, false);	/* Will be free'd by caller. */
 	m_freem(nd->nd_mreq);
 	if (usegssname == 0)
 		AUTH_DESTROY(auth);
diff --git a/sys/fs/nfsclient/nfs_clrpcops.c b/sys/fs/nfsclient/nfs_clrpcops.c
index 0c03ca07d63d..d69c630f5e47 100644
--- a/sys/fs/nfsclient/nfs_clrpcops.c
+++ b/sys/fs/nfsclient/nfs_clrpcops.c
@@ -1838,7 +1838,8 @@ nfsrpc_readrpc(vnode_t vp, struct uio *uiop, struct ucred *cred,
 	mbflag = 0;
 	if (NFSHASRDMA(nmp) && tsiz > 0) {
 		/* Assume the rest of the RPC without data is <= 1024 bytes. */
-		if ((uiop->uio_offset & PAGE_MASK) == 0)
+		if ((uiop->uio_offset & PAGE_MASK) == 0 &&
+		    uiop->uio_segflg == UIO_SYSSPACE)
 			did_rdma = true;
 		else if (tsiz <= PAGE_SIZE - 1024)
 			mbflag = M_PROTO7;
@@ -2054,7 +2055,8 @@ nfsrpc_writerpc(vnode_t vp, struct uio *uiop, int *iomode,
 	*attrflagp = 0;
 	tsiz = uiop->uio_resid;
 	did_rdma = false;
-	if (NFSHASRDMA(nmp) && (uiop->uio_offset & PAGE_MASK) == 0)
+	if (NFSHASRDMA(nmp) && (uiop->uio_offset & PAGE_MASK) == 0 &&
+	    uiop->uio_segflg == UIO_SYSSPACE)
 		did_rdma = true;
 	tmp_off = uiop->uio_offset + tsiz;
 	NFSLOCKMNT(nmp);
diff --git a/sys/rpc/clnt_rc.c b/sys/rpc/clnt_rc.c
index 1a743d2be1e3..c19b3e5eefc8 100644
--- a/sys/rpc/clnt_rc.c
+++ b/sys/rpc/clnt_rc.c
@@ -185,7 +185,7 @@ clnt_reconnect_connect(CLIENT *cl)
 			    rc->rc_rdmamax_io, rc->rc_rdma_cbslots,
 			    &rc->rc_err);
 		} else {
-			stat = RPC_FAILED;
+			rc->rc_err.re_status = stat = RPC_FAILED;
 			newclient = NULL;
 		}
 	} else {
diff --git a/sys/rpc/clntrdma.h b/sys/rpc/clntrdma.h
index 21c61374692b..1b5ecb16f457 100644
--- a/sys/rpc/clntrdma.h
+++ b/sys/rpc/clntrdma.h
@@ -27,13 +27,27 @@ struct rpcrdma_xprt {
 	uint32_t	maxbck;
 	uint32_t	maxio;
 	uint32_t	maxsge;
+	uint32_t	use_bounce;
 	void		*ep;
 };
 
 #define	RPCRDMA_MAX_SEGMENTS	16	/* Limit from RFC8267. */
 #define	RPCRDMA_MAX_INLINE	1024	/* Limit from RFC8267. */
 
-#define	RPCRDMA_MAX_SGE		(16 + 2)
+/*
+ * Since I/O sizes are always a power of 2 and RPCRDMA_MAX_SEGMENTS is a
+ * power of 2, RPCRDMA_MAX_SGE must be a power of 2 plus 1.
+ * The +1 is for the rest of the RPC message that goes along with the data.
+ * This value sets the size of the arrays and, as such, is the upper bound
+ * for xp->maxsge, which is set to a power of 2 + 1 by xprt_rdma_connect().
+ * At this time, the power of 2 is 16 for Mellanox, 8 for Intel and ?? for
+ * Chelsio.  If future NICs support larger powers of 2, based on their
+ * max_sge, max_sge_rd and max_qp_rd_atom values, this constant can be
+ * increased.  The only effect is making the array sizes in
+ * "struct _rpcrdma_chunk_priv" larger.
+ */
+#define	RPCRDMA_MAX_SGE		(16 + 1)
+
 struct rpcrdma_chunk {
 	int			ind;
 	unsigned int		first_off;
@@ -83,7 +97,9 @@ struct rpcrdma_chunk *xprt_rdma_create_chunk(struct rpcrdma_xprt *xp,
     uint32_t num_pg, struct rpcrdma_reduce_pg *rb, struct mbuf *mextpg,
     bool into_mem, int ind);
 
-int xprt_rdma_disconnected(struct rpcrdma_xprt *xp);
+int xprt_rdma_isdisconnected(struct rpcrdma_xprt *xp);
+
+void xprt_rdma_mark_disconnected(struct rpcrdma_xprt *xp);
 
 int xprt_rdma_acquire_buf(struct rpcrdma_xprt *xp, int start, int end);
 
@@ -97,6 +113,8 @@ int rpc_copy_uio_pages(struct mbuf *mr, struct uio *uiop, int len,
     bool from_pages);
 
 void rpc_copy_mbuf_to_rb(struct mbuf *m, struct rpcrdma_reduce_pg *rb);
+
+void rpc_remove_mreduce(struct mbuf *m, bool free_it);
 #endif	/* _KERNEL */
 
 #endif	/* _RPC_CLNTRDMA_H_ */
diff --git a/sys/rpc/rpc_generic.c b/sys/rpc/rpc_generic.c
index 7172a19164c2..c540f86df032 100644
--- a/sys/rpc/rpc_generic.c
+++ b/sys/rpc/rpc_generic.c
@@ -1089,6 +1089,27 @@ rpc_copy_mbuf_to_rb(struct mbuf *m, struct rpcrdma_reduce_pg *rb)
 	}
 }
 
+/*
+ * Remove the reduce mbuf from the list and, optionally, free it.
+ */
+void
+rpc_remove_mreduce(struct mbuf *m, bool free_it)
+{
+	struct mbuf *mreduce, **mreduce_prev;
+
+	mreduce_prev = &m;
+	for (mreduce = m; mreduce != NULL &&
+	    (mreduce->m_flags & M_PROTO10) == 0;
+	    mreduce = mreduce->m_next)
+		mreduce_prev = &mreduce->m_next;
+	if (mreduce != NULL) {
+		*mreduce_prev = mreduce->m_next;
+		mreduce->m_next = NULL;
+		if (free_it)
+			m_free(mreduce);
+	}
+}
+
 /*
  * Kernel module glue
  */