git: b78f8fac5bd3 - main - iw_cxgbe/libcxgb4: Support for RDMA WRITE_COMPL work requests
- Go to: [ bottom of page ] [ top of archives ] [ this month ]
Date: Tue, 15 Sep 2026 14:47:46 UTC
The branch main has been updated by jhb:
URL: https://cgit.FreeBSD.org/src/commit/?id=b78f8fac5bd3d3c44c1e3887eff01f230786ce15
commit b78f8fac5bd3d3c44c1e3887eff01f230786ce15
Author: Steve Wise <swise@opengridcomputing.com>
AuthorDate: 2017-10-30 20:23:45 +0000
Commit: John Baldwin <jhb@FreeBSD.org>
CommitDate: 2026-09-15 14:22:39 +0000
iw_cxgbe/libcxgb4: Support for RDMA WRITE_COMPL work requests
To optimize NVME-oF READ IOPs, use a specialized work request that
combines a RDMA WRITE and SEND_INV chain into a single request.
Sponsored by: Chelsio Communications
Co-authored-by: Potnuri Bharat Teja <bharat@chelsio.com>
---
contrib/ofed/libcxgb4/dev.c | 1 +
contrib/ofed/libcxgb4/libcxgb4.h | 1 +
contrib/ofed/libcxgb4/qp.c | 177 ++++++++++++++++++++++++++++++++++++---
contrib/ofed/libcxgb4/t4.h | 9 +-
sys/dev/cxgbe/iw_cxgbe/device.c | 2 +
sys/dev/cxgbe/iw_cxgbe/qp.c | 147 +++++++++++++++++++++++++++++++-
sys/dev/cxgbe/iw_cxgbe/t4.h | 6 +-
7 files changed, 329 insertions(+), 14 deletions(-)
diff --git a/contrib/ofed/libcxgb4/dev.c b/contrib/ofed/libcxgb4/dev.c
index db728f5627da..a5e4d04f78ab 100644
--- a/contrib/ofed/libcxgb4/dev.c
+++ b/contrib/ofed/libcxgb4/dev.c
@@ -206,6 +206,7 @@ static struct ibv_context *c4iw_alloc_context(struct ibv_device *ibdev,
if (t5_en_wc && !context->status_page->wc_supported) {
t5_en_wc = 0;
}
+ rhp->write_cmpl_supported = context->status_page->write_cmpl_supported;
}
return &context->ibv_ctx;
diff --git a/contrib/ofed/libcxgb4/libcxgb4.h b/contrib/ofed/libcxgb4/libcxgb4.h
index 94700db03e3f..343b65d3dcc1 100644
--- a/contrib/ofed/libcxgb4/libcxgb4.h
+++ b/contrib/ofed/libcxgb4/libcxgb4.h
@@ -62,6 +62,7 @@ struct c4iw_dev {
pthread_spinlock_t lock;
TAILQ_ENTRY(c4iw_dev) list;
int abi_version;
+ bool write_cmpl_supported;
};
static inline int dev_is_t7(struct c4iw_dev *dev)
diff --git a/contrib/ofed/libcxgb4/qp.c b/contrib/ofed/libcxgb4/qp.c
index 36531af67ae8..7f77de77ae9d 100644
--- a/contrib/ofed/libcxgb4/qp.c
+++ b/contrib/ofed/libcxgb4/qp.c
@@ -130,20 +130,29 @@ static int build_immd(struct t4_sq *sq, struct fw_ri_immd *immdp,
return 0;
}
-static int build_isgl(struct fw_ri_isgl *isglp, struct ibv_sge *sg_list,
+static int build_isgl(__be64 *queue_start, __be64 *queue_end,
+ struct fw_ri_isgl *isglp, struct ibv_sge *sg_list,
int num_sge, u32 *plenp)
{
int i;
u32 plen = 0;
- __be64 *flitp = (__be64 *)isglp->sge;
+ __be64 *flitp;
+ if ((__be64 *)isglp == queue_end)
+ isglp = (struct fw_ri_isgl *)queue_start;
+
+ flitp = (__be64 *)isglp->sge;
for (i = 0; i < num_sge; i++) {
if ((plen + sg_list[i].length) < plen)
return -EMSGSIZE;
plen += sg_list[i].length;
- *flitp++ = htobe64(((u64)sg_list[i].lkey << 32) |
- sg_list[i].length);
- *flitp++ = htobe64(sg_list[i].addr);
+ *flitp = htobe64(((u64)sg_list[i].lkey << 32) |
+ sg_list[i].length);
+ if (++flitp == queue_end)
+ flitp = queue_start;
+ *flitp = htobe64(sg_list[i].addr);
+ if (++flitp == queue_end)
+ flitp = queue_start;
}
*flitp = 0;
isglp->op = FW_RI_DATA_ISGL;
@@ -199,7 +208,9 @@ static int build_rdma_send(struct t4_sq *sq, union t4_wr *wqe,
size = sizeof wqe->send + sizeof(struct fw_ri_immd) +
plen;
} else {
- ret = build_isgl(wqe->send.u.isgl_src,
+ ret = build_isgl((__be64 *)sq->queue,
+ (__be64 *)&sq->queue[sq->size],
+ wqe->send.u.isgl_src,
wr->sg_list, wr->num_sge, &plen);
if (ret)
return ret;
@@ -243,7 +254,9 @@ static int build_rdma_write(struct t4_sq *sq, union t4_wr *wqe,
size = sizeof wqe->write + sizeof(struct fw_ri_immd) +
plen;
} else {
- ret = build_isgl(wqe->write.u.isgl_src,
+ ret = build_isgl((__be64 *)sq->queue,
+ (__be64 *)&sq->queue[sq->size],
+ wqe->write.u.isgl_src,
wr->sg_list, wr->num_sge, &plen);
if (ret)
return ret;
@@ -263,6 +276,64 @@ static int build_rdma_write(struct t4_sq *sq, union t4_wr *wqe,
return 0;
}
+static void build_immd_cmpl(struct t4_sq *sq, struct fw_ri_immd_cmpl *immdp,
+ struct ibv_send_wr *wr)
+{
+ memcpy((u8 *)immdp->data, (u8 *)(uintptr_t)wr->sg_list->addr, 16);
+ memset(immdp->r1, 0, 6);
+ immdp->op = FW_RI_DATA_IMMD;
+ immdp->immdlen = 16;
+ return;
+}
+
+#define BUILD_BUG_ON(condition) ((void)sizeof(char[1 - 2*!!(condition)]))
+
+static void build_rdma_write_cmpl(struct t4_sq *sq,
+ struct fw_ri_rdma_write_cmpl_wr *wcwr,
+ struct ibv_send_wr *wr, u8 *len16)
+{
+ u32 plen;
+ int size;
+
+ /*
+ * This code assumes the struct fields preceeding the write isgl
+ * fit in one 64B WR slot. This is because the WQE is built
+ * directly in the dma queue, and wrapping is only handled
+ * by the code buildling sgls. IE the "fixed part" of the wr
+ * structs must all fit in 64B. The WQE build code should probably be
+ * redesigned to avoid this restriction, but for now just add
+ * the BUILD_BUG_ON() to catch if this WQE struct gets too big.
+ */
+ BUILD_BUG_ON(offsetof(struct fw_ri_rdma_write_cmpl_wr, u) > 64);
+
+ wcwr->stag_sink = htobe32(wr->wr.rdma.rkey);
+ wcwr->to_sink = htobe64(wr->wr.rdma.remote_addr);
+ if (wr->next->opcode == IBV_WR_SEND)
+ wcwr->stag_inv = 0;
+ else
+ wcwr->stag_inv = htobe32(wr->next->invalidate_rkey);
+ wcwr->r2 = 0;
+ wcwr->r3 = 0;
+
+ /* SEND_INV SGL */
+ if (wr->next->send_flags & IBV_SEND_INLINE)
+ build_immd_cmpl(sq, &wcwr->u_cmpl.immd_src, wr->next);
+ else
+ build_isgl((__be64 *)sq->queue, (__be64 *)&sq->queue[sq->size],
+ &wcwr->u_cmpl.isgl_src, wr->next->sg_list, 1, NULL);
+
+ /* WRITE SGL */
+ build_isgl((__be64 *)sq->queue, (__be64 *)&sq->queue[sq->size],
+ wcwr->u.isgl_src, wr->sg_list, wr->num_sge, &plen);
+
+ size = sizeof *wcwr + sizeof(struct fw_ri_isgl) +
+ wr->num_sge * sizeof(struct fw_ri_sge);
+ wcwr->plen = htobe32(plen);
+ *len16 = DIV_ROUND_UP(size, 16);
+
+ return;
+}
+
static int build_rdma_read(union t4_wr *wqe, struct ibv_send_wr *wr, u8 *len16)
{
if (wr->num_sge > 1)
@@ -290,12 +361,13 @@ static int build_rdma_read(union t4_wr *wqe, struct ibv_send_wr *wr, u8 *len16)
return 0;
}
-static int build_rdma_recv(struct c4iw_qp *qhp, union t4_recv_wr *wqe,
+static int build_rdma_recv(struct t4_rq *rq, union t4_recv_wr *wqe,
struct ibv_recv_wr *wr, u8 *len16)
{
int ret;
- ret = build_isgl(&wqe->recv.isgl, wr->sg_list, wr->num_sge, NULL);
+ ret = build_isgl((__be64 *)rq->queue, (__be64 *)&rq->queue[rq->size],
+ &wqe->recv.isgl, wr->sg_list, wr->num_sge, NULL);
if (ret)
return ret;
*len16 = DIV_ROUND_UP(sizeof wqe->recv +
@@ -324,6 +396,67 @@ static void ring_kernel_db(struct c4iw_qp *qhp, u32 qid, u16 idx)
assert(!ret);
}
+static void post_write_cmpl(struct c4iw_qp *qhp, struct ibv_send_wr *wr)
+{
+ bool send_signaled = (wr->next->send_flags & IBV_SEND_SIGNALED) ||
+ qhp->sq_sig_all;
+ bool write_signaled = (wr->send_flags & IBV_SEND_SIGNALED) ||
+ qhp->sq_sig_all;
+ struct t4_swsqe *swsqe;
+ union t4_wr *wqe;
+ u16 write_wrid;
+ u8 len16;
+ u16 idx;
+
+ /*
+ * The sw_sq entries still look like a WRITE and a SEND and consume
+ * 2 slots. The FW WR, however, will be a single uber-WR.
+ */
+ wqe = (union t4_wr *)((u8 *)qhp->wq.sq.queue +
+ qhp->wq.sq.wq_pidx * T4_EQ_ENTRY_SIZE);
+ build_rdma_write_cmpl(&qhp->wq.sq, &wqe->write_cmpl, wr, &len16);
+
+ /* WRITE swsqe */
+ swsqe = &qhp->wq.sq.sw_sq[qhp->wq.sq.pidx];
+ swsqe->opcode = FW_RI_RDMA_WRITE;
+ swsqe->idx = qhp->wq.sq.pidx;
+ swsqe->complete = 0;
+ swsqe->signaled = write_signaled;
+ swsqe->flushed = 0;
+ swsqe->wr_id = wr->wr_id;
+
+ write_wrid = qhp->wq.sq.pidx;
+
+ /* just bump the sw_sq */
+ qhp->wq.sq.in_use++;
+ if (++qhp->wq.sq.pidx == qhp->wq.sq.size)
+ qhp->wq.sq.pidx = 0;
+
+ /* SEND swsqe */
+ swsqe = &qhp->wq.sq.sw_sq[qhp->wq.sq.pidx];
+ if (wr->next->opcode == IBV_WR_SEND)
+ swsqe->opcode = FW_RI_SEND;
+ else
+ swsqe->opcode = FW_RI_SEND_WITH_INV;
+ swsqe->idx = qhp->wq.sq.pidx;
+ swsqe->complete = 0;
+ swsqe->signaled = send_signaled;
+ swsqe->flushed = 0;
+ swsqe->wr_id = wr->next->wr_id;
+
+ wqe->write_cmpl.flags_send = send_signaled ? FW_RI_COMPLETION_FLAG : 0;
+ wqe->write_cmpl.wrid_send = qhp->wq.sq.pidx;
+
+ init_wr_hdr(wqe, write_wrid, FW_RI_RDMA_WRITE_CMPL_WR,
+ write_signaled ? FW_RI_COMPLETION_FLAG : 0, len16);
+ t4_sq_produce(&qhp->wq, len16);
+ idx = DIV_ROUND_UP(len16*16, T4_EQ_ENTRY_SIZE);
+
+ t4_ring_sq_db(&qhp->wq, idx, dev_is_t4(qhp->rhp),
+ len16, wqe);
+ return;
+}
+
int c4iw_post_send(struct ibv_qp *ibqp, struct ibv_send_wr *wr,
struct ibv_send_wr **bad_wr)
{
@@ -350,6 +483,30 @@ int c4iw_post_send(struct ibv_qp *ibqp, struct ibv_send_wr *wr,
*bad_wr = wr;
return -ENOMEM;
}
+
+ /*
+ * Fastpath for NVMe-oF target WRITE + SEND_WITH_INV wr chain which is
+ * the response for small NVMEe-oF READ requests. If the chain is
+ * exactly a WRITE->SEND_WITH_INV or a WRITE->SEND and the sgl depths
+ * and lengths meet the requirements of the fw_ri_write_cmpl_wr work
+ * request, then build and post the write_cmpl WR. If any of the tests
+ * below are not true, then we continue on with the tradtional WRITE
+ * and SEND WRs.
+ */
+ if (qhp->rhp->write_cmpl_supported &&
+ qhp->rhp->chip_version >= CHELSIO_T5 &&
+ wr && wr->next && !wr->next->next &&
+ wr->opcode == IBV_WR_RDMA_WRITE && wr->sg_list[0].length &&
+ wr->num_sge <= T4_WRITE_CMPL_MAX_SGL &&
+ (wr->next->opcode == IBV_WR_SEND_WITH_INV ||
+ wr->next->opcode == IBV_WR_SEND) &&
+ wr->next->sg_list[0].length == T4_WRITE_CMPL_MAX_CQE &&
+ wr->next->num_sge == 1 && num_wrs >= 2) {
+ post_write_cmpl(qhp, wr);
+ pthread_spin_unlock(&qhp->lock);
+ return 0;
+ }
+
while (wr) {
if (num_wrs == 0) {
err = -ENOMEM;
@@ -475,7 +632,7 @@ int c4iw_post_receive(struct ibv_qp *ibqp, struct ibv_recv_wr *wr,
}
wqe = &lwqe;
if (num_wrs)
- err = build_rdma_recv(qhp, wqe, wr, &len16);
+ err = build_rdma_recv(&qhp->wq.rq, wqe, wr, &len16);
else
err = -ENOMEM;
if (err) {
diff --git a/contrib/ofed/libcxgb4/t4.h b/contrib/ofed/libcxgb4/t4.h
index 921af87424fe..8027fdd39f8c 100644
--- a/contrib/ofed/libcxgb4/t4.h
+++ b/contrib/ofed/libcxgb4/t4.h
@@ -121,6 +121,9 @@ struct t4_status_page {
#define T4_RQ_NUM_BYTES (T4_EQ_ENTRY_SIZE * T4_RQ_NUM_SLOTS)
#define T4_MAX_RECV_SGE 4
+#define T4_WRITE_CMPL_MAX_SGL 4
+#define T4_WRITE_CMPL_MAX_CQE 16
+
union t4_wr {
struct fw_ri_res_wr res;
struct fw_ri_wr init;
@@ -130,9 +133,10 @@ union t4_wr {
struct fw_ri_bind_mw_wr bind;
struct fw_ri_fr_nsmr_wr fr;
struct fw_ri_inv_lstag_wr inv;
+ struct fw_ri_rdma_write_cmpl_wr write_cmpl;
struct t4_status_page status;
__be64 flits[T4_EQ_ENTRY_SIZE / sizeof(__be64) * T4_SQ_NUM_SLOTS];
-};
+} __attribute__((aligned(T4_EQ_ENTRY_SIZE)));
union t4_recv_wr {
struct fw_ri_recv_wr recv;
@@ -733,7 +737,8 @@ struct t4_dev_status_page
{
u8 db_off;
u8 wc_supported;
- u16 pad2;
+ u8 write_cmpl_supported;
+ u8 pad2;
u32 pad3;
u64 qp_start;
u64 qp_size;
diff --git a/sys/dev/cxgbe/iw_cxgbe/device.c b/sys/dev/cxgbe/iw_cxgbe/device.c
index 4610f91e96ac..9e5ac7c530d0 100644
--- a/sys/dev/cxgbe/iw_cxgbe/device.c
+++ b/sys/dev/cxgbe/iw_cxgbe/device.c
@@ -159,6 +159,8 @@ c4iw_rdev_open(struct c4iw_rdev *rdev)
rdev->status_page->db_off = 0;
rdev->status_page->wc_supported = rdev->adap->iwt.wc_en;
+ rdev->status_page->write_cmpl_supported =
+ rdev->adap->params.write_cmpl_support;
rdev->free_workq = create_singlethread_workqueue("iw_cxgb4_free");
if (!rdev->free_workq) {
diff --git a/sys/dev/cxgbe/iw_cxgbe/qp.c b/sys/dev/cxgbe/iw_cxgbe/qp.c
index 16aed5e423f0..5d4b11faf3cf 100644
--- a/sys/dev/cxgbe/iw_cxgbe/qp.c
+++ b/sys/dev/cxgbe/iw_cxgbe/qp.c
@@ -381,7 +381,12 @@ static int build_isgl(__be64 *queue_start, __be64 *queue_end,
{
int i;
u32 plen = 0;
- __be64 *flitp = (__be64 *)isglp->sge;
+ __be64 *flitp;
+
+ if ((__be64 *)isglp == queue_end)
+ isglp = (struct fw_ri_isgl *)queue_start;
+
+ flitp = (__be64 *)isglp->sge;
for (i = 0; i < num_sge; i++) {
if ((plen + sg_list[i].length) < plen)
@@ -518,6 +523,62 @@ static int build_rdma_write(struct t4_sq *sq, union t4_wr *wqe,
return 0;
}
+static void build_immd_cmpl(struct t4_sq *sq, struct fw_ri_immd_cmpl *immdp,
+ struct ib_send_wr *wr)
+{
+ memcpy((u8 *)immdp->data, (u8 *)(uintptr_t)wr->sg_list->addr, 16);
+ memset(immdp->r1, 0, 6);
+ immdp->op = FW_RI_DATA_IMMD;
+ immdp->immdlen = 16;
+ return;
+}
+
+static void build_rdma_write_cmpl(struct t4_sq *sq,
+ struct fw_ri_rdma_write_cmpl_wr *wcwr,
+ const struct ib_send_wr *wr, u8 *len16)
+{
+ u32 plen;
+ int size;
+
+ /*
+ * This code assumes the struct fields preceeding the write isgl
+ * fit in one 64B WR slot. This is because the WQE is built
+ * directly in the dma queue, and wrapping is only handled
+ * by the code buildling sgls. IE the "fixed part" of the wr
+ * structs must all fit in 64B. The WQE build code should probably be
+ * redesigned to avoid this restriction, but for now just add
+ * the BUILD_BUG_ON() to catch if this WQE struct gets too big.
+ */
+ BUILD_BUG_ON(offsetof(struct fw_ri_rdma_write_cmpl_wr, u) > 64);
+
+ wcwr->stag_sink = cpu_to_be32(rdma_wr(wr)->rkey);
+ wcwr->to_sink = cpu_to_be64(rdma_wr(wr)->remote_addr);
+ if (wr->next->opcode == IB_WR_SEND)
+ wcwr->stag_inv = 0;
+ else
+ wcwr->stag_inv = cpu_to_be32(wr->next->ex.invalidate_rkey);
+ wcwr->r2 = 0;
+ wcwr->r3 = 0;
+
+ /* SEND_INV SGL */
+ if (wr->next->send_flags & IB_SEND_INLINE)
+ build_immd_cmpl(sq, &wcwr->u_cmpl.immd_src, wr->next);
+ else
+ build_isgl((__be64 *)sq->queue, (__be64 *)&sq->queue[sq->size],
+ &wcwr->u_cmpl.isgl_src, wr->next->sg_list, 1, NULL);
+
+ /* WRITE SGL */
+ build_isgl((__be64 *)sq->queue, (__be64 *)&sq->queue[sq->size],
+ wcwr->u.isgl_src, wr->sg_list, wr->num_sge, &plen);
+
+ size = sizeof *wcwr + sizeof(struct fw_ri_isgl) +
+ wr->num_sge * sizeof(struct fw_ri_sge);
+ wcwr->plen = cpu_to_be32(plen);
+ *len16 = DIV_ROUND_UP(size, 16);
+
+ return;
+}
+
static int build_rdma_read(union t4_wr *wqe, const struct ib_send_wr *wr, u8 *len16)
{
if (wr->num_sge > 1)
@@ -671,6 +732,66 @@ static void complete_rq_drain_wr(struct c4iw_qp *qhp, const struct ib_recv_wr *w
spin_unlock_irqrestore(&rchp->comp_handler_lock, flag);
}
+static void post_write_cmpl(struct c4iw_qp *qhp, const struct ib_send_wr *wr)
+{
+ bool send_signaled = (wr->next->send_flags & IB_SEND_SIGNALED) ||
+ qhp->sq_sig_all;
+ bool write_signaled = (wr->send_flags & IB_SEND_SIGNALED) ||
+ qhp->sq_sig_all;
+ struct t4_swsqe *swsqe;
+ union t4_wr *wqe;
+ u16 write_wrid;
+ u8 len16;
+ u16 idx;
+
+ /*
+ * The sw_sq entries still look like a WRITE and a SEND and consume
+ * 2 slots. The FW WR, however, will be a single uber-WR.
+ */
+ wqe = (union t4_wr *)((u8 *)qhp->wq.sq.queue +
+ qhp->wq.sq.wq_pidx * T4_EQ_ENTRY_SIZE);
+ build_rdma_write_cmpl(&qhp->wq.sq, &wqe->write_cmpl, wr, &len16);
+
+ /* WRITE swsqe */
+ swsqe = &qhp->wq.sq.sw_sq[qhp->wq.sq.pidx];
+ swsqe->opcode = FW_RI_RDMA_WRITE;
+ swsqe->idx = qhp->wq.sq.pidx;
+ swsqe->complete = 0;
+ swsqe->signaled = write_signaled;
+ swsqe->flushed = 0;
+ swsqe->wr_id = wr->wr_id;
+
+ write_wrid = qhp->wq.sq.pidx;
+
+ /* just bump the sw_sq */
+ qhp->wq.sq.in_use++;
+ if (++qhp->wq.sq.pidx == qhp->wq.sq.size)
+ qhp->wq.sq.pidx = 0;
+
+ /* SEND_WITH_INV swsqe */
+ swsqe = &qhp->wq.sq.sw_sq[qhp->wq.sq.pidx];
+ if (wr->next->opcode == IB_WR_SEND)
+ swsqe->opcode = FW_RI_SEND;
+ else
+ swsqe->opcode = FW_RI_SEND_WITH_INV;
+ swsqe->idx = qhp->wq.sq.pidx;
+ swsqe->complete = 0;
+ swsqe->signaled = send_signaled;
+ swsqe->flushed = 0;
+ swsqe->wr_id = wr->next->wr_id;
+
+ wqe->write_cmpl.flags_send = send_signaled ? FW_RI_COMPLETION_FLAG : 0;
+ wqe->write_cmpl.wrid_send = qhp->wq.sq.pidx;
+
+ init_wr_hdr(wqe, write_wrid, FW_RI_RDMA_WRITE_CMPL_WR,
+ write_signaled ? FW_RI_COMPLETION_FLAG : 0, len16);
+ t4_sq_produce(&qhp->wq, len16);
+ idx = DIV_ROUND_UP(len16*16, T4_EQ_ENTRY_SIZE);
+
+ t4_ring_sq_db(&qhp->wq, idx, wqe, qhp->rhp->rdev.adap->iwt.wc_en);
+ return;
+}
+
static int build_tpte_memreg(struct fw_ri_fr_nsmr_tpte_wr *fr,
const struct ib_reg_wr *wr, struct c4iw_mr *mhp, u8 *len16)
{
@@ -805,6 +926,30 @@ int c4iw_post_send(struct ib_qp *ibqp, const struct ib_send_wr *wr,
*bad_wr = wr;
return -ENOMEM;
}
+
+ /*
+ * Fastpath for NVMe-oF target WRITE + SEND_WITH_INV wr chain which is
+ * the response for small NVMEe-oF READ requests. If the chain is
+ * exactly a WRITE->SEND_WITH_INV or a WRITE->SEND and the sgl depths
+ * and lengths meet the requirements of the fw_ri_write_cmpl_wr work
+ * request, then build and post the write_cmpl WR. If any of the tests
+ * below are not true, then we continue on with the tradtional WRITE
+ * and SEND WRs.
+ */
+ if (qhp->rhp->rdev.adap->params.write_cmpl_support &&
+ chip_id(qhp->rhp->rdev.adap) >=
+ CHELSIO_T5 &&
+ wr && wr->next && !wr->next->next &&
+ wr->opcode == IB_WR_RDMA_WRITE &&
+ wr->sg_list[0].length && wr->num_sge <= T4_WRITE_CMPL_MAX_SGL &&
+ (wr->next->opcode == IB_WR_SEND ||
+ wr->next->opcode == IB_WR_SEND_WITH_INV) &&
+ wr->next->num_sge == 1 && num_wrs >= 2) {
+ post_write_cmpl(qhp, wr);
+ spin_unlock_irqrestore(&qhp->lock, flag);
+ return 0;
+ }
+
while (wr) {
if (num_wrs == 0) {
err = -ENOMEM;
diff --git a/sys/dev/cxgbe/iw_cxgbe/t4.h b/sys/dev/cxgbe/iw_cxgbe/t4.h
index 1401883a21a7..230076e5e97d 100644
--- a/sys/dev/cxgbe/iw_cxgbe/t4.h
+++ b/sys/dev/cxgbe/iw_cxgbe/t4.h
@@ -107,6 +107,8 @@ struct t4_status_page {
#define T4_RQ_NUM_BYTES (T4_EQ_ENTRY_SIZE * T4_RQ_NUM_SLOTS)
#define T4_MAX_RECV_SGE 4
+#define T4_WRITE_CMPL_MAX_SGL 4
+
union t4_wr {
struct fw_ri_res_wr res;
struct fw_ri_wr ri;
@@ -117,6 +119,7 @@ union t4_wr {
struct fw_ri_fr_nsmr_wr fr;
struct fw_ri_fr_nsmr_tpte_wr fr_tpte;
struct fw_ri_inv_lstag_wr inv;
+ struct fw_ri_rdma_write_cmpl_wr write_cmpl;
struct t4_status_page status;
__be64 flits[T4_EQ_ENTRY_SIZE / sizeof(__be64) * T4_SQ_NUM_SLOTS];
};
@@ -717,7 +720,8 @@ static inline void t4_set_cq_in_error(struct t4_cq *cq)
struct t4_dev_status_page {
u8 db_off;
u8 wc_supported;
- u16 pad2;
+ u8 write_cmpl_supported;
+ u8 pad2;
u32 pad3;
u64 qp_start;
u64 qp_size;