git: 6262cbed3c37 - main - libcxgb4: Support send/recv drain work requests

From: John Baldwin <jhb_at_FreeBSD.org>
Date: Tue, 15 Sep 2026 14:47:48 UTC
The branch main has been updated by jhb:

URL: https://cgit.FreeBSD.org/src/commit/?id=6262cbed3c37aff6ee9ff4dad7b14147bb40ac7e

commit 6262cbed3c37aff6ee9ff4dad7b14147bb40ac7e
Author:     Potnuri Bharat Teja <bharat@chelsio.com>
AuthorDate: 2018-12-17 18:09:41 +0000
Commit:     John Baldwin <jhb@FreeBSD.org>
CommitDate: 2026-09-15 14:22:39 +0000

    libcxgb4: Support send/recv drain work requests
    
    This change allows libcxgb4 to post send/recv drain work requests
    after a QP is flushed and schedules a CQE on the software CQ for the
    application to note the drain's completion.
    
    Sponsored by:   Chelsio Communications
---
 contrib/ofed/libcxgb4/cq.c |  15 ++++++
 contrib/ofed/libcxgb4/qp.c | 132 ++++++++++++++++++++++++++++++++++++++++++---
 contrib/ofed/libcxgb4/t4.h |  14 +++++
 3 files changed, 155 insertions(+), 6 deletions(-)

diff --git a/contrib/ofed/libcxgb4/cq.c b/contrib/ofed/libcxgb4/cq.c
index a5ce335d7e60..b03a52a59159 100644
--- a/contrib/ofed/libcxgb4/cq.c
+++ b/contrib/ofed/libcxgb4/cq.c
@@ -283,6 +283,12 @@ next_cqe:
 
 static int cqe_completes_wr(struct t4_cqe *cqe, struct t4_wq *wq)
 {
+	if (DRAIN_CQE(cqe)) {
+		fprintf(stderr, "Unexpected DRAIN CQE qp id %u!\n",
+			wq->sq.qid);
+		return 0;
+	}
+
 	if (CQE_OPCODE(cqe) == FW_RI_TERMINATE)
 		return 0;
 
@@ -370,6 +376,15 @@ static int poll_cq(struct t4_wq *wq, struct t4_cq *cq, struct t4_cqe *cqe,
 		goto skip_cqe;
 	}
 
+	/*
+	 * Special cqe for drain WR completions...
+	 */
+	if (DRAIN_CQE(hw_cqe)) {
+		*cookie = CQE_DRAIN_COOKIE(hw_cqe);
+		*cqe = *hw_cqe;
+		goto skip_cqe;
+	}
+
 	/*
 	 * Gotta tweak READ completions:
 	 *	1) the cqe doesn't contain the sq_wptr from the wr.
diff --git a/contrib/ofed/libcxgb4/qp.c b/contrib/ofed/libcxgb4/qp.c
index 7f77de77ae9d..a41381bf4907 100644
--- a/contrib/ofed/libcxgb4/qp.c
+++ b/contrib/ofed/libcxgb4/qp.c
@@ -396,6 +396,116 @@ static void ring_kernel_db(struct c4iw_qp *qhp, u32 qid, u16 idx)
 	assert(!ret);
 }
 
+static int ibv_to_fw_opcode(int ib_opcode)
+{
+	int opcode;
+
+	switch (ib_opcode) {
+	case IBV_WR_SEND_WITH_INV:
+		opcode = FW_RI_SEND_WITH_INV;
+		break;
+	case IBV_WR_SEND:
+		opcode = FW_RI_SEND;
+		break;
+	case IBV_WR_RDMA_WRITE:
+		opcode = FW_RI_RDMA_WRITE;
+		break;
+	case IBV_WR_RDMA_WRITE_WITH_IMM:
+		opcode = FW_RI_WRITE_IMMEDIATE;
+		break;
+	case IBV_WR_RDMA_READ:
+		opcode = FW_RI_READ_REQ;
+		break;
+	default:
+		opcode = -EINVAL;
+	}
+	return opcode;
+}
+
+static int complete_sq_drain_wr(struct c4iw_qp *qhp, struct ibv_send_wr *wr)
+{
+	struct t4_cqe cqe = {};
+	struct c4iw_cq *schp;
+	struct t4_cq *cq;
+	int opcode;
+
+	schp = to_c4iw_cq(qhp->ibv_qp.send_cq);
+	cq = &schp->cq;
+
+	opcode = ibv_to_fw_opcode(wr->opcode);
+	if (opcode < 0)
+		return opcode;
+
+	PDBG("drain sq id %u\n", qhp->wq.sq.qid);
+	cqe.u.drain_cookie = wr->wr_id;
+	cqe.header = htobe32(V_CQE_STATUS(T4_ERR_SWFLUSH) |
+				 V_CQE_OPCODE(opcode) |
+				 V_CQE_TYPE(1) |
+				 V_CQE_SWCQE(1) |
+				 V_CQE_DRAIN(1) |
+				 V_CQE_QPID(qhp->wq.sq.qid));
+
+	pthread_spin_lock(&schp->lock);
+	cqe.bits_type_ts = htobe64(V_CQE_GENBIT((u64)cq->gen));
+	cq->sw_queue[cq->sw_pidx] = cqe;
+	t4_swcq_produce(cq);
+	pthread_spin_unlock(&schp->lock);
+
+	t4_clear_cq_armed(&schp->cq);
+	return 0;
+}
+
+static int complete_sq_drain_wrs(struct c4iw_qp *qhp, struct ibv_send_wr *wr,
+				 struct ibv_send_wr **bad_wr)
+{
+	int ret = 0;
+
+	while (wr) {
+		ret = complete_sq_drain_wr(qhp, wr);
+		if (ret) {
+			*bad_wr = wr;
+			break;
+		}
+		wr = wr->next;
+	}
+	return ret;
+}
+
+static void complete_rq_drain_wr(struct c4iw_qp *qhp, struct ibv_recv_wr *wr)
+{
+	struct t4_cqe cqe = {};
+	struct c4iw_cq *rchp;
+	struct t4_cq *cq;
+
+	rchp = to_c4iw_cq(qhp->ibv_qp.recv_cq);
+	cq = &rchp->cq;
+
+	PDBG("drain rq id %u\n", qhp->wq.sq.qid);
+	cqe.u.drain_cookie = wr->wr_id;
+	cqe.header = htobe32(V_CQE_STATUS(T4_ERR_SWFLUSH) |
+				 V_CQE_OPCODE(FW_RI_SEND) |
+				 V_CQE_TYPE(0) |
+				 V_CQE_SWCQE(1) |
+				 V_CQE_DRAIN(1) |
+				 V_CQE_QPID(qhp->wq.sq.qid));
+
+	pthread_spin_lock(&rchp->lock);
+	cqe.bits_type_ts = htobe64(V_CQE_GENBIT((u64)cq->gen));
+	cq->sw_queue[cq->sw_pidx] = cqe;
+	t4_swcq_produce(cq);
+	pthread_spin_unlock(&rchp->lock);
+
+	t4_clear_cq_armed(&rchp->cq);
+}
+
+static void complete_rq_drain_wrs(struct c4iw_qp *qhp, struct ibv_recv_wr *wr)
+{
+	while (wr) {
+		complete_rq_drain_wr(qhp, wr);
+		wr = wr->next;
+	}
+}
+
 static void post_write_cmpl(struct c4iw_qp *qhp, struct ibv_send_wr *wr)
 {
 	bool send_signaled = (wr->next->send_flags & IBV_SEND_SIGNALED) ||
@@ -472,10 +582,15 @@ int c4iw_post_send(struct ibv_qp *ibqp, struct ibv_send_wr *wr,
 
 	qhp = to_c4iw_qp(ibqp);
 	pthread_spin_lock(&qhp->lock);
-	if (t4_wq_in_error(&qhp->wq)) {
+
+	/*
+	 * If the qp has been flushed, then just insert a special
+	 * drain cqe.
+	 */
+	if (qhp->wq.flushed) {
 		pthread_spin_unlock(&qhp->lock);
-		*bad_wr = wr;
-		return -EINVAL;
+		err = complete_sq_drain_wrs(qhp, wr, bad_wr);
+		return err;
 	}
 	num_wrs = t4_sq_avail(&qhp->wq);
 	if (num_wrs == 0) {
@@ -612,10 +727,15 @@ int c4iw_post_receive(struct ibv_qp *ibqp, struct ibv_recv_wr *wr,
 
 	qhp = to_c4iw_qp(ibqp);
 	pthread_spin_lock(&qhp->lock);
-	if (t4_wq_in_error(&qhp->wq)) {
+
+	/*
+	 * If the qp has been flushed, then just insert a special
+	 * drain cqe.
+	 */
+	if (qhp->wq.flushed) {
 		pthread_spin_unlock(&qhp->lock);
-		*bad_wr = wr;
-		return -EINVAL;
+		complete_rq_drain_wrs(qhp, wr);
+		return err;
 	}
 	INC_STAT(recv);
 	num_wrs = t4_rq_avail(&qhp->wq);
diff --git a/contrib/ofed/libcxgb4/t4.h b/contrib/ofed/libcxgb4/t4.h
index 8027fdd39f8c..34f8db74e4e9 100644
--- a/contrib/ofed/libcxgb4/t4.h
+++ b/contrib/ofed/libcxgb4/t4.h
@@ -100,6 +100,7 @@ struct t4_status_page {
 	__be16 pidx;
 	u8 qp_err;	/* flit 1 - sw owns */
 	u8 db_off;
+	u8 cq_armed;
 	u8 pad;
 	u16 host_wq_pidx;
 	u16 host_cidx;
@@ -217,6 +218,7 @@ struct t4_cqe {
 			__be32 wrid_hi;
 			__be32 wrid_low;
 		} gen;
+		__u64 drain_cookie;
 		struct {
 			__be32 mo;
 			__be32 msn;
@@ -239,6 +241,11 @@ struct t4_cqe {
 #define G_CQE_SWCQE(x)    ((((x) >> S_CQE_SWCQE)) & M_CQE_SWCQE)
 #define V_CQE_SWCQE(x)	  ((x)<<S_CQE_SWCQE)
 
+#define S_CQE_DRAIN       10
+#define M_CQE_DRAIN       0x1
+#define G_CQE_DRAIN(x)    ((((x) >> S_CQE_DRAIN)) & M_CQE_DRAIN)
+#define V_CQE_DRAIN(x)    ((x)<<S_CQE_DRAIN)
+
 #define S_CQE_STATUS      5
 #define M_CQE_STATUS      0x1F
 #define G_CQE_STATUS(x)   ((((x) >> S_CQE_STATUS)) & M_CQE_STATUS)
@@ -255,6 +262,7 @@ struct t4_cqe {
 #define V_CQE_OPCODE(x)   ((x)<<S_CQE_OPCODE)
 
 #define SW_CQE(x)         (G_CQE_SWCQE(be32toh((x)->header)))
+#define DRAIN_CQE(x)      (G_CQE_DRAIN(be32toh((x)->header)))
 #define CQE_QPID(x)       (G_CQE_QPID(be32toh((x)->header)))
 #define CQE_TYPE(x)       (G_CQE_TYPE(be32toh((x)->header)))
 #define SQ_TYPE(x)	  (CQE_TYPE((x)))
@@ -281,6 +289,7 @@ struct t4_cqe {
 /* generic accessor macros */
 #define CQE_WRID_HI(x)		((x)->u.gen.wrid_hi)
 #define CQE_WRID_LOW(x)		((x)->u.gen.wrid_low)
+#define CQE_DRAIN_COOKIE(x)    ((x)->u.drain_cookie)
 
 /* macros for flit 3 of the cqe */
 #define S_CQE_GENBIT	63
@@ -604,6 +613,11 @@ struct t4_cq {
 	u8 error;
 };
 
+static inline void t4_clear_cq_armed(struct t4_cq *cq)
+{
+	((struct t4_status_page *)&cq->queue[cq->size])->cq_armed = 0;
+}
+
 static inline int t4_arm_cq(struct t4_cq *cq, int se)
 {
 	u32 val;