git: f08d90e47d92 - main - iw_cxgbe: Fixes around work queue flushing and draining

From: John Baldwin <jhb_at_FreeBSD.org>
Date: Tue, 15 Sep 2026 14:47:47 UTC
The branch main has been updated by jhb:

URL: https://cgit.FreeBSD.org/src/commit/?id=f08d90e47d9227f81ff76c28fc754315b0c405d8

commit f08d90e47d9227f81ff76c28fc754315b0c405d8
Author:     Potnuri Bharat Teja <bharat@chelsio.com>
AuthorDate: 2018-02-13 12:24:00 +0000
Commit:     John Baldwin <jhb@FreeBSD.org>
CommitDate: 2026-09-15 14:22:39 +0000

    iw_cxgbe: Fixes around work queue flushing and draining
    
    - Only insert drain CQEs if a work queue is flushed
    
    - Preserve the request opcode of the original request when flushing a
      request.  Use bit 10 of the CQE header word to indicate the CQE is a
      special drain completion, and save the original WR opcode in the cqe
      header opcode field.
    
    - If a work request chain was posted and needed to be flushed, only
      the first request in the chain was completed with FLUSHED status.
      The rest were never completed.
    
    - When a CQ is shared by multiple QPs, c4iw_flush_hw_cq() needs to
      acquire the corresponding QP lock before moving the CQEs into its
      corresponding SW queue and accessing the SQ contents for completing
      a WR.
    
    - Once a user QP has been flushed, it cannot be flushed again.
    
    Sponsored by:   Chelsio Communications
    Co-authored-by: Steve Wise <swise@opengridcomputing.com>
---
 sys/dev/cxgbe/iw_cxgbe/cq.c       | 22 ++++++---
 sys/dev/cxgbe/iw_cxgbe/iw_cxgbe.h |  4 +-
 sys/dev/cxgbe/iw_cxgbe/qp.c       | 99 +++++++++++++++++++++++++++++++++++----
 sys/dev/cxgbe/iw_cxgbe/t4.h       |  6 +++
 4 files changed, 113 insertions(+), 18 deletions(-)

diff --git a/sys/dev/cxgbe/iw_cxgbe/cq.c b/sys/dev/cxgbe/iw_cxgbe/cq.c
index 65351dbfaafb..bff2c4aeca01 100644
--- a/sys/dev/cxgbe/iw_cxgbe/cq.c
+++ b/sys/dev/cxgbe/iw_cxgbe/cq.c
@@ -357,7 +357,7 @@ static void advance_oldest_read(struct t4_wq *wq)
  * Deal with out-of-order and/or completions that complete
  * prior unsignalled WRs.
  */
-void c4iw_flush_hw_cq(struct c4iw_cq *chp)
+void c4iw_flush_hw_cq(struct c4iw_cq *chp, struct c4iw_qp *flush_qhp)
 {
 	struct t4_cqe *hw_cqe, *swcqe, read_cqe;
 	struct c4iw_qp *qhp;
@@ -382,6 +382,13 @@ void c4iw_flush_hw_cq(struct c4iw_cq *chp)
 		if (qhp == NULL)
 			goto next_cqe;
 
+		if (flush_qhp != qhp) {
+			spin_lock(&qhp->lock);
+
+			if (qhp->wq.flushed == 1)
+				goto next_cqe;
+		}
+
 		if (CQE_OPCODE(hw_cqe) == FW_RI_TERMINATE)
 			goto next_cqe;
 
@@ -433,11 +440,18 @@ void c4iw_flush_hw_cq(struct c4iw_cq *chp)
 next_cqe:
 		t4_hwcq_consume(&chp->cq);
 		ret = t4_next_hw_cqe(&chp->cq, &hw_cqe);
+		if (qhp && flush_qhp != qhp)
+			spin_unlock(&qhp->lock);
 	}
 }
 
 static int cqe_completes_wr(struct t4_cqe *cqe, struct t4_wq *wq)
 {
+	if (DRAIN_CQE(cqe)) {
+		WARN_ONCE(1, "Unexpected DRAIN CQE qp id %u!\n", wq->sq.qid);
+		return 0;
+	}
+
 	if (CQE_OPCODE(cqe) == FW_RI_TERMINATE)
 		return 0;
 
@@ -535,7 +549,7 @@ static int poll_cq(struct t4_wq *wq, struct t4_cq *cq, struct t4_cqe *cqe,
 	/*
 	 * Special cqe for drain WR completions...
 	 */
-	if (CQE_OPCODE(hw_cqe) == C4IW_DRAIN_OPCODE) {
+	if (DRAIN_CQE(hw_cqe)) {
 		*cookie = CQE_DRAIN_COOKIE(hw_cqe);
 		*cqe = *hw_cqe;
 		goto skip_cqe;
@@ -757,7 +771,6 @@ static int c4iw_poll_cq_one(struct c4iw_cq *chp, struct ib_wc *wc)
 
 		switch (CQE_OPCODE(&cqe)) {
 		case FW_RI_SEND:
-		case C4IW_DRAIN_OPCODE:
 			wc->opcode = IB_WC_RECV;
 			break;
 		case FW_RI_SEND_WITH_INV:
@@ -809,9 +822,6 @@ static int c4iw_poll_cq_one(struct c4iw_cq *chp, struct ib_wc *wc)
 				c4iw_invalidate_mr(qhp->rhp,
 						   CQE_WRID_FR_STAG(&cqe));
 			break;
-		case C4IW_DRAIN_OPCODE:
-			wc->opcode = IB_WC_SEND;
-			break;
 		default:
 			printf("Unexpected opcode %d "
 			       "in the CQE received for QPID = 0x%0x\n",
diff --git a/sys/dev/cxgbe/iw_cxgbe/iw_cxgbe.h b/sys/dev/cxgbe/iw_cxgbe/iw_cxgbe.h
index 47ce10562c66..5f3e54439a7f 100644
--- a/sys/dev/cxgbe/iw_cxgbe/iw_cxgbe.h
+++ b/sys/dev/cxgbe/iw_cxgbe/iw_cxgbe.h
@@ -636,8 +636,6 @@ static inline int to_ib_qp_state(int c4iw_qp_state)
 	return IB_QPS_ERR;
 }
 
-#define C4IW_DRAIN_OPCODE FW_RI_SGE_EC_CR_RETURN
-
 static inline u32 c4iw_ib_to_tpt_access(int a)
 {
 	return (a & IB_ACCESS_REMOTE_WRITE ? FW_RI_MEM_ACCESS_REM_WRITE : 0) |
@@ -961,7 +959,7 @@ void c4iw_rqtpool_free(struct c4iw_rdev *rdev, u32 addr, int size);
 u32 c4iw_pblpool_alloc(struct c4iw_rdev *rdev, int size);
 void c4iw_pblpool_free(struct c4iw_rdev *rdev, u32 addr, int size);
 int c4iw_ofld_send(struct c4iw_rdev *rdev, struct mbuf *m);
-void c4iw_flush_hw_cq(struct c4iw_cq *cq);
+void c4iw_flush_hw_cq(struct c4iw_cq *cq, struct c4iw_qp *flush_qhp);
 void c4iw_count_rcqes(struct t4_cq *cq, struct t4_wq *wq, int *count);
 int c4iw_ep_disconnect(struct c4iw_ep *ep, int abrupt, gfp_t gfp);
 int __c4iw_ep_disconnect(struct c4iw_ep *ep, int abrupt, gfp_t gfp);
diff --git a/sys/dev/cxgbe/iw_cxgbe/qp.c b/sys/dev/cxgbe/iw_cxgbe/qp.c
index 5d4b11faf3cf..5167fcdc9112 100644
--- a/sys/dev/cxgbe/iw_cxgbe/qp.c
+++ b/sys/dev/cxgbe/iw_cxgbe/qp.c
@@ -672,22 +672,61 @@ void c4iw_qp_rem_ref(struct ib_qp *qp)
 	kref_put(&to_c4iw_qp(qp)->kref, queue_qp_free);
 }
 
-static void complete_sq_drain_wr(struct c4iw_qp *qhp, const struct ib_send_wr *wr)
+static int ib_to_fw_opcode(int ib_opcode)
+{
+	int opcode;
+
+	switch (ib_opcode) {
+	case IB_WR_SEND_WITH_INV:
+		opcode = FW_RI_SEND_WITH_INV;
+		break;
+	case IB_WR_SEND:
+		opcode = FW_RI_SEND;
+		break;
+	case IB_WR_RDMA_WRITE:
+		opcode = FW_RI_RDMA_WRITE;
+		break;
+	case IB_WR_RDMA_WRITE_WITH_IMM:
+		opcode = FW_RI_WRITE_IMMEDIATE;
+		break;
+	case IB_WR_RDMA_READ:
+	case IB_WR_RDMA_READ_WITH_INV:
+		opcode = FW_RI_READ_REQ;
+		break;
+	case IB_WR_REG_MR:
+		opcode = FW_RI_FAST_REGISTER;
+		break;
+	case IB_WR_LOCAL_INV:
+		opcode = FW_RI_LOCAL_INV;
+		break;
+	default:
+		opcode = -EINVAL;
+	}
+	return opcode;
+}
+
+static int complete_sq_drain_wr(struct c4iw_qp *qhp, const struct ib_send_wr *wr)
 {
 	struct t4_cqe cqe = {};
 	struct c4iw_cq *schp;
 	unsigned long flag;
 	struct t4_cq *cq;
+	int opcode;
 
 	schp = to_c4iw_cq(qhp->ibqp.send_cq);
 	cq = &schp->cq;
 
+	opcode = ib_to_fw_opcode(wr->opcode);
+	if (opcode < 0)
+		return opcode;
+
 	PDBG("%s drain sq id %u\n", __func__, qhp->wq.sq.qid);
 	cqe.u.drain_cookie = wr->wr_id;
 	cqe.header = cpu_to_be32(V_CQE_STATUS(T4_ERR_SWFLUSH) |
-				 V_CQE_OPCODE(C4IW_DRAIN_OPCODE) |
+				 V_CQE_OPCODE(opcode) |
 				 V_CQE_TYPE(1) |
 				 V_CQE_SWCQE(1) |
+				 V_CQE_DRAIN(1) |
 				 V_CQE_QPID(qhp->wq.sq.qid));
 
 	spin_lock_irqsave(&schp->lock, flag);
@@ -700,6 +739,23 @@ static void complete_sq_drain_wr(struct c4iw_qp *qhp, const struct ib_send_wr *w
 	(*schp->ibcq.comp_handler)(&schp->ibcq,
 				   schp->ibcq.cq_context);
 	spin_unlock_irqrestore(&schp->comp_handler_lock, flag);
+	return 0;
+}
+
+static int complete_sq_drain_wrs(struct c4iw_qp *qhp, const struct ib_send_wr *wr,
+				 const struct ib_send_wr **bad_wr)
+{
+	int ret = 0;
+
+	while (wr) {
+		ret = complete_sq_drain_wr(qhp, wr);
+		if (ret) {
+			*bad_wr = wr;
+			break;
+		}
+		wr = wr->next;
+	}
+	return ret;
 }
 
 static void complete_rq_drain_wr(struct c4iw_qp *qhp, const struct ib_recv_wr *wr)
@@ -715,9 +771,10 @@ static void complete_rq_drain_wr(struct c4iw_qp *qhp, const struct ib_recv_wr *w
 	PDBG("%s drain rq id %u\n", __func__, qhp->wq.sq.qid);
 	cqe.u.drain_cookie = wr->wr_id;
 	cqe.header = cpu_to_be32(V_CQE_STATUS(T4_ERR_SWFLUSH) |
-				 V_CQE_OPCODE(C4IW_DRAIN_OPCODE) |
+				 V_CQE_OPCODE(FW_RI_SEND) |
 				 V_CQE_TYPE(0) |
 				 V_CQE_SWCQE(1) |
+				 V_CQE_DRAIN(1) |
 				 V_CQE_QPID(qhp->wq.sq.qid));
 
 	spin_lock_irqsave(&rchp->lock, flag);
@@ -732,6 +789,14 @@ static void complete_rq_drain_wr(struct c4iw_qp *qhp, const struct ib_recv_wr *w
 	spin_unlock_irqrestore(&rchp->comp_handler_lock, flag);
 }
 
+static void complete_rq_drain_wrs(struct c4iw_qp *qhp, const struct ib_recv_wr *wr)
+{
+	while (wr) {
+		complete_rq_drain_wr(qhp, wr);
+		wr = wr->next;
+	}
+}
+
 static void post_write_cmpl(struct c4iw_qp *qhp, const struct ib_send_wr *wr)
 {
 	bool send_signaled = (wr->next->send_flags & IB_SEND_SIGNALED) ||
@@ -915,9 +980,14 @@ int c4iw_post_send(struct ib_qp *ibqp, const struct ib_send_wr *wr,
 	if (__predict_false(c4iw_stopped(rdev)))
 		return -EIO;
 	spin_lock_irqsave(&qhp->lock, flag);
-	if (t4_wq_in_error(&qhp->wq)) {
+
+	/*
+	 * If the qp has been flushed, then just insert a special
+	 * drain cqe.
+	 */
+	if (qhp->wq.flushed) {
 		spin_unlock_irqrestore(&qhp->lock, flag);
-		complete_sq_drain_wr(qhp, wr);
+		err = complete_sq_drain_wrs(qhp, wr, bad_wr);
 		return err;
 	}
 	num_wrs = t4_sq_avail(&qhp->wq);
@@ -1083,9 +1153,14 @@ int c4iw_post_receive(struct ib_qp *ibqp, const struct ib_recv_wr *wr,
 	if (__predict_false(c4iw_stopped(&qhp->rhp->rdev)))
 		return -EIO;
 	spin_lock_irqsave(&qhp->lock, flag);
-	if (t4_wq_in_error(&qhp->wq)) {
+
+	/*
+	 * If the qp has been flushed, then just insert a special
+	 * drain cqe.
+	 */
+	if (qhp->wq.flushed) {
 		spin_unlock_irqrestore(&qhp->lock, flag);
-		complete_rq_drain_wr(qhp, wr);
+		complete_rq_drain_wrs(qhp, wr);
 		return err;
 	}
 	num_wrs = t4_rq_avail(&qhp->wq);
@@ -1334,7 +1409,7 @@ static void __flush_qp(struct c4iw_qp *qhp, struct c4iw_cq *rchp,
 	}
 	qhp->wq.flushed = 1;
 
-	c4iw_flush_hw_cq(rchp);
+	c4iw_flush_hw_cq(rchp, qhp);
 	c4iw_count_rcqes(&rchp->cq, &qhp->wq, &count);
 	rq_flushed = c4iw_flush_rq(&qhp->wq, &rchp->cq, count);
 	spin_unlock(&qhp->lock);
@@ -1344,7 +1419,7 @@ static void __flush_qp(struct c4iw_qp *qhp, struct c4iw_cq *rchp,
 	spin_lock_irqsave(&schp->lock, flag);
 	spin_lock(&qhp->lock);
 	if (schp != rchp)
-		c4iw_flush_hw_cq(schp);
+		c4iw_flush_hw_cq(schp, qhp);
 	sq_flushed = c4iw_flush_sq(qhp);
 	spin_unlock(&qhp->lock);
 	spin_unlock_irqrestore(&schp->lock, flag);
@@ -1383,6 +1458,12 @@ static void flush_qp(struct c4iw_qp *qhp)
 
 	t4_set_wq_in_error(&qhp->wq);
 	if (qhp->ibqp.uobject) {
+
+		/* qhp->wq.flush is protected by qhp->mutex */
+		if (qhp->wq.flushed)
+			return;
+
+		qhp->wq.flushed = 1;
 		t4_set_cq_in_error(&rchp->cq);
 		spin_lock_irqsave(&rchp->comp_handler_lock, flag);
 		(*rchp->ibcq.comp_handler)(&rchp->ibcq, rchp->ibcq.cq_context);
diff --git a/sys/dev/cxgbe/iw_cxgbe/t4.h b/sys/dev/cxgbe/iw_cxgbe/t4.h
index 230076e5e97d..88089213c22b 100644
--- a/sys/dev/cxgbe/iw_cxgbe/t4.h
+++ b/sys/dev/cxgbe/iw_cxgbe/t4.h
@@ -230,6 +230,11 @@ struct t4_cqe {
 #define G_CQE_SWCQE(x)    ((((x) >> S_CQE_SWCQE)) & M_CQE_SWCQE)
 #define V_CQE_SWCQE(x)	  ((x)<<S_CQE_SWCQE)
 
+#define S_CQE_DRAIN       10
+#define M_CQE_DRAIN       0x1
+#define G_CQE_DRAIN(x)    ((((x) >> S_CQE_DRAIN)) & M_CQE_DRAIN)
+#define V_CQE_DRAIN(x)	  ((x)<<S_CQE_DRAIN)
+
 #define S_CQE_STATUS      5
 #define M_CQE_STATUS      0x1F
 #define G_CQE_STATUS(x)   ((((x) >> S_CQE_STATUS)) & M_CQE_STATUS)
@@ -246,6 +251,7 @@ struct t4_cqe {
 #define V_CQE_OPCODE(x)   ((x)<<S_CQE_OPCODE)
 
 #define SW_CQE(x)         (G_CQE_SWCQE(be32_to_cpu((x)->header)))
+#define DRAIN_CQE(x)      (G_CQE_DRAIN(be32_to_cpu((x)->header)))
 #define CQE_QPID(x)       (G_CQE_QPID(be32_to_cpu((x)->header)))
 #define CQE_TYPE(x)       (G_CQE_TYPE(be32_to_cpu((x)->header)))
 #define SQ_TYPE(x)	  (CQE_TYPE((x)))