git: 4b5e09be147c - main - iflib: Separate software admission from legacy driver flags
- Go to: [ bottom of page ] [ top of archives ] [ this month ]
Date: Tue, 15 Sep 2026 18:57:49 UTC
The branch main has been updated by kbowling:
URL: https://cgit.FreeBSD.org/src/commit/?id=4b5e09be147c7303147b1742845ffe4417c4aabf
commit 4b5e09be147c7303147b1742845ffe4417c4aabf
Author: Kevin Bowling <kbowling@FreeBSD.org>
AuthorDate: 2026-09-11 19:00:09 +0000
Commit: Kevin Bowling <kbowling@FreeBSD.org>
CommitDate: 2026-09-15 18:55:49 +0000
iflib: Separate software admission from legacy driver flags
Publish an atomic software run state snapshot through iflib_is_running().
Use it for iflib traffic admission, queue tasks, timers, debugnet and
live-configuration checks instead of reading the unsynchronized
ifnet driver flags. Keep admission closed after failed initialization
and close it when a watchdog requests deferred recovery.
Retain the context-locked datapath state for DMA ownership: a closed
admission gate does not establish that the hardware is stopped. Open
the gate after receive buffer setup and before interrupt enable, at the
existing RUNNING publication point. Serialize the writers with the
state mutex and continue publishing RUNNING/OACTIVE for network stack
consumers. The accessor takes no lock and is usable from filters, but
is only a snapshot, not a context reference or a queue-user drain.
Recheck multicast and VFLR admission under the context lock. Replace
the OACTIVE drain check with the same private admission gate, retaining
the separate per-queue descriptor backpressure policy. Rename its
debug counter to txq_drain_stopped.
Document the accessor and its synchronization limits. This does not
remove legacy flag reads elsewhere in the network stack or change the
existing queue-drain and device-stop contracts. External drivers will
still see the IFF_DRV_ publications but should migrate to this API.
Reviewed by: iflib (gallatin)
MFC after: 2 weeks
Sponsored by: BBOX.io
Differential Revision: https://reviews.freebsd.org/D59598
---
share/man/man9/iflibdi.9 | 29 ++++++++++-
sys/net/iflib.c | 129 +++++++++++++++++++++++++++++------------------
sys/net/iflib.h | 7 +++
3 files changed, 116 insertions(+), 49 deletions(-)
diff --git a/share/man/man9/iflibdi.9 b/share/man/man9/iflibdi.9
index edc2c7f172f1..f5039dd41c3f 100644
--- a/share/man/man9/iflibdi.9
+++ b/share/man/man9/iflibdi.9
@@ -1,4 +1,4 @@
-.Dd August 8, 2026
+.Dd September 11, 2026
.Dt IFLIBDI 9
.Os
.Sh NAME
@@ -92,6 +92,10 @@
.Fo iflib_init_failed
.Fa "if_ctx_t ctx"
.Fc
+.Ft bool
+.Fo iflib_is_running
+.Fa "if_ctx_t ctx"
+.Fc
.Ft void
.Fo iflib_add_int_delay_sysctl
.Fa "if_ctx_t ctx"
@@ -260,6 +264,29 @@ Iflib then leaves
.Dv IFF_DRV_RUNNING
clear and does not enable interrupts or periodic timers.
Output remains blocked so explicitly scheduled admin recovery work can run.
+.It Fn iflib_is_running
+Return an atomic snapshot of whether iflib admits software traffic to its
+queues.
+This function may be called from an interrupt filter without acquiring an
+iflib lock.
+Admission opens after successful driver initialization and receive-buffer
+setup, before enabling interrupts, and closes before stop or when the watchdog
+requests recovery.
+It remains closed after failed initialization.
+.Pp
+This is neither the physical link state nor proof that queue DMA has stopped.
+In particular, admission closes before deferred watchdog recovery stops the
+device.
+The snapshot does not hold a reference to the context or wait for existing
+queue users; callers must retain their normal lifetime and queue
+synchronization.
+Administrative and firmware recovery work may still be needed while admission
+is closed.
+.Pp
+Drivers should use this accessor instead of reading
+.Dv IFF_DRV_RUNNING
+for software run-state checks, and must not modify the driver flags.
+Iflib continues to publish the legacy driver flags for network-stack consumers.
.It Fn iflib_add_int_delay_sysctl
Modifies settings to user defined values for a given set of variables.
.El
diff --git a/sys/net/iflib.c b/sys/net/iflib.c
index 012c4ab5f54a..f3f963e5ab0e 100644
--- a/sys/net/iflib.c
+++ b/sys/net/iflib.c
@@ -164,9 +164,9 @@ struct iflib_ctx;
*
* Only STOPPED establishes that the device can no longer access the mappings;
* a failed initialization can leave queues active. IFF_UP separately records
- * administrative intent. Existing datapath users still use IFF_DRV_RUNNING,
- * but clearing that flag does not establish quiescence: the watchdog clears
- * it before the admin task stops the hardware. Use this state for lifecycle
+ * administrative intent. ifc_running separately gates software traffic,
+ * but clearing it does not establish quiescence: the watchdog closes that
+ * gate before the admin task stops the hardware. Use this state for lifecycle
* decisions under ifc_ctx_sx, not as an unlocked datapath admission check.
*/
enum iflib_datapath_state {
@@ -225,6 +225,8 @@ struct iflib_ctx {
uint32_t ifc_flags;
enum iflib_datapath_state ifc_datapath_state;
enum iflib_pm_state ifc_pm_state;
+ /* Atomic software admission snapshot, not proof of DMA quiescence. */
+ u_int ifc_running;
uint32_t ifc_max_fl_buf_size;
uint32_t ifc_rx_mbuf_sz;
@@ -302,6 +304,13 @@ iflib_get_ifp(if_ctx_t ctx)
return (ctx->ifc_ifp);
}
+bool
+iflib_is_running(if_ctx_t ctx)
+{
+
+ return (atomic_load_acq_int(&ctx->ifc_running) != 0);
+}
+
struct ifmedia *
iflib_get_media(if_ctx_t ctx)
{
@@ -585,7 +594,7 @@ typedef struct if_rxsd {
#define MAX_SINGLE_PACKET_FRACTION 12
#define IF_BAD_DMA ((bus_addr_t)-1)
-#define CTX_ACTIVE(ctx) ((if_getdrvflags((ctx)->ifc_ifp) & IFF_DRV_RUNNING))
+#define CTX_ACTIVE(ctx) iflib_is_running(ctx)
#define CTX_LOCK_INIT(_sc) sx_init(&(_sc)->ifc_ctx_sx, "iflib ctx lock")
#define CTX_LOCK(ctx) sx_xlock(&(ctx)->ifc_ctx_sx)
@@ -756,13 +765,13 @@ SYSCTL_INT(_net_iflib, OID_AUTO, fl_refills_large, CTLFLAG_RD,
&iflib_fl_refills_large, 0, "# large refills");
static int iflib_txq_drain_flushing;
-static int iflib_txq_drain_oactive;
+static int iflib_txq_drain_stopped;
static int iflib_txq_drain_notready;
SYSCTL_INT(_net_iflib, OID_AUTO, txq_drain_flushing, CTLFLAG_RD,
&iflib_txq_drain_flushing, 0, "# drain flushes");
-SYSCTL_INT(_net_iflib, OID_AUTO, txq_drain_oactive, CTLFLAG_RD,
- &iflib_txq_drain_oactive, 0, "# drain oactives");
+SYSCTL_INT(_net_iflib, OID_AUTO, txq_drain_stopped, CTLFLAG_RD,
+ &iflib_txq_drain_stopped, 0, "# drains interrupted by stop");
SYSCTL_INT(_net_iflib, OID_AUTO, txq_drain_notready, CTLFLAG_RD,
&iflib_txq_drain_notready, 0, "# drain notready");
@@ -813,7 +822,7 @@ iflib_debug_reset(void)
{
iflib_tx_seen = iflib_tx_sent = iflib_tx_encap = iflib_rx_allocs =
iflib_fl_refills = iflib_fl_refills_large = iflib_tx_frees =
- iflib_txq_drain_flushing = iflib_txq_drain_oactive =
+ iflib_txq_drain_flushing = iflib_txq_drain_stopped =
iflib_txq_drain_notready =
iflib_encap_load_mbuf_fail = iflib_encap_pad_mbuf_fail =
iflib_encap_txq_avail_fail = iflib_encap_txd_encap_fail =
@@ -2570,6 +2579,28 @@ iflib_rx_sds_free(iflib_rxq_t rxq)
}
}
+/*
+ * Serialize admission changes and publication of the legacy driver flags.
+ * Opening admission publishes queue setup to lockless readers before device
+ * interrupts are enabled. Closing it does not wait for existing users:
+ * queue locks, task drains and the driver stop contract still apply.
+ */
+static void
+iflib_set_running(if_ctx_t ctx, bool running)
+{
+
+ mtx_assert(&ctx->ifc_state_mtx, MA_OWNED);
+ if (running) {
+ if_setdrvflagbits(ctx->ifc_ifp, IFF_DRV_RUNNING,
+ IFF_DRV_OACTIVE);
+ atomic_store_rel_int(&ctx->ifc_running, 1);
+ } else {
+ atomic_store_rel_int(&ctx->ifc_running, 0);
+ if_setdrvflagbits(ctx->ifc_ifp, IFF_DRV_OACTIVE,
+ IFF_DRV_RUNNING);
+ }
+}
+
/*
* Timer routine
*/
@@ -2581,7 +2612,7 @@ iflib_timer(void *arg)
if_softc_ctx_t sctx = &ctx->ifc_softc_ctx;
uint64_t this_tick = ticks;
- if (!(if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_RUNNING))
+ if (!iflib_is_running(ctx))
return;
/*
@@ -2663,8 +2694,7 @@ iflib_timer(void *arg)
txq->ift_id, TXQ_AVAIL(txq),
txq->ift_pidx);
STATE_LOCK(ctx);
- if_setdrvflagbits(ctx->ifc_ifp,
- IFF_DRV_OACTIVE, IFF_DRV_RUNNING);
+ iflib_set_running(ctx, false);
ctx->ifc_flags |=
(IFC_DO_WATCHDOG | IFC_DO_RESET);
iflib_admin_intr_deferred(ctx);
@@ -2684,7 +2714,7 @@ iflib_timer(void *arg)
GROUPTASK_ENQUEUE(&txq->ift_task);
sctx->isc_pause_frames = 0;
- if (if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_RUNNING)
+ if (iflib_is_running(ctx))
callout_reset_on(&txq->ift_timer, iflib_timer_default, iflib_timer,
txq, txq->ift_timer.c_cpu);
}
@@ -2738,7 +2768,9 @@ iflib_init_locked(if_ctx_t ctx)
("iflib init from datapath state %d", ctx->ifc_datapath_state));
ctx->ifc_datapath_state = IFLIB_DP_STARTING;
- if_setdrvflagbits(ifp, IFF_DRV_OACTIVE, IFF_DRV_RUNNING);
+ STATE_LOCK(ctx);
+ iflib_set_running(ctx, false);
+ STATE_UNLOCK(ctx);
IFDI_INTR_DISABLE(ctx);
/*
@@ -2816,7 +2848,9 @@ iflib_init_locked(if_ctx_t ctx)
}
}
}
- if_setdrvflagbits(ctx->ifc_ifp, IFF_DRV_RUNNING, IFF_DRV_OACTIVE);
+ STATE_LOCK(ctx);
+ iflib_set_running(ctx, true);
+ STATE_UNLOCK(ctx);
IFDI_INTR_ENABLE(ctx);
txq = ctx->ifc_txqs;
for (i = 0; i < scctx->isc_ntxqsets; i++, txq++) {
@@ -2903,7 +2937,9 @@ iflib_stop(if_ctx_t ctx)
}
/* Tell the stack that the interface is no longer active */
- if_setdrvflagbits(ctx->ifc_ifp, IFF_DRV_OACTIVE, IFF_DRV_RUNNING);
+ STATE_LOCK(ctx);
+ iflib_set_running(ctx, false);
+ STATE_UNLOCK(ctx);
if (stop_hardware) {
ctx->ifc_datapath_state = IFLIB_DP_STOPPING;
@@ -4214,8 +4250,7 @@ iflib_txq_drain(struct ifmp_ring *r, uint32_t cidx, uint32_t pidx)
int mcast_sent, pkt_sent, reclaimed;
bool do_prefetch, rang, ring;
- if (__predict_false(!(if_getdrvflags(ifp) & IFF_DRV_RUNNING) ||
- !LINK_ACTIVE(ctx))) {
+ if (__predict_false(!iflib_is_running(ctx) || !LINK_ACTIVE(ctx))) {
DBG_COUNTER_INC(txq_drain_notready);
return (0);
}
@@ -4236,11 +4271,11 @@ iflib_txq_drain(struct ifmp_ring *r, uint32_t cidx, uint32_t pidx)
return (avail);
}
- if (__predict_false(if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_OACTIVE)) {
+ if (__predict_false(!iflib_is_running(ctx))) {
CALLOUT_LOCK(txq);
callout_stop(&txq->ift_timer);
CALLOUT_UNLOCK(txq);
- DBG_COUNTER_INC(txq_drain_oactive);
+ DBG_COUNTER_INC(txq_drain_stopped);
return (0);
}
@@ -4284,7 +4319,7 @@ iflib_txq_drain(struct ifmp_ring *r, uint32_t cidx, uint32_t pidx)
DBG_COUNTER_INC(tx_sent);
mcast_sent += !!(m->m_flags & M_MCAST);
- if (__predict_false(!(if_getdrvflags(ifp) & IFF_DRV_RUNNING)))
+ if (__predict_false(!iflib_is_running(ctx)))
break;
ETHER_BPF_MTAP(ifp, m);
rang = iflib_txd_db_check(txq, false);
@@ -4355,13 +4390,15 @@ _task_fn_tx(void *context)
{
iflib_txq_t txq = context;
if_ctx_t ctx = txq->ift_ctx;
+#if defined(DEV_NETMAP) || defined(ALTQ)
if_t ifp = ctx->ifc_ifp;
+#endif
int abdicate = ctx->ifc_sysctl_tx_abdicate;
#ifdef IFLIB_DIAGNOSTICS
txq->ift_cpu_exec_count[curcpu]++;
#endif
- if (!(if_getdrvflags(ifp) & IFF_DRV_RUNNING))
+ if (!iflib_is_running(ctx))
return;
#ifdef DEV_NETMAP
if ((if_getcapenable(ifp) & IFCAP_NETMAP) &&
@@ -4409,7 +4446,7 @@ _task_fn_rx(void *context)
rxq->ifr_cpu_exec_count[curcpu]++;
#endif
DBG_COUNTER_INC(task_fn_rxs);
- if (__predict_false(!(if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_RUNNING)))
+ if (__predict_false(!iflib_is_running(ctx)))
return;
#ifdef DEV_NETMAP
nmirq = netmap_rx_irq(ctx->ifc_ifp, rxq->ifr_id, &work);
@@ -4432,7 +4469,7 @@ skip_rxeof:
IFDI_RX_QUEUE_INTR_ENABLE(ctx, rxq->ifr_id);
DBG_COUNTER_INC(rx_intr_enables);
}
- if (__predict_false(!(if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_RUNNING)))
+ if (__predict_false(!iflib_is_running(ctx)))
return;
if (more & IFLIB_RXEOF_MORE)
@@ -4511,12 +4548,10 @@ _task_fn_iov(void *context, int pending)
if (iflib_in_detach(ctx))
return;
- if (!(if_getdrvflags(ctx->ifc_ifp) & IFF_DRV_RUNNING) &&
- !(ctx->ifc_sctx->isc_flags & IFLIB_ADMIN_ALWAYS_RUN))
- return;
-
CTX_LOCK(ctx);
- if (ctx->ifc_pm_state != IFLIB_PM_ACTIVE) {
+ if (ctx->ifc_pm_state != IFLIB_PM_ACTIVE ||
+ (!iflib_is_running(ctx) &&
+ !(ctx->ifc_sctx->isc_flags & IFLIB_ADMIN_ALWAYS_RUN))) {
CTX_UNLOCK(ctx);
return;
}
@@ -4575,7 +4610,7 @@ iflib_if_transmit(if_t ifp, struct mbuf *m)
int err, qidx;
int abdicate;
- if (__predict_false((if_getdrvflags(ifp) & IFF_DRV_RUNNING) == 0 || !LINK_ACTIVE(ctx))) {
+ if (__predict_false(!iflib_is_running(ctx) || !LINK_ACTIVE(ctx))) {
DBG_COUNTER_INC(tx_frees);
m_freem(m);
return (ENETDOWN);
@@ -4793,7 +4828,7 @@ iflib_if_ioctl(if_t ifp, u_long command, caddr_t data)
*/
if (avoid_reset) {
if_setflagbits(ifp, IFF_UP, 0);
- if (!(if_getdrvflags(ifp) & IFF_DRV_RUNNING))
+ if (!iflib_is_running(ctx))
reinit = 1;
#ifdef INET
if (!(if_getflags(ifp) & IFF_NOARP))
@@ -4830,7 +4865,7 @@ iflib_if_ioctl(if_t ifp, u_long command, caddr_t data)
case SIOCSIFFLAGS:
CTX_LOCK(ctx);
if (if_getflags(ifp) & IFF_UP) {
- if (if_getdrvflags(ifp) & IFF_DRV_RUNNING) {
+ if (iflib_is_running(ctx)) {
if ((if_getflags(ifp) ^ ctx->ifc_if_flags) &
(IFF_PROMISC | IFF_ALLMULTI)) {
CTX_UNLOCK(ctx);
@@ -4848,13 +4883,13 @@ iflib_if_ioctl(if_t ifp, u_long command, caddr_t data)
break;
case SIOCADDMULTI:
case SIOCDELMULTI:
- if (if_getdrvflags(ifp) & IFF_DRV_RUNNING) {
- CTX_LOCK(ctx);
+ CTX_LOCK(ctx);
+ if (iflib_is_running(ctx)) {
IFDI_INTR_DISABLE(ctx);
IFDI_MULTI_SET(ctx);
IFDI_INTR_ENABLE(ctx);
- CTX_UNLOCK(ctx);
}
+ CTX_UNLOCK(ctx);
break;
case SIOCSIFMEDIA:
CTX_LOCK(ctx);
@@ -6056,7 +6091,9 @@ iflib_device_resume_locked(if_ctx_t ctx)
ctx->ifc_pm_state = IFLIB_PM_ACTIVE;
if ((if_getflags(ifp) & IFF_UP) == 0) {
- if_setdrvflagbits(ifp, IFF_DRV_OACTIVE, IFF_DRV_RUNNING);
+ STATE_LOCK(ctx);
+ iflib_set_running(ctx, false);
+ STATE_UNLOCK(ctx);
return (0);
}
@@ -6191,10 +6228,10 @@ iflib_device_iov_init_restart(device_t dev, uint16_t num_vfs,
/*
* Drivers which change the PF queue layout need the complete iflib
* stop/init sequence around their IOV callback. Administrative state
- * and IFF_DRV_RUNNING do not establish that the queues are stopped:
+ * and software admission do not establish DMA quiescence:
* failed initialization or a pending watchdog reset can leave DMA
* active. Let iflib_stop() decide whether hardware needs quiescing,
- * and preserve administrative state across the layout change.
+ * and preserve administrative intent across the layout change.
*/
restart = (if_getflags(ifp) & IFF_UP) != 0;
iflib_stop(ctx);
@@ -6225,9 +6262,9 @@ iflib_device_iov_uninit_restart(device_t dev)
CTX_LOCK(ctx);
/*
- * RUNNING can be clear while a watchdog reset is pending but the
- * hardware is still live. Always stop before the driver changes its
- * queue layout, and use IFF_UP only to preserve administrative state.
+ * Software admission can be closed while a watchdog reset is pending
+ * but the hardware is still live. Always stop before the driver changes
+ * its queue layout, and use IFF_UP only to preserve administrative intent.
*/
restart = (if_getflags(ctx->ifc_ifp) & IFF_UP) != 0;
iflib_stop(ctx);
@@ -7941,8 +7978,7 @@ iflib_debugnet_transmit(if_t ifp, struct mbuf *m)
int pkt_sent = 0;
ctx = if_getsoftc(ifp);
- if ((if_getdrvflags(ifp) & (IFF_DRV_RUNNING | IFF_DRV_OACTIVE)) !=
- IFF_DRV_RUNNING)
+ if (!iflib_is_running(ctx))
return (EBUSY);
txq = &ctx->ifc_txqs[0];
@@ -7964,8 +8000,7 @@ iflib_debugnet_poll(if_t ifp, int count)
ctx = if_getsoftc(ifp);
scctx = &ctx->ifc_softc_ctx;
- if ((if_getdrvflags(ifp) & (IFF_DRV_RUNNING | IFF_DRV_OACTIVE)) !=
- IFF_DRV_RUNNING)
+ if (!iflib_is_running(ctx))
return (EBUSY);
txq = &ctx->ifc_txqs[0];
@@ -8153,8 +8188,7 @@ iflib_simple_drbr_drain(iflib_txq_t txq, u_int quota, int *bytes, int *pkts,
mtx_assert(&txq->ift_mtx, MA_OWNED);
ctx = txq->ift_ctx;
ifp = ctx->ifc_ifp;
- if (__predict_false(!(if_getdrvflags(ifp) & IFF_DRV_RUNNING) ||
- !LINK_ACTIVE(ctx)))
+ if (__predict_false(!iflib_is_running(ctx) || !LINK_ACTIVE(ctx)))
return;
#ifdef ALTQ
@@ -8269,8 +8303,7 @@ iflib_simple_transmit(if_t ifp, struct mbuf *m)
int bytes_sent = 0, pkt_sent = 0, mcast_sent = 0;
ctx = if_getsoftc(ifp);
- if (__predict_false((if_getdrvflags(ifp) & IFF_DRV_RUNNING) == 0
- || !LINK_ACTIVE(ctx)))
+ if (__predict_false(!iflib_is_running(ctx) || !LINK_ACTIVE(ctx)))
goto net_down;
txq = iflib_simple_select_queue(ctx, m);
diff --git a/sys/net/iflib.h b/sys/net/iflib.h
index cab5c6232f9c..9bf1a1d8dbbf 100644
--- a/sys/net/iflib.h
+++ b/sys/net/iflib.h
@@ -436,6 +436,13 @@ device_t iflib_get_dev(if_ctx_t ctx);
if_t iflib_get_ifp(if_ctx_t ctx);
+/*
+ * Lockless software admission snapshot. This neither pins the context nor
+ * establishes that queue DMA has stopped; callers retain their usual lifetime
+ * and queue synchronization requirements.
+ */
+bool iflib_is_running(if_ctx_t ctx);
+
struct ifmedia *iflib_get_media(if_ctx_t ctx);
if_softc_ctx_t iflib_get_softc_ctx(if_ctx_t ctx);