git: 89ce1aa07241 - main - igbv: Support Hyper-V virtual functions

From: Kevin Bowling <kbowling_at_FreeBSD.org>
Date: Wed, 16 Sep 2026 19:50:11 UTC
The branch main has been updated by kbowling:

URL: https://cgit.FreeBSD.org/src/commit/?id=89ce1aa07241abbcea1703b5f2b85b88460d584f

commit 89ce1aa07241abbcea1703b5f2b85b88460d584f
Author:     Kevin Bowling <kbowling@FreeBSD.org>
AuthorDate: 2026-09-16 17:13:00 +0000
Commit:     Kevin Bowling <kbowling@FreeBSD.org>
CommitDate: 2026-09-16 19:47:56 +0000

    igbv: Support Hyper-V virtual functions
    
    Use the Hyper-V reset/MAC exchange for 82576 and I350 VFs instead of the
    native posted mailbox protocol, which the Windows PF does not service.
    Read the host assigned address through configuration bytes 0x201 through
    0x206 only during reset, and use it to identify the matching synthetic
    hn(4) interface.  The operations are local to the VF frontend.
    
    Poll hardware link status rather than retaining a native mailbox link
    handshake.  Leave MAC, multicast, promiscuous-mode, and VLAN membership
    policy with the host.  Disable guest VLAN registration and native receive
    limit requests, and limit the VF to an MTU of 1500 bytes.
    
    Preserve accumulated statistics across host resets without counting a
    counter clear as a wrap.  Reject inaccessible register samples and rebase
    after a reset indication or a disabled transmit queue, including when the
    PF blocks the queue for malicious driver detection.
    
    Document single queue support and host assigned access VLANs.  Guest VLAN
    trunks are not supported: tagged transmissions with the Windows I350 PF
    driver 14.1.5.0 can disable VF queues even on a trunk-configured port, and
    the limitation was also reproduced with Windows VF drivers.  Such trunks
    must use the synthetic path with SR-IOV disabled for that virtual adapter.
    
    Tested on Windows Server 2025 Hyper-V with both 82576 and I350 PFs.
    
    Relnotes:       yes
    Sponsored by:   BBOX.io
---
 share/man/man4/em.4     |  29 ++++++++++++--
 sys/dev/e1000/if_em.c   |  82 +++++++++++++++++++++++++++++++++++---
 sys/dev/e1000/if_em.h   |  11 ++++++
 sys/dev/e1000/if_igbv.c | 103 ++++++++++++++++++++++++++++++++++++++++++++++++
 4 files changed, 217 insertions(+), 8 deletions(-)

diff --git a/share/man/man4/em.4 b/share/man/man4/em.4
index 49b2e95e208c..187930de087d 100644
--- a/share/man/man4/em.4
+++ b/share/man/man4/em.4
@@ -32,7 +32,7 @@
 .\"
 .\" * Other names and brands may be claimed as the property of others.
 .\"
-.Dd August 30, 2026
+.Dd September 15, 2026
 .Dt EM 4
 .Os
 .Sh NAME
@@ -90,8 +90,30 @@ Virtual functions provided by 82576 and I350 physical functions appear as
 .Nm igbv
 interfaces.
 .Pp
-The driver supports Transmit/Receive checksum offload and Jumbo Frames
-on all but 82542-based adapters.
+On Hyper-V,
+.Nm igbv
+supports 82576 and I350 virtual functions using the host-assigned MAC address
+and a single transmit/receive queue pair.
+The VF can be used by
+.Xr hn 4
+as the accelerated data path for its synthetic interface.
+The host controls MAC, multicast, promiscuous-mode, and VLAN-filter policy;
+changing these on the VF does not reconfigure the host.
+Hyper-V VFs currently support an MTU of at most 1500 bytes.
+.Pp
+Only host-assigned access VLANs with untagged guest traffic are supported
+on Hyper-V; guest VLAN registration is ignored.
+With Windows I350 PF driver version 14.1.5.0, tagged VF transmissions can
+trigger host malicious driver detection and
+disable the VF's queues even when the Hyper-V port is configured for trunk mode.
+This limitation is also reproducible with Windows VF drivers.
+For guest VLAN trunks, disable SR-IOV for the virtual network adapter on
+the host and use the
+.Xr hn 4
+synthetic data path.
+.Pp
+The driver supports Transmit/Receive checksum offload.
+Jumbo Frames are supported except on 82542-based adapters and Hyper-V VFs.
 .Pp
 Physical functions on supported boards and ports advertise one or more of the
 .Cm wol_magic ,
@@ -515,6 +537,7 @@ issue to
 .Sh SEE ALSO
 .Xr altq 4 ,
 .Xr arp 4 ,
+.Xr hn 4 ,
 .Xr iflib 4 ,
 .Xr led 4 ,
 .Xr netintro 4 ,
diff --git a/sys/dev/e1000/if_em.c b/sys/dev/e1000/if_em.c
index 3ca8253a1457..9e4d3e8cfda2 100644
--- a/sys/dev/e1000/if_em.c
+++ b/sys/dev/e1000/if_em.c
@@ -395,11 +395,11 @@ static const pci_vendor_info_t igbv_vendor_info_array[] = {
 	PVID(0x8086, E1000_DEV_ID_82576_VF,
 	    "Intel(R) PRO/1000 82576 Virtual Function"),
 	PVID(0x8086, E1000_DEV_ID_82576_VF_HV,
-	    "Intel(R) PRO/1000 82576 Virtual Function"),
+	    "Intel(R) PRO/1000 82576 Hyper-V Virtual Function"),
 	PVID(0x8086, E1000_DEV_ID_I350_VF,
 	    "Intel(R) I350 Virtual Function"),
 	PVID(0x8086, E1000_DEV_ID_I350_VF_HV,
-	    "Intel(R) I350 Virtual Function"),
+	    "Intel(R) I350 Hyper-V Virtual Function"),
 	PVID_END
 };
 
@@ -1323,6 +1323,12 @@ em_if_attach_pre(if_ctx_t ctx)
 		scctx->isc_tx_tso_segsize_max = EM_TSO_SEG_SIZE;
 		scctx->isc_capabilities = scctx->isc_capenable =
 		    sc->vf_ifp ? IGBV_CAPS : IGB_CAPS;
+		if (igbv_is_hyperv(sc)) {
+			/* The host owns VLAN filters and the receive-frame limit. */
+			scctx->isc_capabilities &=
+			    ~(IFCAP_VLAN_HWFILTER | IFCAP_JUMBO_MTU);
+			scctx->isc_capenable = scctx->isc_capabilities;
+		}
 		scctx->isc_tx_csum_flags = CSUM_TCP | CSUM_UDP | CSUM_TSO |
 		     CSUM_IP6_TCP | CSUM_IP6_UDP;
 		if (hw->mac.type != e1000_82575)
@@ -1500,6 +1506,9 @@ em_if_attach_pre(if_ctx_t ctx)
 		goto err_pci;
 	}
 
+	if (igbv_is_hyperv(sc))
+		igbv_init_hv_ops(hw);
+
 	em_setup_msix(ctx);
 	e1000_get_bus_info(hw);
 
@@ -1637,7 +1646,7 @@ em_if_attach_pre(if_ctx_t ctx)
 	/* Copy the permanent MAC address out of the EEPROM */
 	if (e1000_read_mac_addr(hw) < 0) {
 		device_printf(dev,
-		    "EEPROM read error while reading MAC address\n");
+		    "Unable to read MAC address\n");
 		error = EIO;
 		goto err_late;
 	}
@@ -1855,6 +1864,10 @@ em_if_mtu_set(if_ctx_t ctx, uint32_t mtu)
 
 	IOCTL_DEBUGOUT("ioctl rcv'd: SIOCSIFMTU (Set Interface MTU)");
 
+	/* No native SET_LPE exchange is available with a Hyper-V PF. */
+	if (igbv_is_hyperv(sc) && mtu > ETHERMTU)
+		return (EINVAL);
+
 	switch (sc->hw.mac.type) {
 	case e1000_82571:
 	case e1000_82572:
@@ -1930,7 +1943,7 @@ em_if_init(if_ctx_t ctx)
 
 	/*
 	 * A VF restores its address only after its reset handshake establishes
-	 * CTS.  The PF path programs RAR[0] directly here.
+	 * its PF-assigned state.  The PF path programs RAR[0] directly here.
 	 */
 	if (!sc->vf_ifp)
 		e1000_rar_set(&sc->hw, sc->hw.mac.addr, 0);
@@ -3696,6 +3709,10 @@ em_if_set_promisc_impl(if_ctx_t ctx, int flags)
 	u32 reg_rctl;
 	int mcnt = 0;
 
+	/* Hyper-V receive-mode policy is configured through the host. */
+	if (igbv_is_hyperv(sc))
+		return (0);
+
 	if (sc->vf_ifp) {
 		if (flags & IFF_PROMISC)
 			type = e1000_promisc_enabled;
@@ -6141,6 +6158,10 @@ em_if_vlan_register(if_ctx_t ctx, u16 vtag)
 	bool present;
 	u32 index, mask;
 
+	/* Hyper-V supports host-assigned access VLANs, not guest VLANs. */
+	if (igbv_is_hyperv(sc))
+		return;
+
 	index = (vtag >> 5) & 0x7F;
 	mask = 1U << (vtag & 0x1F);
 	present = (sc->shadow_vfta[index] & mask) != 0;
@@ -6174,6 +6195,9 @@ em_if_vlan_unregister(if_ctx_t ctx, u16 vtag)
 	bool present;
 	u32 index, mask;
 
+	if (igbv_is_hyperv(sc))
+		return;
+
 	index = (vtag >> 5) & 0x7F;
 	mask = 1U << (vtag & 0x1F);
 	present = (sc->shadow_vfta[index] & mask) != 0;
@@ -6288,6 +6312,10 @@ em_setup_vlan_hw_support(if_ctx_t ctx)
 	u16 vid;
 	int restore_failures;
 
+	/* Hyper-V programs the VF's receive limit and VLAN membership. */
+	if (igbv_is_hyperv(sc))
+		return;
+
 	/*
 	 * Only PFs have control over VLAN HW filtering
 	 * configuration. VFs have to act as if it's always
@@ -7437,6 +7465,11 @@ static void
 em_rebase_vf_stats(struct e1000_softc *sc)
 {
 	struct e1000_vf_stats *stats;
+	bool hyperv = igbv_is_hyperv(sc);
+
+	sc->vf_stats_valid = false;
+	if (hyperv && E1000_READ_REG(&sc->hw, E1000_STATUS) == 0xffffffff)
+		return;
 
 	/*
 	 * A PF reset starts a new VF counter epoch.  Preserve the accumulated
@@ -7465,14 +7498,30 @@ em_rebase_vf_stats(struct e1000_softc *sc)
 	INIT_VF_REG(E1000_VFGORLBC, gorlbc);
 	INIT_VF_REG(E1000_VFGPRLBC, gprlbc);
 #undef INIT_VF_REG
+	sc->vf_stats_valid = !hyperv ||
+	    E1000_READ_REG(&sc->hw, E1000_STATUS) != 0xffffffff;
 }
 
 static void
 em_update_vf_stats_counters(struct e1000_softc *sc)
 {
+	struct e1000_vf_stats sample;
 	struct e1000_vf_stats *stats;
+	bool hyperv, reset;
 
-	stats = &sc->ustats.vf_stats;
+	hyperv = igbv_is_hyperv(sc);
+	if (hyperv && E1000_READ_REG(&sc->hw, E1000_STATUS) == 0xffffffff) {
+		sc->vf_stats_valid = false;
+		return;
+	}
+	reset = hyperv && e1000_check_for_rst(&sc->hw, 0) == E1000_SUCCESS;
+	if (hyperv && !sc->vf_stats_valid)
+		reset = true;
+	if (hyperv && (E1000_READ_REG(&sc->hw, E1000_TXDCTL(0)) &
+	    E1000_TXDCTL_QUEUE_ENABLE) == 0)
+		reset = true;
+	sample = sc->ustats.vf_stats;
+	stats = &sample;
 
 	/*
 	 * Internal VF loopback traffic can continue without physical link,
@@ -7497,6 +7546,29 @@ em_update_vf_stats_counters(struct e1000_softc *sc)
 	    stats->last_gorlbc, stats->gorlbc);
 	UPDATE_VF_REG(E1000_VFGPRLBC,
 	    stats->last_gprlbc, stats->gprlbc);
+	/*
+	 * Hyper-V can reset the VF without a native mailbox handshake.  Do not
+	 * count that counter clear as a 32-bit wrap, including a reset during
+	 * this sweep.  The PF can also clear counters while blocking a queue
+	 * for MDD without leaving a reset indication.  Hyper-V has one queue;
+	 * a disabled queue cannot supply a valid running counter epoch.
+	 */
+	if (hyperv) {
+		if (e1000_check_for_rst(&sc->hw, 0) == E1000_SUCCESS)
+			reset = true;
+		if ((E1000_READ_REG(&sc->hw, E1000_TXDCTL(0)) &
+		    E1000_TXDCTL_QUEUE_ENABLE) == 0)
+			reset = true;
+		if (E1000_READ_REG(&sc->hw, E1000_STATUS) == 0xffffffff) {
+			/* Rebase on a good sample before accounting further deltas. */
+			sc->vf_stats_valid = false;
+			return;
+		}
+	}
+	if (reset)
+		em_rebase_vf_stats(sc);
+	else
+		sc->ustats.vf_stats = sample;
 }
 
 static uint64_t
diff --git a/sys/dev/e1000/if_em.h b/sys/dev/e1000/if_em.h
index 3f3663b8d9a0..e73d359de19a 100644
--- a/sys/dev/e1000/if_em.h
+++ b/sys/dev/e1000/if_em.h
@@ -721,10 +721,20 @@ struct e1000_softc {
 	bool			vf_mbx_retry_initialized;
 	bool			vf_queues_sanitized;
 	bool			vf_reset_pending;
+	bool			vf_stats_valid;
 	/* A PF can retain auxiliary filters across a VF reset. */
 	bool			vf_uc_filters_set;
 };
 
+static inline bool
+igbv_is_hyperv(const struct e1000_softc *sc)
+{
+
+	return (sc->vf_ifp &&
+	    (sc->hw.device_id == E1000_DEV_ID_82576_VF_HV ||
+	    sc->hw.device_id == E1000_DEV_ID_I350_VF_HV));
+}
+
 /*
  * Shared PF/VF mechanisms and VF policy entry points.  The latter live in
  * if_igbv.c so the VF method table cannot accidentally select PF policy.
@@ -742,6 +752,7 @@ int	igbv_get_regs(SYSCTL_HANDLER_ARGS);
 int	igbv_if_attach_pre(if_ctx_t);
 int	igbv_if_attach_post(if_ctx_t);
 int	igbv_if_media_change(if_ctx_t);
+void	igbv_init_hv_ops(struct e1000_hw *);
 void	igbv_if_intr_enable(if_ctx_t);
 void	igbv_if_intr_disable(if_ctx_t);
 void	igbv_if_update_admin_status(if_ctx_t);
diff --git a/sys/dev/e1000/if_igbv.c b/sys/dev/e1000/if_igbv.c
index 355b37fdcbee..62b09973b301 100644
--- a/sys/dev/e1000/if_igbv.c
+++ b/sys/dev/e1000/if_igbv.c
@@ -41,6 +41,9 @@
 #define	IGBV_VLAN_RETRY_BATCH	4
 #define	IGBV_VLAN_RETRY_WINDOW	(8 * SBT_1S)
 
+/* Reading these six configuration bytes performs the Hyper-V VF handshake. */
+#define	IGBV_HV_RESET_OFFSET	0x201
+
 static const struct timeval igbv_queue_log_interval = { 2, 0 };
 static const struct timeval igbv_mbx_log_interval = { 60, 0 };
 static const sbintime_t igbv_queue_retry_delay[] = {
@@ -65,6 +68,103 @@ static bool	igbv_tx_pending(struct e1000_softc *);
 static bool	igbv_vlan_retry_pending(const struct e1000_softc *);
 static void	igbv_vlan_retry_tick(struct e1000_softc *);
 
+static s32
+igbv_hv_read_mac_addr(struct e1000_hw *hw)
+{
+
+	if (!em_is_valid_ether_addr(hw->mac.perm_addr))
+		return (-E1000_ERR_MAC_INIT);
+	memcpy(hw->mac.addr, hw->mac.perm_addr, ETHER_ADDR_LEN);
+	return (E1000_SUCCESS);
+}
+
+static s32
+igbv_hv_reset_hw(struct e1000_hw *hw)
+{
+	struct e1000_osdep *osdep;
+	u8 addr[ETHER_ADDR_LEN];
+	u32 ctrl;
+	int i;
+
+	osdep = hw->back;
+	hw->mac.get_link_status = true;
+	memset(hw->mac.perm_addr, 0, ETHER_ADDR_LEN);
+	/* Hyper-V does not service the native VF posted-message protocol. */
+	hw->mbx.timeout = 0;
+	ctrl = E1000_READ_REG(hw, E1000_CTRL);
+	if (ctrl == 0xffffffff)
+		return (-E1000_ERR_RESET);
+	E1000_WRITE_REG(hw, E1000_CTRL, ctrl | E1000_CTRL_RST);
+	E1000_WRITE_FLUSH(hw);
+	for (i = 0; i < E1000_VF_INIT_TIMEOUT; i++) {
+		if (hw->mbx.ops.check_for_rst(hw, 0) != E1000_SUCCESS)
+			break;
+		DELAY(5);
+	}
+
+	/*
+	 * The PF handles 0x201..0x206 as a reset/MAC exchange, including
+	 * partial reads.  Do not use these bytes for routine status polling.
+	 * The host-assigned address also identifies the matching hn interface.
+	 */
+	for (i = 0; i < ETHER_ADDR_LEN; i++)
+		addr[i] = pci_read_config(osdep->dev, IGBV_HV_RESET_OFFSET + i, 1);
+	if (!em_is_valid_ether_addr(addr))
+		return (-E1000_ERR_MAC_INIT);
+	memcpy(hw->mac.perm_addr, addr, ETHER_ADDR_LEN);
+	return (E1000_SUCCESS);
+}
+
+static s32
+igbv_hv_check_for_link(struct e1000_hw *hw)
+{
+	u32 status;
+
+	/* Leave reset indications for the statistics counter-epoch check. */
+	status = E1000_READ_REG(hw, E1000_STATUS);
+	hw->mac.get_link_status = true;
+	if (status == 0xffffffff)
+		return (-E1000_ERR_MAC_INIT);
+	/* Poll even after link-up so a host port link-down cannot stay cached. */
+	hw->mac.get_link_status = (status & E1000_STATUS_LU) == 0;
+	return (E1000_SUCCESS);
+}
+
+static int
+igbv_hv_rar_set(struct e1000_hw *hw, u8 *addr, u32 index)
+{
+	int error;
+
+	error = index == 0 &&
+	    memcmp(addr, hw->mac.perm_addr, ETHER_ADDR_LEN) == 0 ?
+	    E1000_SUCCESS : -E1000_ERR_MAC_INIT;
+	if (igbv_hv_read_mac_addr(hw) != E1000_SUCCESS)
+		return (-E1000_ERR_MAC_INIT);
+	return (error);
+}
+
+static void
+igbv_hv_update_mc_addr_list(struct e1000_hw *hw __unused,
+    u8 *addrs __unused, u32 count __unused)
+{
+
+	/* The host owns receive-filter policy for the synthetic/VF pair. */
+}
+
+void
+igbv_init_hv_ops(struct e1000_hw *hw)
+{
+
+	/* Install after the ordinary VF MAC and mailbox parameter setup. */
+	hw->mac.ops.reset_hw = igbv_hv_reset_hw;
+	/* Native init_hw calls rar_set_vf directly, bypassing the RAR op. */
+	hw->mac.ops.init_hw = igbv_hv_read_mac_addr;
+	hw->mac.ops.read_mac_addr = igbv_hv_read_mac_addr;
+	hw->mac.ops.rar_set = igbv_hv_rar_set;
+	hw->mac.ops.check_for_link = igbv_hv_check_for_link;
+	hw->mac.ops.update_mc_addr_list = igbv_hv_update_mc_addr_list;
+}
+
 static void
 igbv_queue_retry_callout(void *arg)
 {
@@ -772,6 +872,9 @@ igbv_update_uc_addr_list(struct e1000_softc *sc, if_t ifp)
 	};
 	u_int count;
 
+	if (igbv_is_hyperv(sc))
+		return;
+
 	count = if_foreach_lladdr(ifp, igbv_copy_uc_addr, &list);
 	if (count > IGBV_MAX_MAC_FILTERS) {
 		device_printf(sc->dev,