git: 2019f3d86f78 - main - rtnetlink: Add native SR-IOV VF status

From: Kevin Bowling <kbowling_at_FreeBSD.org>
Date: Sun, 06 Sep 2026 13:52:00 UTC
The branch main has been updated by kbowling:

URL: https://cgit.FreeBSD.org/src/commit/?id=2019f3d86f783711e9969d72cc423b26a4ce2f5a

commit 2019f3d86f783711e9969d72cc423b26a4ce2f5a
Author:     Kevin Bowling <kbowling@FreeBSD.org>
AuthorDate: 2026-08-10 22:29:23 +0000
Commit:     Kevin Bowling <kbowling@FreeBSD.org>
CommitDate: 2026-09-06 13:51:52 +0000

    rtnetlink: Add native SR-IOV VF status
    
    Add a transport neutral kernel snapshot for NIC-specific SR-IOV VF
    status and an optional iflib provider method.  Providers gather state
    under driver defined synchronization.
    
    Honor RTEXT_FILTER_VF on RTM_GETLINK requests and encode the status as
    native typed route Netlink attributes.  Represent VFs, driver
    namespaces, and namespace fields as directly repeated nested attributes.
    Presence masks in consumers can distinguish omission from false or zero.
    
    Drivers may add custom status under stable, versioned namespaces.  The
    named, typed representation lets generic transports and consumers carry
    or display fields without knowing their driver-specific schemas, while
    the driver retains ownership of their names and meanings.
    
    Document the ABI and add parser and RTM_GETLINK coverage.
    
    Reviewed by:    melifaro, iflib (gallatin), kgalazka (previous version)
    Sponsored by:   BBOX.io
    Differential Revision:  https://reviews.freebsd.org/D58776
---
 share/man/man4/rtnetlink.4                    | 188 +++++++-
 sys/net/if.c                                  | 165 +++++++
 sys/net/if_dead.c                             |   8 +
 sys/net/if_private.h                          |   1 +
 sys/net/if_var.h                              |   4 +
 sys/net/if_vf_status.h                        | 157 +++++++
 sys/net/ifdi_if.m                             |  13 +
 sys/net/iflib.c                               |  17 +
 sys/netlink/netlink_snl.h                     |  26 ++
 sys/netlink/netlink_snl_route_parsers.h       | 355 +++++++++++++++
 sys/netlink/route/iface.c                     | 604 +++++++++++++++++++++++++-
 sys/netlink/route/interface.h                 |  83 +++-
 sys/netlink/route/route_var.h                 |   1 +
 tests/atf_python/sys/netlink/attrs.py         |  28 ++
 tests/atf_python/sys/netlink/netlink_route.py | 122 ++++++
 tests/sys/netlink/test_rtnl_iface.py          |  13 +-
 tests/sys/netlink/test_snl.c                  | 413 ++++++++++++++++++
 17 files changed, 2175 insertions(+), 23 deletions(-)

diff --git a/share/man/man4/rtnetlink.4 b/share/man/man4/rtnetlink.4
index 3d76c66c1917..76028858456a 100644
--- a/share/man/man4/rtnetlink.4
+++ b/share/man/man4/rtnetlink.4
@@ -22,7 +22,7 @@
 .\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
 .\" SUCH DAMAGE.
 .\"
-.Dd July 28, 2026
+.Dd September 6, 2026
 .Dt RTNETLINK 4
 .Os
 .Sh NAME
@@ -325,7 +325,17 @@ The following filters are recognised by the kernel:
 ifi_index	interface index
 IFLA_IFNAME	interface name
 IFLA_ALT_IFNAME	interface name
+IFLA_EXT_MASK	extended information selection bitmap
 .Ed
+.Pp
+Setting the
+.Dv RTEXT_FILTER_VF
+bit in
+.Dv IFLA_EXT_MASK
+requests SR-IOV VF status.
+The query is opt-in because obtaining status can require entering the PF
+driver.
+VF status is not included in unsolicited link notifications.
 .Ss TLVs
 .Bl -tag -width indent
 .It Dv IFLA_ADDRESS
@@ -340,6 +350,182 @@ IFLA_ALT_IFNAME	interface name
 (uint32_t) (readonly) Interface index.
 .It Dv IFLA_MASTER
 (uint32_t) Parent interface index.
+.It Dv IFLA_NUM_VF
+(uint32_t) (readonly) Number of VF records returned.
+This attribute is returned with a successful
+.Dv RTEXT_FILTER_VF
+query.
+.It Dv IFLA_FREEBSD
+(nested) Local interface attributes.
+When requested,
+.Dv IFLAF_VF_STATUS
+contains the following nested attributes:
+.Bd -literal -offset indent -compact
+IFLAF_VFS_ERROR		(uint32_t) errno if the PF query failed
+IFLAF_VFS_PF_LINK_STATE	(uint8_t) IFLAF_VF_LINK_*
+IFLAF_VFS_PF_LINK_SPEED	(uint64_t) bits per second
+.Ed
+.Pp
+If the PF status provider does not implement this query, the status block is
+omitted.
+If an implemented PF status query fails,
+.Dv IFLAF_VFS_ERROR
+contains the errno value and the other status attributes are omitted.
+An empty status snapshot reports
+.Dv IFLA_NUM_VF
+as zero and contains no
+.Dv IFLA_FREEBSD_VF
+attributes.
+.Pp
+Invalid or oversized data returned by a provider is reported through
+.Dv IFLAF_VFS_ERROR .
+Allocation failures while encoding the reply terminate the requested
+.Dv RTM_GETLINK
+operation with an error.
+For a multipart dump, valid interface records may precede the terminating
+error.
+.It Dv IFLA_FREEBSD_VF
+(nested) (readonly) One VF record containing
+.Dv IFLAF_VF_*
+attributes.
+The attribute is repeated once for every VF in a successful status snapshot.
+Keeping each VF in an independent attribute permits the complete Netlink
+message to exceed the 16-bit length limit of an individual attribute.
+Each
+.Dv IFLA_FREEBSD_VF
+may contain the following attributes:
+.Bd -literal -offset indent -compact
+IFLAF_VF_INDEX			(uint32_t) PF-local VF index
+IFLAF_VF_CONFIGURED		(bool) PF accepted configuration
+IFLAF_VF_INITIALIZED		(bool) VF handshake completed
+IFLAF_VF_MAC			(binary) PF-known primary MAC address
+IFLAF_VF_VLAN_MODE		(uint8_t) IFLAF_VF_VLAN_*
+IFLAF_VF_VLAN			(uint16_t) access VLAN identifier
+IFLAF_VF_VLAN_PCP		(uint8_t) access VLAN priority code point
+IFLAF_VF_VLAN_PROTO		(uint16_t) host-order access VLAN EtherType
+IFLAF_VF_VLAN_COUNT		(uint32_t) explicit VLAN filters
+IFLAF_VF_VLAN_LIMIT		(uint32_t) explicit VLAN-filter limit
+IFLAF_VF_NUM_TX_QUEUES		(uint16_t) allocated transmit queues
+IFLAF_VF_NUM_RX_QUEUES		(uint16_t) allocated receive queues
+IFLAF_VF_MIN_TX_RATE		(uint64_t) minimum transmit bits per second
+IFLAF_VF_MAX_TX_RATE		(uint64_t) maximum transmit bits per second
+IFLAF_VF_ALLOW_SET_MAC		(bool) administrative permission
+IFLAF_VF_ALLOW_SET_VLAN		(bool) administrative permission
+IFLAF_VF_MAC_ANTI_SPOOF		(bool) MAC anti-spoofing enabled
+IFLAF_VF_ALLOW_PROMISC		(bool) administrative permission
+IFLAF_VF_TRAFFIC_ALLOWED	(bool) PF permits VF data traffic
+IFLAF_VF_FAULT_BLOCKED		(bool) fault-containment block active
+IFLAF_VF_QUARANTINED		(bool) persistent quarantine active
+IFLAF_VF_API_VERSION		(string) negotiated mailbox API
+IFLAF_VF_LINK_STATE_POLICY	(uint8_t) IFLAF_VF_LINK_*
+IFLAF_VF_DRIVER			(nested) driver namespace; repeated
+.Ed
+.Pp
+.Dv IFLAF_VF_INDEX
+is required in every list entry.
+Other fields are optional and are omitted when the PF driver cannot observe
+them; omission does not mean false or zero.
+Boolean attributes contain a native Netlink boolean value.
+.Dv IFLAF_VF_CONFIGURED
+means that the PF accepted the VF configuration, while
+.Dv IFLAF_VF_INITIALIZED
+means that the VF completed its driver or mailbox handshake since its last
+reset.
+.Pp
+.Dv IFLAF_VF_TRAFFIC_ALLOWED
+means that the PF currently permits the VF to send and receive data traffic.
+It does not imply that the VF completed initialization, enabled queues, has
+link, or is actively passing packets.
+.Dv IFLAF_VF_FAULT_BLOCKED
+means that the PF isolated the VF because of a detected fault or abusive
+behavior.
+It does not include ordinary unconfigured, reset, or link-down states.
+The driver namespace describes the particular cause and recovery policy.
+When both fields are present, a fault block implies that traffic is not
+allowed.
+.Dv IFLAF_VF_QUARANTINED
+describes a persistent form of containment and, when both fields are present,
+implies that a fault block is active.
+.Pp
+.Dv IFLAF_VF_NUM_TX_QUEUES
+and
+.Dv IFLAF_VF_NUM_RX_QUEUES
+are the numbers of transmit and receive queues allocated to the VF, not
+necessarily the numbers currently used by its driver.
+The transmit-rate attributes describe configured aggregate VF policy.
+A present zero minimum means no guaranteed allocation; a present zero maximum
+means no rate limit.
+When both are present and the maximum is nonzero, the minimum does not exceed
+the maximum.
+Access VLAN mode means that the PF imposes a single port VLAN policy.
+The optional
+.Dv IFLAF_VF_VLAN ,
+.Dv IFLAF_VF_VLAN_PCP ,
+and
+.Dv IFLAF_VF_VLAN_PROTO
+attributes describe its VLAN identifier, IEEE 802.1p priority code point, and
+host-order tag EtherType, respectively.
+A present zero VLAN identifier can represent a priority-only access tag, and a
+present zero priority is distinct from an omitted priority.
+These attributes do not describe the VF's trunk filters.
+Trunk mode means that no access VLAN is imposed and does not promise unlimited
+filter capacity.
+.Dv IFLAF_VF_VLAN_COUNT
+counts explicit filters recorded by the PF and excludes implicit untagged and
+priority-tag membership.
+The permission attributes describe requests the VF may make, not requests it
+has made.
+The top-level PF link state and speed are values normally advertised to VFs,
+not evidence that a VF driver is operational.
+.Pp
+Link-state values are:
+.Bd -literal -offset indent -compact
+IFLAF_VF_LINK_UNKNOWN	state is unavailable
+IFLAF_VF_LINK_DOWN	link is forced or observed down
+IFLAF_VF_LINK_UP	link is forced or observed up
+IFLAF_VF_LINK_AUTO	VF follows PF link state
+.Ed
+.Pp
+VLAN-mode values are:
+.Bd -literal -offset indent -compact
+IFLAF_VF_VLAN_UNKNOWN	mode is unavailable
+IFLAF_VF_VLAN_ACCESS	PF imposes an access VLAN
+IFLAF_VF_VLAN_TRUNK	no access VLAN is imposed
+.Ed
+.Pp
+Driver-specific data is returned under
+.Dv IFLAF_VF_DRIVER ,
+which may be repeated directly in a VF record.
+Each driver object contains:
+.Bd -literal -offset indent -compact
+IFLAF_VFD_NAME		(string) stable namespace name
+IFLAF_VFD_VERSION	(uint32_t) namespace schema version
+IFLAF_VFD_FIELD		(nested) namespace field; repeated
+.Ed
+.Pp
+Each field contains its namespace-local name in
+.Dv IFLAF_VFDF_NAME
+and exactly one of
+.Dv IFLAF_VFDF_BOOL ,
+.Dv IFLAF_VFDF_NUMBER ,
+.Dv IFLAF_VFDF_STRING ,
+or
+.Dv IFLAF_VFDF_BINARY .
+The namespace defines the types and meanings of its fields.
+The named, typed form permits generic consumers to carry or display an
+extension without knowing its driver-specific schema.
+Consumers that interpret extensions must ignore unknown namespaces and
+fields.
+Each attribute, including each containing nested attribute, is limited to
+65535 bytes by the Netlink encoding.
+If an individual VF record, driver extension, or field cannot be encoded
+within that limit,
+.Dv IFLAF_VFS_ERROR
+reports
+.Er EMSGSIZE .
+The enclosing
+.Dv RTM_NEWLINK
+message may exceed 65535 bytes.
 .It Dv IFLA_LINKINFO
 (nested) Interface type-specific attributes:
 .Bd -literal -offset indent -compact
diff --git a/sys/net/if.c b/sys/net/if.c
index 644a039c8d25..552ae04d3215 100644
--- a/sys/net/if.c
+++ b/sys/net/if.c
@@ -81,6 +81,7 @@
 #include <net/if_strings.h>
 #include <net/if_types.h>
 #include <net/if_var.h>
+#include <net/if_vf_status.h>
 #include <net/if_media.h>
 #include <net/if_mib.h>
 #include <net/if_private.h>
@@ -2360,6 +2361,164 @@ if_capint_to_capnv(nvlist_t *nv, const struct ifcap_nv_bit_name *nn,
 	}
 }
 
+struct if_vf_status *
+if_vf_status_alloc(uint32_t num_vfs)
+{
+	struct if_vf_status *status;
+	size_t size;
+
+	KASSERT(num_vfs <= IFVF_MAX_VFS,
+	    ("invalid VF count %u", num_vfs));
+	size = sizeof(struct if_vf_status) +
+	    num_vfs * sizeof(struct if_vf_info);
+	status = malloc(size, M_IFNET, M_WAITOK | M_ZERO);
+	status->num_vfs = num_vfs;
+	return (status);
+}
+
+void
+if_vf_status_free(struct if_vf_status *status)
+{
+	struct if_vf_ext_field *field;
+	struct if_vf_extension *extension;
+	uint32_t i, j;
+
+	KASSERT(status != NULL, ("NULL VF status"));
+	for (i = 0; i < status->num_vfs; i++) {
+		for (j = 0; j < status->vfs[i].num_extensions; j++) {
+			extension = &status->vfs[i].extensions[j];
+			for (uint32_t k = 0; k < extension->num_fields; k++) {
+				field = &extension->fields[k];
+				if (field->type == IFVF_EXT_STRING)
+					free(field->value.string, M_IFNET);
+				else if (field->type == IFVF_EXT_BINARY)
+					free(field->value.binary.data, M_IFNET);
+			}
+			free(extension->fields, M_IFNET);
+		}
+		free(status->vfs[i].extensions, M_IFNET);
+	}
+
+	free(status, M_IFNET);
+}
+
+struct if_vf_extension *
+if_vf_status_add_extension(struct if_vf_info *vf, const char *name,
+    uint32_t version, uint32_t num_fields)
+{
+	struct if_vf_extension *extensions, *extension;
+	uint32_t count;
+
+	KASSERT(vf != NULL, ("NULL VF information"));
+	KASSERT(name != NULL, ("VF extension without a name"));
+	KASSERT(vf->num_extensions < IFVF_MAX_EXTENSIONS,
+	    ("too many VF extensions"));
+	KASSERT(num_fields > 0 && num_fields <= IFVF_MAX_EXTENSION_FIELDS,
+	    ("invalid VF extension field count %u", num_fields));
+	count = vf->num_extensions + 1;
+	extensions = mallocarray(count, sizeof(*extensions), M_IFNET,
+	    M_WAITOK | M_ZERO);
+	if (vf->num_extensions != 0) {
+		memcpy(extensions, vf->extensions,
+		    vf->num_extensions * sizeof(*extensions));
+		free(vf->extensions, M_IFNET);
+	}
+	vf->extensions = extensions;
+	vf->num_extensions = count;
+	extension = &extensions[count - 1];
+	extension->name = name;
+	extension->version = version;
+	extension->num_fields = num_fields;
+	extension->fields = mallocarray(num_fields, sizeof(*extension->fields),
+	    M_IFNET, M_WAITOK | M_ZERO);
+	return (extension);
+}
+
+static struct if_vf_ext_field *
+if_vf_extension_field(struct if_vf_extension *extension, uint32_t index,
+    const char *name, enum if_vf_ext_type type)
+{
+	struct if_vf_ext_field *field;
+
+	KASSERT(extension != NULL && index < extension->num_fields,
+	    ("invalid VF extension field"));
+	KASSERT(name != NULL, ("VF extension field without a name"));
+	field = &extension->fields[index];
+	KASSERT(field->type == 0, ("VF extension field initialized twice"));
+	field->name = name;
+	field->type = type;
+	return (field);
+}
+
+void
+if_vf_extension_set_bool(struct if_vf_extension *extension, uint32_t index,
+    const char *name, bool value)
+{
+	struct if_vf_ext_field *field;
+
+	field = if_vf_extension_field(extension, index, name, IFVF_EXT_BOOL);
+	field->value.boolean = value;
+}
+
+void
+if_vf_extension_set_number(struct if_vf_extension *extension, uint32_t index,
+    const char *name, uint64_t value)
+{
+	struct if_vf_ext_field *field;
+
+	field = if_vf_extension_field(extension, index, name, IFVF_EXT_NUMBER);
+	field->value.number = value;
+}
+
+void
+if_vf_extension_set_string(struct if_vf_extension *extension, uint32_t index,
+    const char *name, const char *value)
+{
+	struct if_vf_ext_field *field;
+
+	KASSERT(value != NULL, ("NULL VF extension string"));
+	field = if_vf_extension_field(extension, index, name, IFVF_EXT_STRING);
+	field->value.string = strdup(value, M_IFNET);
+}
+
+void
+if_vf_extension_set_binary(struct if_vf_extension *extension, uint32_t index,
+    const char *name, const void *value, uint32_t length)
+{
+	struct if_vf_ext_field *field;
+
+	KASSERT(value != NULL && length != 0, ("empty VF extension binary"));
+	field = if_vf_extension_field(extension, index, name, IFVF_EXT_BINARY);
+	field->value.binary.data = malloc(length, M_IFNET, M_WAITOK);
+	memcpy(field->value.binary.data, value, length);
+	field->value.binary.length = length;
+}
+
+int
+if_get_vf_status(if_t ifp, struct if_vf_status **statusp)
+{
+	struct if_vf_status *status;
+	int error;
+
+	KASSERT(statusp != NULL, ("NULL VF status output"));
+	if (ifp->if_vf_status == NULL)
+		return (EOPNOTSUPP);
+	status = NULL;
+	error = ifp->if_vf_status(ifp, &status);
+	KASSERT((error == 0) == (status != NULL),
+	    ("VF status provider returned error %d with status %p", error,
+	    status));
+	if (error != 0) {
+		if (status != NULL)
+			if_vf_status_free(status);
+		return (error);
+	}
+	if (status == NULL)
+		return (EBADMSG);
+	*statusp = status;
+	return (0);
+}
+
 /*
  * Hardware specific interface ioctls.
  */
@@ -4886,6 +5045,12 @@ if_setioctlfn(if_t ifp, if_ioctl_fn_t ioctl_fn)
 	ifp->if_ioctl = ioctl_fn;
 }
 
+void
+if_setvfstatusfn(if_t ifp, if_vf_status_fn_t vf_status_fn)
+{
+	ifp->if_vf_status = vf_status_fn;
+}
+
 void
 if_setoutputfn(if_t ifp, if_output_fn_t output_fn)
 {
diff --git a/sys/net/if_dead.c b/sys/net/if_dead.c
index 37e83bdc70bb..a5ea2b56a56b 100644
--- a/sys/net/if_dead.c
+++ b/sys/net/if_dead.c
@@ -107,6 +107,13 @@ ifdead_snd_tag_alloc(struct ifnet *ifp, union if_snd_tag_alloc_params *params,
 	return (EOPNOTSUPP);
 }
 
+static int
+ifdead_vf_status(struct ifnet *ifp __unused,
+    struct if_vf_status **statusp __unused)
+{
+	return (EOPNOTSUPP);
+}
+
 static void
 ifdead_ratelimit_query(struct ifnet *ifp __unused,
       struct if_ratelimit_query_results *q)
@@ -138,4 +145,5 @@ if_dead(struct ifnet *ifp)
 	ifp->if_get_counter = ifdead_get_counter;
 	ifp->if_snd_tag_alloc = ifdead_snd_tag_alloc;
 	ifp->if_ratelimit_query = ifdead_ratelimit_query;
+	ifp->if_vf_status = ifdead_vf_status;
 }
diff --git a/sys/net/if_private.h b/sys/net/if_private.h
index 5a76fd788c1b..3c718a072e15 100644
--- a/sys/net/if_private.h
+++ b/sys/net/if_private.h
@@ -127,6 +127,7 @@ struct ifnet {
 	void (*if_bridge_linkstate)(struct ifnet *ifp);
 	if_start_fn_t	if_start;	/* initiate output routine */
 	if_ioctl_fn_t	if_ioctl;	/* ioctl routine */
+	if_vf_status_fn_t if_vf_status;	/* snapshot SR-IOV VF state */
 	if_init_fn_t	if_init;	/* Init routine */
 	int	(*if_resolvemulti)	/* validate/resolve multicast */
 		(struct ifnet *, struct sockaddr **, struct sockaddr *);
diff --git a/sys/net/if_var.h b/sys/net/if_var.h
index d3d4b1e2a36c..8297520f7607 100644
--- a/sys/net/if_var.h
+++ b/sys/net/if_var.h
@@ -131,6 +131,8 @@ typedef void (*if_qflush_fn_t)(if_t);
 typedef int (*if_transmit_fn_t)(if_t, struct mbuf *);
 typedef	uint64_t (*if_get_counter_t)(if_t, ift_counter);
 typedef	void (*if_reassign_fn_t)(if_t, struct vnet *, char *);
+struct if_vf_status;
+typedef	int (*if_vf_status_fn_t)(if_t, struct if_vf_status **);
 typedef int (*if_spdadd_fn_t)(if_t ifp, void *sp, void *inp, void **priv);
 typedef int (*if_spddel_fn_t)(if_t ifp, void *sp, void *priv);
 typedef int (*if_sa_newkey_fn_t)(if_t ifp, void *sav, u_int drv_spi,
@@ -729,6 +731,7 @@ void if_setinitfn(if_t ifp, if_init_fn_t);
 void if_setinputfn(if_t ifp, if_input_fn_t);
 if_input_fn_t if_getinputfn(if_t ifp);
 void if_setioctlfn(if_t ifp, if_ioctl_fn_t);
+void if_setvfstatusfn(if_t ifp, if_vf_status_fn_t);
 void if_setoutputfn(if_t ifp, if_output_fn_t);
 void if_setstartfn(if_t ifp, if_start_fn_t);
 if_start_fn_t if_getstartfn(if_t ifp);
@@ -765,6 +768,7 @@ void *ifr_buffer_get_buffer(void *data);
 size_t ifr_buffer_get_length(void *data);
 
 int ifhwioctl(u_long, if_t, caddr_t, struct thread *);
+int if_get_vf_status(if_t, struct if_vf_status **);
 
 #ifdef DEVICE_POLLING
 enum poll_cmd { POLL_ONLY, POLL_AND_CHECK_STATUS };
diff --git a/sys/net/if_vf_status.h b/sys/net/if_vf_status.h
new file mode 100644
index 000000000000..57c625ff530c
--- /dev/null
+++ b/sys/net/if_vf_status.h
@@ -0,0 +1,157 @@
+/*
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2026 Kevin Bowling <kbowling@FreeBSD.org>
+ */
+
+#ifndef _NET_IF_VF_STATUS_H_
+#define	_NET_IF_VF_STATUS_H_
+
+#include <sys/types.h>
+#ifndef _KERNEL
+#include <stdbool.h>
+#endif
+
+#include <net/ethernet.h>
+
+/*
+ * Kernel-only snapshot of SR-IOV VF state.  The fields member is a bit mask
+ * identifying which other members contain valid data.  This distinguishes a
+ * missing value from a value of false or zero.  Drivers set bits only for
+ * information they can report, and each transport converts the snapshot to
+ * its own ABI.  VLAN PCP and protocol describe the PF-administered access
+ * VLAN, not the VF's trunk filters.  Transmit rates describe an aggregate VF
+ * policy.  When both are present and the maximum is nonzero, the minimum must
+ * not exceed it.  A successful snapshot may contain zero VFs, allowing a
+ * provider to report that SR-IOV is available but is not currently configured.
+ */
+#define	IFVF_MAX_VFS			UINT16_MAX
+#define	IFVF_MAX_EXTENSIONS		16
+#define	IFVF_MAX_EXTENSION_FIELDS	64
+
+enum if_vf_field {
+	IFVF_F_CONFIGURED		= 1ULL << 0,
+	IFVF_F_INITIALIZED		= 1ULL << 1,
+	IFVF_F_MAC			= 1ULL << 2,
+	IFVF_F_VLAN_MODE		= 1ULL << 3,
+	IFVF_F_VLAN			= 1ULL << 4,
+	IFVF_F_VLAN_PCP			= 1ULL << 5,
+	IFVF_F_VLAN_PROTO		= 1ULL << 6,
+	IFVF_F_VLAN_COUNT		= 1ULL << 7,
+	IFVF_F_VLAN_LIMIT		= 1ULL << 8,
+	IFVF_F_NUM_TX_QUEUES		= 1ULL << 9,
+	IFVF_F_NUM_RX_QUEUES		= 1ULL << 10,
+	IFVF_F_MIN_TX_RATE		= 1ULL << 11,
+	IFVF_F_MAX_TX_RATE		= 1ULL << 12,
+	IFVF_F_ALLOW_SET_MAC		= 1ULL << 13,
+	IFVF_F_ALLOW_SET_VLAN		= 1ULL << 14,
+	IFVF_F_MAC_ANTI_SPOOF		= 1ULL << 15,
+	IFVF_F_ALLOW_PROMISC		= 1ULL << 16,
+	IFVF_F_TRAFFIC_ALLOWED		= 1ULL << 17,
+	IFVF_F_FAULT_BLOCKED		= 1ULL << 18,
+	IFVF_F_QUARANTINED		= 1ULL << 19,
+	IFVF_F_API_VERSION		= 1ULL << 20,
+	IFVF_F_LINK_STATE_POLICY	= 1ULL << 21,
+};
+
+enum if_vf_vlan_mode {
+	IFVF_VLAN_UNKNOWN = 0,
+	IFVF_VLAN_ACCESS,
+	IFVF_VLAN_TRUNK,
+};
+
+enum if_vf_link_state {
+	IFVF_LINK_UNKNOWN = 0,
+	IFVF_LINK_DOWN,
+	IFVF_LINK_UP,
+	IFVF_LINK_AUTO,
+};
+
+/*
+ * Drivers may add fields under a stable, versioned namespace.  The named,
+ * typed representation lets transports and generic consumers carry and
+ * display fields whose driver-specific schema they do not know.  The driver
+ * owns the field names and meanings; common code does not interpret them.
+ * Namespace and field names must remain valid until the snapshot is freed.
+ */
+enum if_vf_ext_type {
+	IFVF_EXT_BOOL = 1,
+	IFVF_EXT_NUMBER,
+	IFVF_EXT_STRING,
+	IFVF_EXT_BINARY,
+};
+
+struct if_vf_ext_field {
+	const char *name;
+	enum if_vf_ext_type type;
+	union {
+		bool boolean;
+		uint64_t number;
+		char *string;
+		struct {
+			void *data;
+			uint32_t length;
+		} binary;
+	} value;
+};
+
+struct if_vf_extension {
+	const char *name;
+	uint32_t version;
+	uint32_t num_fields;
+	struct if_vf_ext_field *fields;
+};
+
+#define	IFVF_API_VERSION_MAX	16
+
+struct if_vf_info {
+	uint64_t fields;
+	uint64_t min_tx_rate_bps;	/* Zero means no guaranteed allocation. */
+	uint64_t max_tx_rate_bps;	/* Zero means unlimited. */
+	uint32_t index;
+	uint32_t vlan_count;
+	uint32_t vlan_limit;
+	uint16_t tx_queue_count;
+	uint16_t rx_queue_count;
+	uint16_t vlan;
+	uint16_t vlan_proto;		/* Host-order Ethernet type. */
+	uint8_t vlan_pcp;
+	uint8_t mac[ETHER_ADDR_LEN];
+	enum if_vf_vlan_mode vlan_mode;
+	enum if_vf_link_state link_state_policy;
+	bool configured:1;
+	bool initialized:1;
+	bool allow_set_mac:1;
+	bool allow_set_vlan:1;
+	bool mac_anti_spoof:1;
+	bool allow_promisc:1;
+	bool traffic_allowed:1;
+	bool fault_blocked:1;
+	bool quarantined:1;
+	/* Negotiated PF/VF mailbox API, not a version of this structure. */
+	char api_version[IFVF_API_VERSION_MAX];
+	uint32_t num_extensions;
+	struct if_vf_extension *extensions;
+};
+
+struct if_vf_status {
+	uint32_t num_vfs;
+	struct if_vf_info vfs[];
+};
+
+#ifdef _KERNEL
+struct if_vf_status *if_vf_status_alloc(uint32_t);
+void if_vf_status_free(struct if_vf_status *);
+struct if_vf_extension *if_vf_status_add_extension(struct if_vf_info *,
+    const char *, uint32_t, uint32_t);
+void if_vf_extension_set_bool(struct if_vf_extension *, uint32_t,
+    const char *, bool);
+void if_vf_extension_set_number(struct if_vf_extension *, uint32_t,
+    const char *, uint64_t);
+void if_vf_extension_set_string(struct if_vf_extension *, uint32_t,
+    const char *, const char *);
+void if_vf_extension_set_binary(struct if_vf_extension *, uint32_t,
+    const char *, const void *, uint32_t);
+#endif
+
+#endif /* _NET_IF_VF_STATUS_H_ */
diff --git a/sys/net/ifdi_if.m b/sys/net/ifdi_if.m
index e9db929e1900..fdec1b866746 100644
--- a/sys/net/ifdi_if.m
+++ b/sys/net/ifdi_if.m
@@ -36,6 +36,7 @@
 #include <net/ethernet.h>
 #include <net/if.h>
 #include <net/if_var.h>
+#include <net/if_vf_status.h>
 #include <net/if_media.h>
 #include <net/iflib.h>
 #include <net/if_private.h>
@@ -118,6 +119,13 @@ CODE {
 		return (ENOTSUP);
 	}
 
+	static int
+	null_vf_status(if_ctx_t _ctx __unused,
+	    struct if_vf_status **_status __unused)
+	{
+		return (ENOTSUP);
+	}
+
 	static bool
 	null_needs_restart(if_ctx_t _ctx __unused, enum iflib_restart_event _event __unused)
 	{
@@ -385,3 +393,8 @@ METHOD int get_downreason {
 	if_ctx_t _ctx;
 	struct ifdownreason *_ifdr;
 } DEFAULT null_get_downreason;
+
+METHOD int vf_status {
+	if_ctx_t _ctx;
+	struct if_vf_status **_status;
+} DEFAULT null_vf_status;
diff --git a/sys/net/iflib.c b/sys/net/iflib.c
index 0c2b03c0b5f3..4131f6dd61b3 100644
--- a/sys/net/iflib.c
+++ b/sys/net/iflib.c
@@ -88,6 +88,7 @@
 #include <dev/pci/pci_private.h>
 
 #include <net/iflib.h>
+#include <net/if_vf_status.h>
 
 #include "ifdi_if.h"
 
@@ -4741,6 +4742,19 @@ iflib_if_ioctl(if_t ifp, u_long command, caddr_t data)
 	return (err);
 }
 
+static int
+iflib_if_vf_status(if_t ifp, struct if_vf_status **statusp)
+{
+	if_ctx_t ctx;
+	int error;
+
+	ctx = if_getsoftc(ifp);
+	CTX_LOCK(ctx);
+	error = IFDI_VF_STATUS(ctx, statusp);
+	CTX_UNLOCK(ctx);
+	return (error);
+}
+
 static uint64_t
 iflib_if_get_counter(if_t ifp, ift_counter cnt)
 {
@@ -6020,6 +6034,9 @@ iflib_register(if_ctx_t ctx)
 	if_setdev(ifp, dev);
 	if_setinitfn(ifp, iflib_if_init);
 	if_setioctlfn(ifp, iflib_if_ioctl);
+	/* VF status describes children of an SR-IOV PF. */
+	if (!CTX_IS_VF(ctx))
+		if_setvfstatusfn(ifp, iflib_if_vf_status);
 #ifdef ALTQ
 	if_setstartfn(ifp, iflib_altq_if_start);
 	if_settransmitfn(ifp, iflib_altq_if_transmit);
diff --git a/sys/netlink/netlink_snl.h b/sys/netlink/netlink_snl.h
index ecc980387fa8..7800e2c50f49 100644
--- a/sys/netlink/netlink_snl.h
+++ b/sys/netlink/netlink_snl.h
@@ -325,6 +325,32 @@ snl_send_message(struct snl_state *ss, struct nlmsghdr *hdr)
 	return (send(ss->fd, hdr, sz, 0) == sz);
 }
 
+/* Ensure the receive buffer can hold the next complete Netlink message. */
+static inline int
+snl_grow_rxbuf_to_next_message(struct snl_state *ss)
+{
+	char *buf;
+	ssize_t len;
+
+	if (ss->off != ss->datalen)
+		return (0);
+	do {
+		len = recv(ss->fd, NULL, 0, MSG_PEEK | MSG_TRUNC);
+	} while (len < 0 && errno == EINTR);
+	if (len < 0)
+		return (errno);
+	if (len == 0)
+		return (EIO);
+	if ((size_t)len <= ss->bufsize)
+		return (0);
+	buf = realloc(ss->buf, (size_t)len);
+	if (buf == NULL)
+		return (ENOMEM);
+	ss->buf = buf;
+	ss->bufsize = (size_t)len;
+	return (0);
+}
+
 static inline uint32_t
 snl_get_seq(struct snl_state *ss)
 {
diff --git a/sys/netlink/netlink_snl_route_parsers.h b/sys/netlink/netlink_snl_route_parsers.h
index 495dee5ec862..bc34dbf0c9a7 100644
--- a/sys/netlink/netlink_snl_route_parsers.h
+++ b/sys/netlink/netlink_snl_route_parsers.h
@@ -180,6 +180,352 @@ static const struct snl_attr_parser _nla_p_ifgroups[] = {
 SNL_DECLARE_ATTR_PARSER_EXT(_ifgroups_parser, sizeof(char[IFNAMSIZ]), _nla_p_ifgroups, NULL);
 
 /* RTM_<NEW|DEL|GET>LINK message parser */
+_Static_assert(IFLAF_VF_MAX < 64,
+    "VF attribute numbers must fit the presence mask");
+
+enum snl_vf_driver_field_type {
+	SNL_VFDF_NONE = 0,
+	SNL_VFDF_BOOL,
+	SNL_VFDF_NUMBER,
+	SNL_VFDF_STRING,
+	SNL_VFDF_BINARY,
+};
+
+struct snl_parsed_vf_driver_field {
+	char			*name;
+	enum snl_vf_driver_field_type type;
+	bool			boolean;
+	uint64_t		number;
+	char			*string;
+	struct nlattr		*binary;
+};
+
+struct snl_parsed_vf_driver {
+	char			*name;
+	uint32_t		version;
+	bool			version_present;
+	struct snl_parray	fields;
+};
+
+struct snl_parsed_vf {
+	uint64_t		attrs;	/* Bits indexed by IFLAF_VF_*. */
+	uint64_t		min_tx_rate_bps;
+	uint64_t		max_tx_rate_bps;
+	uint32_t		index;
+	bool			index_present;
+	uint32_t		vlan_count;
+	uint32_t		vlan_limit;
+	uint16_t		tx_queue_count;
+	uint16_t		rx_queue_count;
+	uint16_t		vlan;
+	uint16_t		vlan_proto;
+	uint8_t			vlan_pcp;
+	bool			configured;
+	bool			initialized;
+	uint8_t			vlan_mode;
+	bool			allow_set_mac;
+	bool			allow_set_vlan;
+	bool			mac_anti_spoof;
+	bool			allow_promisc;
+	bool			traffic_allowed;
+	bool			fault_blocked;
+	bool			quarantined;
+	uint8_t			link_state_policy;
+	char			*api_version;
+	struct nlattr		*mac;
+	struct snl_parray	drivers;
+};
+
+static inline bool
+_snl_vf_get_index(struct snl_state *ss, struct nlattr *nla,
+    const void *arg __unused, void *target)
+{
+	struct snl_parsed_vf *vf = target;
+
+	if (!snl_attr_get_uint32(ss, nla, NULL, &vf->index))
+		return (false);
+	vf->index_present = true;
+	return (true);
+}
+
+static inline bool
+_cb_p_vf(struct snl_state *ss __unused, void *target)
+{
+	struct snl_parsed_vf *vf = target;
+
+	return (vf->index_present);
+}
+
+#define	_SNL_VF_ARG(_attr, _field)	((const void *)(uintptr_t)(	\
+	((uint32_t)(_attr) << 16) | offsetof(struct snl_parsed_vf, _field)))
+#define	_SNL_VF_ATTR(_arg)		((uintptr_t)(_arg) >> 16)
+#define	_SNL_VF_OFF(_arg)		((uintptr_t)(_arg) & 0xffff)
+
+#define	_SNL_VF_GETTER(_name, _getter)					\
+static inline bool							\
+_name(struct snl_state *ss, struct nlattr *nla, const void *arg,	\
+    void *target)							\
+{									\
+	struct snl_parsed_vf *vf = target;				\
+	void *value = (char *)target + _SNL_VF_OFF(arg);			\
+									\
+	if (!_getter(ss, nla, NULL, value))				\
+		return (false);						\
+	vf->attrs |= 1ULL << _SNL_VF_ATTR(arg);				\
+	return (true);							\
+}
+
+_SNL_VF_GETTER(_snl_vf_get_bool, snl_attr_get_bool)
+_SNL_VF_GETTER(_snl_vf_get_u8, snl_attr_get_uint8)
+_SNL_VF_GETTER(_snl_vf_get_u16, snl_attr_get_uint16)
+_SNL_VF_GETTER(_snl_vf_get_u32, snl_attr_get_uint32)
+_SNL_VF_GETTER(_snl_vf_get_u64, snl_attr_get_uint64)
+_SNL_VF_GETTER(_snl_vf_get_string, snl_attr_dup_string)
+_SNL_VF_GETTER(_snl_vf_get_nla, snl_attr_dup_nla)
+
+#undef _SNL_VF_GETTER
+
+#define	_SNL_VFDF_GETTER(_name, _getter, _type, _field)		\
+static inline bool							\
+_name(struct snl_state *ss, struct nlattr *nla, const void *arg,	\
+    void *target)							\
+{									\
+	struct snl_parsed_vf_driver_field *field = target;		\
+									\
+	if (field->type != SNL_VFDF_NONE ||				\
+	    !_getter(ss, nla, arg, &field->_field))			\
+		return (false);						\
+	field->type = _type;						\
+	return (true);							\
+}
+
+_SNL_VFDF_GETTER(_snl_vfdf_get_bool, snl_attr_get_bool, SNL_VFDF_BOOL,
+    boolean)
+_SNL_VFDF_GETTER(_snl_vfdf_get_number, snl_attr_get_uint64,
+    SNL_VFDF_NUMBER, number)
+_SNL_VFDF_GETTER(_snl_vfdf_get_string, snl_attr_dup_string,
+    SNL_VFDF_STRING, string)
+_SNL_VFDF_GETTER(_snl_vfdf_get_binary, snl_attr_dup_nla,
+    SNL_VFDF_BINARY, binary)
+
+#undef _SNL_VFDF_GETTER
+
+static inline bool
+_cb_p_vf_driver_field(struct snl_state *ss __unused, void *target)
+{
+	struct snl_parsed_vf_driver_field *field = target;
+
+	return (field->name != NULL);
+}
+
+#define	_OUT(_field)	offsetof(struct snl_parsed_vf_driver_field, _field)
+static const struct snl_attr_parser _nla_p_vf_driver_field[] = {
+	{ .type = IFLAF_VFDF_NAME, .off = _OUT(name),
+	    .cb = snl_attr_dup_string },
+	{ .type = IFLAF_VFDF_BOOL, .off = 0, .cb = _snl_vfdf_get_bool },
+	{ .type = IFLAF_VFDF_NUMBER, .off = 0,
+	    .cb = _snl_vfdf_get_number },
+	{ .type = IFLAF_VFDF_STRING, .off = 0,
+	    .cb = _snl_vfdf_get_string },
+	{ .type = IFLAF_VFDF_BINARY, .off = 0,
+	    .cb = _snl_vfdf_get_binary },
+};
+#undef _OUT
+SNL_DECLARE_ATTR_PARSER_EXT(_vf_driver_field_parser,
+    sizeof(struct snl_parsed_vf_driver_field), _nla_p_vf_driver_field,
+    _cb_p_vf_driver_field);
+
+static inline bool
+_snl_vfd_get_field_multi(struct snl_state *ss, struct nlattr *nla,
+    const void *arg __unused, void *target)
+{
+	struct snl_parsed_vf_driver_field *field;
+	struct snl_parray *fields = target;
+	struct nlattr *field_nla;
+	bool unknown_value;
+
+	field = snl_allocz(ss, sizeof(*field));
+	if (field == NULL ||
+	    !snl_parse_header(ss, NLA_DATA(nla), NLA_DATA_LEN(nla),
+	    &_vf_driver_field_parser, field))
+		return (false);
+	if (field->type != SNL_VFDF_NONE)
+		return (snl_parray_append(ss, fields, field, 8));
+
+	/* Ignore a field whose value type was added by a newer kernel. */
+	unknown_value = false;
+	NLA_FOREACH(field_nla, NLA_DATA(nla), NLA_DATA_LEN(nla)) {
+		if (NLA_TYPE(field_nla) > IFLAF_VFDF_MAX) {
+			unknown_value = true;
+			break;
+		}
+	}
+	return (unknown_value);
+}
*** 1737 LINES SKIPPED ***