git: fa78e1d0b15c - main - pci: Reserve bus numbers required by SR-IOV VFs
- Go to: [ bottom of page ] [ top of archives ] [ this month ]
Date: Tue, 29 Sep 2026 06:38:43 UTC
The branch main has been updated by kbowling:
URL: https://cgit.FreeBSD.org/src/commit/?id=fa78e1d0b15ca2867f92eaa980eda939e09cdc77
commit fa78e1d0b15ca2867f92eaa980eda939e09cdc77
Author: Kevin Bowling <kbowling@FreeBSD.org>
AuthorDate: 2026-08-03 14:33:51 +0000
Commit: Kevin Bowling <kbowling@FreeBSD.org>
CommitDate: 2026-09-29 06:25:45 +0000
pci: Reserve bus numbers required by SR-IOV VFs
Some firmware assigns only one bus number to each PCI-PCI bridge. This
prevents later SR-IOV VF enumeration when a VF routing ID falls on a bus
number already allocated to a sibling bridge.
Reserve only the additional bus numbers required by SR-IOV PFs.
Enumerate all directly attached functions before child drivers and
bridges attach, inspect their device_t objects for SR-IOV, and grow the
PCI bus resource through the highest possible VF routing ID.
First VF Offset and VF Stride may change when NumVFs changes. Probe
every valid NumVFs value and preserve the original setting. When the
upstream hierarchy uses ARI, temporarily enable the SR-IOV ARI Hierarchy
control in the lowest-numbered PF while sizing, then restore it. Scope
active-VF detection to each conventional PCI slot; an ARI bus remains
one slot-0 hierarchy. If firmware left VFs enabled on a device, do not
modify it and reserve only its active layout.
During runtime configuration, consult the PCI bus's owned resource
range rather than PCI-PCI bridge registers. This recognizes an existing
boot-time reservation beneath both PCI-PCI and host bridges and avoids a
second, overlapping bus-number allocation.
This avoids consuming bus numbers behind unrelated bridges. The runtime
allocation in pci_iov.c remains as a fallback when the boot-time range
cannot be enlarged. hw.pci.clear_buses remains useful when firmware
assigned a required number to another bridge before enumeration.
Validated the targeted implementation on an Intel E810-XXV behind a
non-ARI root port. With hw.pci.clear_buses=1 and no global reserve
tunable, the PF bridge received buses 1-2 while three unrelated bridges
each received one bus. A VF with First VF Offset 0x100 attached as
iavf0 at pci0:2:0:0, and detached cleanly.
Reviewed by: jhb, manpages (ziaee)
MFC after: 2 weeks
Sponsored by: BBOX.io
Differential Revision: https://reviews.freebsd.org/D58624
---
share/man/man4/pci.4 | 11 ++-
sys/dev/pci/pci.c | 189 +++++++++++++++++++++++++++++++++++++++++++
sys/dev/pci/pci_iov.c | 18 ++---
sys/dev/pci/pci_private.h | 3 +
sys/powerpc/ofw/ofw_pcibus.c | 3 +
5 files changed, 214 insertions(+), 10 deletions(-)
diff --git a/share/man/man4/pci.4 b/share/man/man4/pci.4
index c84051b1b988..e81d99a31839 100644
--- a/share/man/man4/pci.4
+++ b/share/man/man4/pci.4
@@ -22,7 +22,7 @@
.\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
.\" SUCH DAMAGE.
.\"
-.Dd August 6, 2026
+.Dd August 27, 2026
.Dt PCI 4
.Os
.Sh NAME
@@ -115,6 +115,15 @@ various platform-specific Host-PCI bridges,
and basic support for
.Tn PCI
VGA adapters.
+.Pp
+When PCI SR-IOV support is present, the
+.Nm
+driver makes a best-effort attempt at boot to reserve additional bus numbers
+required by virtual functions of directly attached physical functions.
+This reservation is independent of
+.Va hw.pci.clear_buses .
+If firmware assigned a required bus number to another bridge, enabling that
+tunable can make the bus number available for reservation.
.Sh IOCTLS
The following
.Xr ioctl 2
diff --git a/sys/dev/pci/pci.c b/sys/dev/pci/pci.c
index 3ed95ca2bf87..88a1bb376fca 100644
--- a/sys/dev/pci/pci.c
+++ b/sys/dev/pci/pci.c
@@ -4967,6 +4967,192 @@ pci_probe(device_t dev)
return (BUS_PROBE_GENERIC);
}
+#ifdef PCI_IOV
+/*
+ * Compute the maximum PCI bus number needed for the currently configured VFs.
+ */
+static int
+pci_iov_max_vf_bus_cur(device_t pf, int iov_pos, uint16_t num_vfs,
+ int *max_bus)
+{
+ uint64_t last_rid;
+ uint32_t pf_rid;
+ uint16_t rid_offset, rid_stride;
+
+ pf_rid = pci_get_rid(pf);
+ rid_offset = pci_read_config(pf, iov_pos + PCIR_SRIOV_VF_OFF, 2);
+ rid_stride = pci_read_config(pf, iov_pos + PCIR_SRIOV_VF_STRIDE, 2);
+ if (rid_offset == 0 || (num_vfs > 1 && rid_stride == 0))
+ return (EINVAL);
+ last_rid = (uint64_t)pf_rid + rid_offset +
+ (uint64_t)(num_vfs - 1) * rid_stride;
+ if (last_rid > UINT16_MAX)
+ return (ERANGE);
+ *max_bus = MAX(*max_bus, PCI_RID2BUS(last_rid));
+ return (0);
+}
+
+/*
+ * First VF Offset and VF Stride may change with NumVFs. Probe every valid
+ * NumVFs value while VF Enable is clear and return the largest VF bus.
+ */
+static int
+pci_iov_max_vf_bus(device_t pf, int iov_pos, int *max_bus)
+{
+ uint16_t ctl, num_vfs, original_num_vfs, total_vfs;
+ int error, pf_max_bus;
+
+ total_vfs = pci_read_config(pf, iov_pos + PCIR_SRIOV_TOTAL_VFS, 2);
+ if (total_vfs == 0 || total_vfs == UINT16_MAX)
+ return (0);
+
+ ctl = pci_read_config(pf, iov_pos + PCIR_SRIOV_CTL, 2);
+ original_num_vfs = pci_read_config(pf,
+ iov_pos + PCIR_SRIOV_NUM_VFS, 2);
+ pf_max_bus = pci_get_bus(pf);
+ error = 0;
+
+ if ((ctl & PCIM_SRIOV_VF_EN) != 0) {
+ /* NumVFs is writable only while VF Enable is clear. */
+ num_vfs = original_num_vfs;
+ if (num_vfs == 0 || num_vfs > total_vfs)
+ return (EINVAL);
+ return (pci_iov_max_vf_bus_cur(pf, iov_pos, num_vfs, max_bus));
+ }
+
+ for (num_vfs = total_vfs; num_vfs != 0; num_vfs--) {
+ pci_write_config(pf, iov_pos + PCIR_SRIOV_NUM_VFS,
+ num_vfs, 2);
+ error = pci_iov_max_vf_bus_cur(pf, iov_pos, num_vfs,
+ &pf_max_bus);
+ if (error != 0)
+ break;
+ }
+ pci_write_config(pf, iov_pos + PCIR_SRIOV_NUM_VFS,
+ original_num_vfs, 2);
+ if (error == 0)
+ *max_bus = MAX(*max_bus, pf_max_bus);
+ return (error);
+}
+
+/*
+ * Reserve bus numbers required by VFs after the bus has identified all of
+ * its children, but before their drivers attach. This is best effort;
+ * pci_iov.c retains a runtime allocation path for configurations that cannot
+ * be sized during boot.
+ */
+void
+pci_reserve_iov_buses(device_t dev, int busno)
+{
+ struct pci_iov_bus_group {
+ bool have_iov;
+ bool vfs_enabled;
+ bool ctl_changed;
+ device_t lowest_pf;
+ int lowest_iov_pos;
+ uint16_t saved_ctl;
+ } groups[PCI_SLOTMAX + 1], *group;
+ device_t child, pcib, *devlist;
+ struct pci_softc *sc;
+ rman_res_t start;
+ uint16_t ctl;
+ int devcount, error, group_slot, i, iov_pos, max_bus, slot;
+ bool ari, have_iov;
+
+ pcib = device_get_parent(dev);
+ sc = device_get_softc(dev);
+ max_bus = busno;
+ ari = PCIB_ARI_ENABLED(pcib);
+ bzero(groups, sizeof(groups));
+ have_iov = false;
+ error = device_get_children(dev, &devlist, &devcount);
+ if (error != 0)
+ return;
+ for (i = 0; i < devcount; i++) {
+ child = devlist[i];
+ if (pci_find_extcap(child, PCIZ_SRIOV, &iov_pos) != 0 ||
+ PCI_EXTCAP_VER(pci_read_config(child, iov_pos, 4)) != 1)
+ continue;
+ slot = pci_get_slot(child);
+ MPASS(!ari || slot == 0);
+ group_slot = ari ? 0 : slot;
+ group = &groups[group_slot];
+ ctl = pci_read_config(child, iov_pos + PCIR_SRIOV_CTL, 2);
+ if ((ctl & PCIM_SRIOV_VF_EN) != 0)
+ group->vfs_enabled = true;
+ if (!group->have_iov || pci_get_function(child) <
+ pci_get_function(group->lowest_pf)) {
+ group->have_iov = true;
+ have_iov = true;
+ group->lowest_pf = child;
+ group->lowest_iov_pos = iov_pos;
+ }
+ }
+ if (!have_iov) {
+ free(devlist, M_TEMP);
+ return;
+ }
+
+ /*
+ * The ARI Hierarchy bit is writable only in the lowest-numbered PF
+ * and controls the VF routing layout for all PFs on the device.
+ * Temporarily enable it while sizing an ARI hierarchy whose VFs are
+ * not already active.
+ */
+ if (ari) {
+ group = &groups[0];
+ MPASS(group->have_iov);
+ group->saved_ctl = pci_read_config(group->lowest_pf,
+ group->lowest_iov_pos + PCIR_SRIOV_CTL, 2);
+ ctl = group->saved_ctl;
+ ctl |= PCIM_SRIOV_ARI_EN;
+ group->ctl_changed = !group->vfs_enabled &&
+ ctl != group->saved_ctl;
+ if (group->ctl_changed)
+ pci_write_config(group->lowest_pf,
+ group->lowest_iov_pos + PCIR_SRIOV_CTL, ctl, 2);
+ }
+
+ for (i = 0; i < devcount; i++) {
+ child = devlist[i];
+ if (pci_find_extcap(child, PCIZ_SRIOV, &iov_pos) != 0 ||
+ PCI_EXTCAP_VER(pci_read_config(child, iov_pos, 4)) != 1)
+ continue;
+ slot = pci_get_slot(child);
+ MPASS(!ari || slot == 0);
+ group = &groups[ari ? 0 : slot];
+ ctl = pci_read_config(child, iov_pos + PCIR_SRIOV_CTL, 2);
+ if (group->vfs_enabled && (ctl & PCIM_SRIOV_VF_EN) == 0)
+ continue;
+ error = pci_iov_max_vf_bus(child, iov_pos, &max_bus);
+ if (error != 0 && bootverbose)
+ device_printf(child,
+ "cannot size SR-IOV function: %d\n", error);
+ }
+ if (ari && groups[0].ctl_changed) {
+ group = &groups[0];
+ pci_write_config(group->lowest_pf,
+ group->lowest_iov_pos + PCIR_SRIOV_CTL,
+ group->saved_ctl, 2);
+ }
+ free(devlist, M_TEMP);
+
+ start = rman_get_start(sc->sc_bus);
+ if (max_bus <= rman_get_end(sc->sc_bus))
+ return;
+ error = bus_adjust_resource(dev, sc->sc_bus, start, max_bus);
+ if (error != 0) {
+ device_printf(dev,
+ "failed to reserve bus numbers %ju-%d for SR-IOV: %d\n",
+ (uintmax_t)start, max_bus, error);
+ return;
+ }
+ if (bootverbose)
+ device_printf(dev, "reserved bus numbers %ju-%d for SR-IOV\n",
+ (uintmax_t)start, max_bus);
+}
+#endif
+
int
pci_attach_common(device_t dev)
{
@@ -5009,6 +5195,9 @@ pci_attach(device_t dev)
domain = pcib_get_domain(dev);
busno = pcib_get_bus(dev);
pci_add_children(dev, domain, busno);
+#ifdef PCI_IOV
+ pci_reserve_iov_buses(dev, busno);
+#endif
bus_attach_children(dev);
return (0);
}
diff --git a/sys/dev/pci/pci_iov.c b/sys/dev/pci/pci_iov.c
index 393582193318..8fd5a1672ee7 100644
--- a/sys/dev/pci/pci_iov.c
+++ b/sys/dev/pci/pci_iov.c
@@ -748,21 +748,21 @@ pci_iov_config(struct cdev *cdev, struct pci_iov_arg *arg)
last_rid = first_rid + (num_vfs - 1) * rid_stride;
if (pci_get_bus(dev) != PCI_RID2BUS(last_rid)) {
- device_t pcib = device_get_parent(bus);
- uint8_t secbus = pci_read_config(pcib, PCIR_SECBUS_1, 1);
- uint8_t subbus = pci_read_config(pcib, PCIR_SUBBUS_1, 1);
+ struct pci_softc *sc = device_get_softc(bus);
uint16_t vf_bus = PCI_RID2BUS(last_rid);
- /*
- * XXX: This should not be directly accessing the bridge registers and does
- * nothing to prevent some other device from releasing this bus number while
- * another PF is using it.
+ /*
+ * The PCI bus owns its entire sc_bus range in the parent
+ * resource manager. Boot-time SR-IOV sizing may already have
+ * extended that resource through vf_bus, so do not claim the
+ * same bus number a second time.
*/
- if (secbus == 0 || vf_bus < secbus || vf_bus > subbus) {
+ if (vf_bus < rman_get_start(sc->sc_bus) ||
+ vf_bus > rman_get_end(sc->sc_bus)) {
int rid = 0;
iov->iov_bus_res = bus_alloc_resource(bus, PCI_RES_BUS, &rid,
- vf_bus, vf_bus, 1, RF_ACTIVE);
+ vf_bus, vf_bus, 1, RF_ACTIVE);
if (iov->iov_bus_res == NULL) {
device_printf(dev,
"failed to allocate PCIe bus number for VFs\n");
diff --git a/sys/dev/pci/pci_private.h b/sys/dev/pci/pci_private.h
index bfa7df899322..737847576b09 100644
--- a/sys/dev/pci/pci_private.h
+++ b/sys/dev/pci/pci_private.h
@@ -116,6 +116,9 @@ pci_create_iov_child_t pci_create_iov_child_method;
void pci_add_children(device_t dev, int domain, int busno);
void pci_add_child(device_t bus, struct pci_devinfo *dinfo);
+#ifdef PCI_IOV
+void pci_reserve_iov_buses(device_t dev, int busno);
+#endif
/* Call after cold enumeration and before attaching the bus's children. */
void pcie_reconcile_link_mps(device_t bus);
device_t pci_iov_get_pf(device_t dev);
diff --git a/sys/powerpc/ofw/ofw_pcibus.c b/sys/powerpc/ofw/ofw_pcibus.c
index e6ddbdf0e835..a2d9efe07b59 100644
--- a/sys/powerpc/ofw/ofw_pcibus.c
+++ b/sys/powerpc/ofw/ofw_pcibus.c
@@ -147,6 +147,9 @@ ofw_pcibus_attach(device_t dev)
if (!ofw_devices_only)
ofw_pcibus_enum_bus(dev, domain, busno);
+#ifdef PCI_IOV
+ pci_reserve_iov_buses(dev, busno);
+#endif
pcie_reconcile_link_mps(dev);
bus_attach_children(dev);
return (0);