git: 9d3064010590 - main - hwpmc_amd: add PerfMonV2 global-control path

From: Mitchell Horne <mhorne_at_FreeBSD.org>
Date: Mon, 28 Sep 2026 19:24:10 UTC
The branch main has been updated by mhorne:

URL: https://cgit.FreeBSD.org/src/commit/?id=9d30640105900a72437380636f70dd1872ca9e22

commit 9d30640105900a72437380636f70dd1872ca9e22
Author:     Andre Silva <andasilv@amd.com>
AuthorDate: 2026-09-28 19:07:22 +0000
Commit:     Mitchell Horne <mhorne@FreeBSD.org>
CommitDate: 2026-09-28 19:23:34 +0000

    hwpmc_amd: add PerfMonV2 global-control path
    
    Add support for AMD PerfMonV2 (Family 19h+) core counters, which need
    both the per-counter EVSEL enable bit and the global GLOBAL_CTL bit set
    to count. Detects PerfMonV2 at init and switches to v2-specific
    start/stop/interrupt handlers; older CPUs and L3/DF counters keep using
    the classic path unchanged.
    
    Adds a read-only sysctl, kern.hwpmc.amd_perfmon_v2, to report which path
    is active.
    
    Signed-off-by:  Andre Silva <andasilv@amd.com>
    Reviewed by:    Ali Mashtizadeh <ali@mashtizadeh.com>
    Sponsored by:   AMD
    Differential Revision:  https://reviews.freebsd.org/D58256
---
 sys/dev/hwpmc/hwpmc_amd.c | 507 ++++++++++++++++++++++++++++++++++++++++++++--
 sys/dev/hwpmc/hwpmc_amd.h |  11 +
 sys/dev/hwpmc/hwpmc_mod.c |  69 +++++++
 sys/sys/pmc.h             |   5 +
 4 files changed, 574 insertions(+), 18 deletions(-)

diff --git a/sys/dev/hwpmc/hwpmc_amd.c b/sys/dev/hwpmc/hwpmc_amd.c
index 77d4a4230e49..5de283946384 100644
--- a/sys/dev/hwpmc/hwpmc_amd.c
+++ b/sys/dev/hwpmc/hwpmc_amd.c
@@ -46,6 +46,7 @@
 #define	EXTERR_CATEGORY	EXTERR_CAT_HWPMC_AMD
 #include <sys/exterrvar.h>
 
+#include <machine/atomic.h>
 #include <machine/cpu.h>
 #include <machine/cpufunc.h>
 #include <machine/md_var.h>
@@ -63,9 +64,11 @@ struct amd_descr {
 	enum sub_class	pm_subclass; /* register subclass */
 };
 
+static struct amd_descr *amd_pmcdesc;
+
 static int amd_npmcs;
 static int amd_core_npmcs, amd_l3_npmcs, amd_df_npmcs, amd_umc_npmcs;
-static struct amd_descr *amd_pmcdesc;
+
 struct amd_event_code_map {
 	enum pmc_event	pe_ev;	 /* enum value */
 	uint16_t	pe_code; /* encoded event mask */
@@ -175,10 +178,20 @@ const struct amd_event_code_map amd_event_codes[] = {
 const int amd_event_codes_size = nitems(amd_event_codes);
 
 /*
- * Per-processor information
+ * This struct holds per-processor information.
+ *
+ * pc_global_mask marks the core counters that must run on this CPU now.
+ * pc_virtual_mask marks per-process counters that wait to start at the
+ * next context switch.  The switch clears this mask each time, so it
+ * never holds a system-wide counter.  System counters and per-process
+ * counters always use different row indices.  They never share one
+ * GLOBAL_CTL bit.
  */
 struct amd_cpu {
 	struct pmc_hw	*pc_amdpmcs;
+	volatile u_int	pc_global_mask;
+	volatile u_int	pc_virtual_mask;
+	volatile u_int	pc_gate_depth;
 };
 static struct amd_cpu **amd_pcpu;
 
@@ -193,6 +206,10 @@ static uint64_t amd_l3_extra_mask;
 static uint64_t amd_df_extra_mask;
 static uint64_t amd_umc_extra_mask;
 
+static bool amd_perfmon_v2;
+/* Mask with one bit for each core counter */
+static uint64_t amd_global_cntr_mask;
+
 SYSCTL_DECL(_kern_hwpmc);
 
 SYSCTL_U64(_kern_hwpmc, OID_AUTO, amd_core_extra_mask, CTLFLAG_RDTUN,
@@ -211,6 +228,10 @@ SYSCTL_U64(_kern_hwpmc, OID_AUTO, amd_umc_extra_mask, CTLFLAG_RDTUN,
     &amd_umc_extra_mask, 0,
     "Extra allowed bits in AMD UMC PMU control (override; default 0)");
 
+SYSCTL_BOOL(_kern_hwpmc, OID_AUTO, amd_perfmon_v2, CTLFLAG_RD,
+    &amd_perfmon_v2, 0,
+    "AMD PerfMonV2 global-control path selected (read-only)");
+
 static void
 amd_init_policy(void)
 {
@@ -249,6 +270,132 @@ amd_config_mask(enum sub_class subclass, uint64_t caps)
 	}
 }
 
+static __inline u_int
+amd_v2_counter_mask(int ri)
+{
+	KASSERT(ri >= 0 && ri < amd_core_npmcs,
+	    ("[amd,%d] illegal core row-index %d", __LINE__, ri));
+	return (1U << ri);
+}
+
+static __inline void
+amd_v2_assert_mask(u_int mask)
+{
+	KASSERT((mask & ~(u_int)amd_global_cntr_mask) == 0,
+	    ("[amd,%d] invalid GLOBAL_CTL mask %#x", __LINE__, mask));
+}
+
+static __inline void
+amd_v2_publish_mask(int cpu)
+{
+	struct amd_cpu *pac;
+	u_int mask;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] illegal CPU value %d", __LINE__, cpu));
+	pac = amd_pcpu[cpu];
+	KASSERT(pac != NULL,
+	    ("[amd,%d] null per-cpu, cpu %d", __LINE__, cpu));
+	mask = atomic_load_acq_int(&pac->pc_global_mask);
+	amd_v2_assert_mask(mask);
+	if (atomic_load_acq_int(&pac->pc_gate_depth) == 0)
+		wrmsr(AMD_PMC_GLOBAL_CTL, mask);
+}
+
+static void
+amd_v2_stage_virtual(int cpu, int ri)
+{
+	struct amd_cpu *pac;
+	u_int mask;
+
+	pac = amd_pcpu[cpu];
+	mask = amd_v2_counter_mask(ri);
+	atomic_set_int(&pac->pc_virtual_mask, mask);
+}
+
+static int
+amd_start_pmc_all_v2(int cpu)
+{
+	struct amd_cpu *pac;
+	u_int mask;
+
+	pac = amd_pcpu[cpu];
+	mask = atomic_load_acq_int(&pac->pc_virtual_mask);
+	if (mask == 0)
+		return (0);
+	amd_v2_assert_mask(mask);
+	atomic_set_int(&pac->pc_global_mask, mask);
+	amd_v2_publish_mask(cpu);
+	return (0);
+}
+
+/* Clear only the staged virtual bits.  System-wide counters keep running. */
+static int
+amd_stop_pmc_all_v2(int cpu)
+{
+	struct amd_cpu *pac;
+	u_int mask;
+
+	pac = amd_pcpu[cpu];
+	mask = atomic_readandclear_int(&pac->pc_virtual_mask);
+	if (mask == 0)
+		return (0);
+	amd_v2_assert_mask(mask);
+	atomic_clear_int(&pac->pc_global_mask, mask);
+	amd_v2_publish_mask(cpu);
+	return (0);
+}
+
+static void
+amd_v2_forget_core(int cpu, int ri, struct pmc *pm)
+{
+	struct amd_cpu *pac;
+	u_int mask;
+
+	pac = amd_pcpu[cpu];
+	mask = amd_v2_counter_mask(ri);
+	if (PMC_IS_VIRTUAL_MODE(PMC_TO_MODE(pm)))
+		atomic_clear_int(&pac->pc_virtual_mask, mask);
+	if ((atomic_load_acq_int(&pac->pc_global_mask) & mask) != 0) {
+		atomic_clear_int(&pac->pc_global_mask, mask);
+		amd_v2_publish_mask(cpu);
+	}
+}
+
+static __inline void
+amd_v2_disable_all(void)
+{
+	wrmsr(AMD_PMC_GLOBAL_CTL, 0);
+}
+
+static void
+amd_v2_freeze_core(int cpu)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] illegal CPU value %d", __LINE__, cpu));
+	atomic_add_int(&amd_pcpu[cpu]->pc_gate_depth, 1);
+	amd_v2_disable_all();
+}
+
+static void
+amd_v2_thaw_core(int cpu)
+{
+	u_int old;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] illegal CPU value %d", __LINE__, cpu));
+	old = atomic_fetchadd_int(&amd_pcpu[cpu]->pc_gate_depth, -1);
+	KASSERT(old > 0,
+	    ("AMD PMC gate-depth underflow on CPU %d", cpu));
+	if (old == 1) {
+		u_int mask;
+
+		mask = atomic_load_acq_int(&amd_pcpu[cpu]->pc_global_mask);
+		amd_v2_assert_mask(mask);
+		wrmsr(AMD_PMC_GLOBAL_CTL, mask);
+	}
+}
+
 /*
  * Read a PMC value from the MSR.
  */
@@ -344,6 +491,10 @@ amd_config_pmc(int cpu, int ri, struct pmc *pm)
 	    ("[amd,%d] pm=%p phw->pm=%p hwpmc not unconfigured",
 		__LINE__, pm, phw->phw_pmc));
 
+	if (amd_perfmon_v2 && pm == NULL && phw->phw_pmc != NULL &&
+	    amd_pmcdesc[ri].pm_subclass == PMC_AMD_SUB_CLASS_CORE)
+		amd_v2_forget_core(cpu, ri, phw->phw_pmc);
+
 	phw->phw_pmc = pm;
 	return (0);
 }
@@ -393,11 +544,10 @@ amd_switch_out(struct pmc_cpu *pc __pmcdbg_used,
 }
 
 /*
- * Check if a given PMC allocation is feasible.
+ * Check if an event can use row index 'ri'.
  */
 static int
-amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
-    const struct pmc_op_pmcallocate *a)
+amd_can_assign_pmc(int ri, struct pmc *pm, const struct pmc_op_pmcallocate *a)
 {
 	const struct pmc_descr *pd;
 	uint64_t allowed_unitmask, caps, config, unitmask;
@@ -409,14 +559,11 @@ amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
 
 	pd = &amd_pmcdesc[ri].pm_descr;
 
-	/* check class match */
 	if (pd->pd_class != a->pm_class)
 		return (EINVAL);
 
 	caps = pm->pm_caps;
 
-	PMCDBG2(MDP, ALL, 1,"amd-allocate ri=%d caps=0x%x", ri, caps);
-
 	/* Validate sub-class. */
 	if (amd_pmcdesc[ri].pm_subclass != a->pm_md.pm_amd.pm_amd_sub_class)
 		return (EINVAL);
@@ -434,18 +581,10 @@ amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
 			    "AMD PMU config has unsupported bits %#jx",
 			    (uintmax_t)(config & ~amd_config_mask(
 			    amd_pmcdesc[ri].pm_subclass, caps))));
-		pm->pm_md.pm_amd.pm_amd_evsel = config;
-		PMCDBG2(MDP, ALL, 2, "amd-allocate ri=%d -> config=0x%jx",
-		    ri, (uintmax_t)config);
 		return (0);
 	}
 
-	/*
-	 * Everything below this is for supporting older processors.
-	 */
 	pe = a->pm_ev;
-
-	/* map ev to the correct event mask code */
 	config = allowed_unitmask = 0;
 	for (i = 0; i < amd_event_codes_size; i++) {
 		if (amd_event_codes[i].pe_ev == pe) {
@@ -467,13 +606,55 @@ amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
 		    "AMD unitmask %#jx exceeds allowed mask %#jx",
 		    (uintmax_t)unitmask, (uintmax_t)allowed_unitmask));
 
+	return (0);
+}
+
+/*
+ * Check if a given PMC allocation is feasible.
+ */
+static int
+amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
+    const struct pmc_op_pmcallocate *a)
+{
+	uint64_t allowed_unitmask, caps, config, unitmask;
+	enum pmc_event pe;
+	int error, i;
+
+	error = amd_can_assign_pmc(ri, pm, a);
+	if (error != 0)
+		return (error);
+
+	caps = pm->pm_caps;
+	PMCDBG2(MDP, ALL, 1,"amd-allocate ri=%d caps=0x%x", ri, caps);
+
+	/* PMC_F_EV_PMU: config comes from pmu-events tables. */
+	if ((a->pm_flags & PMC_F_EV_PMU) != 0) {
+		config = a->pm_md.pm_amd.pm_amd_config;
+		pm->pm_md.pm_amd.pm_amd_evsel = config;
+		PMCDBG2(MDP, ALL, 2, "amd-allocate ri=%d -> config=0x%jx",
+		    ri, (uintmax_t)config);
+		return (0);
+	}
+
+	pe = a->pm_ev;
+	config = allowed_unitmask = 0;
+	for (i = 0; i < amd_event_codes_size; i++) {
+		if (amd_event_codes[i].pe_ev == pe) {
+			config =
+			    AMD_PMC_TO_EVENTMASK(amd_event_codes[i].pe_code);
+			allowed_unitmask =
+			    AMD_PMC_TO_UNITMASK(amd_event_codes[i].pe_mask);
+			break;
+		}
+	}
+
+	unitmask = a->pm_md.pm_amd.pm_amd_config & AMD_PMC_UNITMASK;
 	if (unitmask && (caps & PMC_CAP_QUALIFIER) != 0)
 		config |= unitmask;
 
 	if ((caps & PMC_CAP_THRESHOLD) != 0)
 		config |= a->pm_md.pm_amd.pm_amd_config & AMD_PMC_COUNTERMASK;
 
-	/* Set at least one of the 'usr' or 'os' caps. */
 	if ((caps & PMC_CAP_USER) != 0)
 		config |= AMD_PMC_USR;
 	if ((caps & PMC_CAP_SYSTEM) != 0)
@@ -488,7 +669,7 @@ amd_allocate_pmc(int cpu __unused, int ri, struct pmc *pm,
 	if ((caps & PMC_CAP_INTERRUPT) != 0)
 		config |= AMD_PMC_INT;
 
-	pm->pm_md.pm_amd.pm_amd_evsel = config; /* save config value */
+	pm->pm_md.pm_amd.pm_amd_evsel = config;
 
 	PMCDBG2(MDP, ALL, 2, "amd-allocate ri=%d -> config=0x%x", ri, config);
 
@@ -561,6 +742,68 @@ amd_start_pmc(int cpu __diagused, int ri, struct pmc *pm)
 	return (0);
 }
 
+/* Start one PMC.  pcd_start_all commits the virtual GLOBAL_CTL bits. */
+static int
+amd_start_pmc_v2(int cpu __diagused, int ri, struct pmc *pm)
+{
+	const struct amd_descr *pd;
+	enum pmc_mode mode;
+	uint64_t config;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < amd_npmcs,
+	    ("[amd,%d] illegal row-index %d", __LINE__, ri));
+
+	pd = &amd_pmcdesc[ri];
+	mode = PMC_TO_MODE(pm);
+
+	PMCDBG2(MDP, STA, 1, "amd-start-v2 cpu=%d ri=%d", cpu, ri);
+
+	if (pd->pm_subclass == PMC_AMD_SUB_CLASS_CORE &&
+	    PMC_IS_VIRTUAL_MODE(mode))
+		amd_v2_stage_virtual(cpu, ri);
+
+	/*
+	 * Drop a stale overflow status bit before the row goes live.
+	 * amd_stop_pmc_v2() waits only OVERFLOW_WAIT_COUNT microseconds
+	 * for the old occupant's in-flight NMI.  Group rotation reuses
+	 * rows often, so a surviving GLOBAL_STATUS bit would reach the
+	 * late handler after the row restarts.  The handler would then
+	 * blame this PMC: a false sample, plus a mid-window reload for
+	 * the wrong event.  With the bit clear, the late handler finds
+	 * no work and treats the NMI as spurious.  Legacy AMD has no
+	 * status plane to check, so it only relies on the bounded wait.
+	 */
+	if (pd->pm_subclass == PMC_AMD_SUB_CLASS_CORE &&
+	    PMC_IS_SAMPLING_MODE(mode) &&
+	    (rdmsr(AMD_PMC_GLOBAL_STATUS) & (1ULL << ri)) != 0) {
+		wrmsr(AMD_PMC_GLOBAL_STATUS_CLR, 1ULL << ri);
+
+		/*
+		 * That NMI may still be in flight.  No status bit is left
+		 * to claim it, so amd_intr_v2() would return 0 and the NMI
+		 * would go unhandled (machdep.panic_on_nmi).  Give the stray
+		 * absorber one credit on this CPU.
+		 */
+		if (DPCPU_GET(nmi_counter) == 0)
+			DPCPU_SET(nmi_counter, 1);
+	}
+
+	/* Enable EVSEL while the virtual slot's global bit is off. */
+	config = pm->pm_md.pm_amd.pm_amd_evsel | AMD_PMC_ENABLE;
+	wrmsr(pd->pm_evsel, config);
+
+	if (pd->pm_subclass == PMC_AMD_SUB_CLASS_CORE &&
+	    PMC_IS_SYSTEM_MODE(mode)) {
+		atomic_set_int(&amd_pcpu[cpu]->pc_global_mask,
+		    amd_v2_counter_mask(ri));
+		amd_v2_publish_mask(cpu);
+	}
+
+	return (0);
+}
+
 /*
  * Stop a PMC.
  */
@@ -613,6 +856,49 @@ amd_stop_pmc(int cpu __diagused, int ri, struct pmc *pm)
 	return (0);
 }
 
+/* Stop one PMC.  pcd_stop_all already zeroed the virtual GLOBAL_CTL bits. */
+static int
+amd_stop_pmc_v2(int cpu __diagused, int ri, struct pmc *pm)
+{
+	const struct amd_descr *pd;
+	enum pmc_mode mode;
+	int i;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < amd_npmcs,
+	    ("[amd,%d] illegal row-index %d", __LINE__, ri));
+
+	pd = &amd_pmcdesc[ri];
+	mode = PMC_TO_MODE(pm);
+
+	PMCDBG1(MDP, STO, 1, "amd-stop-v2 ri=%d", ri);
+
+	if (pd->pm_subclass == PMC_AMD_SUB_CLASS_CORE &&
+	    PMC_IS_SYSTEM_MODE(mode)) {
+		atomic_clear_int(&amd_pcpu[cpu]->pc_global_mask,
+		    amd_v2_counter_mask(ri));
+		amd_v2_publish_mask(cpu);
+	}
+
+	/* Disable EVSEL after the counter's global bit is off. */
+	wrmsr(pd->pm_evsel,
+	    pm->pm_md.pm_amd.pm_amd_evsel & ~AMD_PMC_ENABLE);
+
+	/* Wait out an in-flight overflow NMI.  The handler clears the status bit. */
+	if (pd->pm_subclass == PMC_AMD_SUB_CLASS_CORE &&
+	    PMC_IS_SAMPLING_MODE(mode)) {
+		for (i = 0; i < OVERFLOW_WAIT_COUNT; i++) {
+			if ((rdmsr(AMD_PMC_GLOBAL_STATUS) & (1ULL << ri)) == 0)
+				break;
+
+			DELAY(1);
+		}
+	}
+
+	return (0);
+}
+
 /*
  * Interrupt handler.  This function needs to return '1' if the
  * interrupt was this CPU's PMCs or '0' otherwise.  It is not allowed
@@ -733,6 +1019,109 @@ done:
 	return (retval);
 }
 
+/*
+ * This is the v2 interrupt handler: freeze counters; read GLOBAL_STATUS once;
+ * reload; thaw.
+ */
+static int
+amd_intr_v2(struct trapframe *tf)
+{
+	struct amd_cpu *pac;
+	struct pmc *pm;
+	pmc_value_t v;
+	uint64_t status, pending;
+	uint32_t active = 0, count = 0;
+	int i, error, retval, cpu;
+
+	cpu = curcpu;
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[amd,%d] out of range CPU %d", __LINE__, cpu));
+
+	PMCDBG3(MDP, INT, 1, "cpu=%d tf=%p um=%d", cpu, tf, TRAPF_USERMODE(tf));
+
+	retval = 0;
+	pac = amd_pcpu[cpu];
+
+	retval = pmc_ibs_intr(tf);
+	if (retval)
+		goto done;
+
+	amd_v2_freeze_core(cpu);
+
+	/* Read the overflow bitmap once. */
+	status = rdmsr(AMD_PMC_GLOBAL_STATUS);
+	status &= amd_global_cntr_mask;
+
+	/*
+	 * Count all active sampling PMCs, not just the ones that
+	 * overflowed.  Count the in-flight NMI too, its counter has
+	 * not overflowed yet.
+	 */
+	for (i = 0; i < amd_core_npmcs; i++) {
+		pm = pac->pc_amdpmcs[i].phw_pmc;
+		if (pm != NULL && PMC_IS_SAMPLING_MODE(PMC_TO_MODE(pm)))
+			active++;
+	}
+
+	/* ffsll() returns a 1-based bit index, or 0 when no bits are set. */
+	pending = status;
+	while ((i = ffsll(pending) - 1) != -1) {
+		pending &= ~(1ULL << i);
+
+		if ((pm = pac->pc_amdpmcs[i].phw_pmc) == NULL ||
+		    !PMC_IS_SAMPLING_MODE(PMC_TO_MODE(pm))) {
+			continue;
+		}
+
+		retval = 1;
+
+		if (pm->pm_state != PMC_STATE_RUNNING)
+			continue;
+
+		/* Reload the counter. */
+		v = pm->pm_sc.pm_reloadcount;
+		wrmsr(amd_pmcdesc[i].pm_perfctr,
+		    AMD_RELOAD_COUNT_TO_PERFCTR_VALUE(v));
+
+		/*
+		 * On a log failure, leave the PMC disabled. MI code restarts
+		 * it via pcd_start_pmc.
+		  */
+		error = pmc_process_interrupt(PMC_HR, pm, tf);
+		if (error != 0)
+			wrmsr(amd_pmcdesc[i].pm_evsel,
+			    pm->pm_md.pm_amd.pm_amd_evsel & ~AMD_PMC_ENABLE);
+	}
+
+	/* Acknowledge the overflow bits through GLOBAL_STATUS_CLR. */
+	wrmsr(AMD_PMC_GLOBAL_STATUS_CLR, status);
+
+	/* Thaw the core counters. */
+	amd_v2_thaw_core(cpu);
+
+	/*
+	 * An earlier NMI may have serviced this overflow. Absorb the stray
+	 * NMI.
+	 */
+	if (retval) {
+		DPCPU_SET(nmi_counter, min(2, active));
+	} else {
+		if ((count = DPCPU_GET(nmi_counter))) {
+			retval = 1;
+			DPCPU_SET(nmi_counter, --count);
+		}
+	}
+
+done:
+	if (retval)
+		counter_u64_add(pmc_stats.pm_intr_processed, 1);
+	else
+		counter_u64_add(pmc_stats.pm_intr_ignored, 1);
+
+	PMCDBG1(MDP, INT, 2, "retval=%d", retval);
+	return (retval);
+}
+
 /*
  * Describe a PMC.
  */
@@ -843,6 +1232,19 @@ amd_pcpu_init(struct pmc_mdep *md, int cpu)
 	pac->pc_amdpmcs = mallocarray(amd_npmcs, sizeof(*pac->pc_amdpmcs),
 	    M_PMC, M_WAITOK | M_ZERO);
 
+	if (amd_perfmon_v2) {
+		KASSERT(atomic_load_acq_int(&pac->pc_gate_depth) == 0,
+		    ("[amd,%d] nonzero initial gate depth on CPU %d",
+		    __LINE__, cpu));
+		KASSERT(atomic_load_acq_int(&pac->pc_global_mask) == 0,
+		    ("[amd,%d] nonzero initial desired mask on CPU %d",
+		    __LINE__, cpu));
+		KASSERT(atomic_load_acq_int(&pac->pc_virtual_mask) == 0,
+		    ("[amd,%d] nonzero initial virtual mask on CPU %d",
+		    __LINE__, cpu));
+		amd_v2_disable_all();
+	}
+
 	/*
 	 * Set the content of the hardware descriptors to a known
 	 * state and initialize pointers in the MI per-cpu descriptor.
@@ -883,6 +1285,15 @@ amd_pcpu_fini(struct pmc_mdep *md, int cpu)
 	if ((pac = amd_pcpu[cpu]) == NULL)
 		return (0);
 
+	if (amd_perfmon_v2) {
+		KASSERT(atomic_load_acq_int(&pac->pc_gate_depth) == 0,
+		    ("[amd,%d] nonzero gate depth on CPU %d", __LINE__, cpu));
+		KASSERT(atomic_load_acq_int(&pac->pc_global_mask) == 0,
+		    ("[amd,%d] nonzero desired mask on CPU %d", __LINE__, cpu));
+		KASSERT(atomic_load_acq_int(&pac->pc_virtual_mask) == 0,
+		    ("[amd,%d] nonzero virtual mask on CPU %d", __LINE__, cpu));
+		amd_v2_disable_all();
+	}
 	amd_pcpu[cpu] = NULL;
 
 #ifdef	HWPMC_DEBUG
@@ -910,6 +1321,30 @@ amd_pcpu_fini(struct pmc_mdep *md, int cpu)
 	return (0);
 }
 
+struct amd_v2_hwcheck_state {
+	volatile u_int avh_read_error;
+	volatile u_int avh_enabled;
+};
+
+static void
+amd_v2_hwcheck_cpu(void *arg)
+{
+	struct amd_v2_hwcheck_state *state;
+	uint64_t reg;
+	int error, i;
+
+	state = arg;
+	for (i = 0; i < amd_core_npmcs; i++) {
+		error = rdmsr_safe(amd_pmcdesc[i].pm_evsel, &reg);
+		if (error != 0) {
+			atomic_set_int(&state->avh_read_error, 1);
+			continue;
+		}
+		if ((reg & AMD_PMC_ENABLE) != 0)
+			atomic_set_int(&state->avh_enabled, 1);
+	}
+}
+
 /*
  * Check that the PMC hardware is safe to use.  First, we check that the PMCs
  * are not in use by firmware or another module.  Second, if none of the PMC
@@ -919,9 +1354,29 @@ amd_pcpu_fini(struct pmc_mdep *md, int cpu)
 static int
 amd_hwcheck(void)
 {
+	struct amd_v2_hwcheck_state state;
 	uint64_t reg;
 	int error, i;
 
+	if (amd_perfmon_v2) {
+		state.avh_read_error = 0;
+		state.avh_enabled = 0;
+		smp_rendezvous_cpus(all_cpus, smp_no_rendezvous_barrier,
+		    amd_v2_hwcheck_cpu, smp_no_rendezvous_barrier, &state);
+		if (state.avh_read_error != 0) {
+			printf("hwpmc: AMD PerfMonV2 EVSEL read failed on one "
+			    "or more CPUs!\n");
+			return (-1);
+		}
+		if (state.avh_enabled != 0) {
+			printf("hwpmc: PMCs maybe in use by firmware!\n");
+			printf("hwpmc: Disable the PMC use in the BIOS before "
+			    "loading\n");
+			return (-1);
+		}
+		return (0);
+	}
+
 	/*
 	 * Some PC vendors enable the core counters in firmware to track
 	 * performance.  The best guess is that this is being used to control
@@ -1048,6 +1503,10 @@ pmc_amd_initialize(void)
 			amd_umc_npmcs = EXTPERFMON_UMC_PMCS(regs[1]);
 			pmcs_per_umc = amd_umc_npmcs / popcntq(regs[2]);
 		}
+
+		/* EAX bit 0 holds the PerfMonV2 flag. */
+		if (EXTPERFMON_PERFMONV2(regs[0]) && family >= 0x19)
+			amd_perfmon_v2 = true;
 	}
 
 	/*
@@ -1095,6 +1554,9 @@ pmc_amd_initialize(void)
 	}
 	amd_npmcs = amd_core_npmcs;
 
+	if (amd_perfmon_v2)
+		amd_global_cntr_mask = (1ULL << amd_core_npmcs) - 1;
+
 	if ((amd_feature2 & AMDID2_PTSCEL2I) != 0) {
 		/* Enable the LLC/L3 counters */
 		for (i = 0; i < amd_l3_npmcs; i++) {
@@ -1223,6 +1685,15 @@ pmc_amd_initialize(void)
 	pmc_mdep->pmd_switch_in	= amd_switch_in;
 	pmc_mdep->pmd_switch_out = amd_switch_out;
 
+	/* For v2, override core control.  L3 and DF keep the classic path. */
+	if (amd_perfmon_v2) {
+		pcd->pcd_start_pmc = amd_start_pmc_v2;
+		pcd->pcd_stop_pmc  = amd_stop_pmc_v2;
+		pcd->pcd_start_all = amd_start_pmc_all_v2;
+		pcd->pcd_stop_all  = amd_stop_pmc_all_v2;
+		pmc_mdep->pmd_intr = amd_intr_v2;
+	}
+
 	pmc_mdep->pmd_npmc	+= amd_npmcs;
 
 	amd_init_policy();
diff --git a/sys/dev/hwpmc/hwpmc_amd.h b/sys/dev/hwpmc/hwpmc_amd.h
index 4a77ca210f97..4936ea620884 100644
--- a/sys/dev/hwpmc/hwpmc_amd.h
+++ b/sys/dev/hwpmc/hwpmc_amd.h
@@ -36,6 +36,7 @@
 #define	EXTPERFMON_CORE_PMCS(x)	((x) & 0x0F)
 #define	EXTPERFMON_DF_PMCS(x)	(((x) >> 10) & 0x3F)
 #define	EXTPERFMON_UMC_PMCS(x)	(((x) >> 16) & 0xFF)
+#define	EXTPERFMON_PERFMONV2(x)	((x) & 0x1)
 
 /* AMD K8 PMCs */
 #define	AMD_PMC_EVSEL_0		0xC0010000
@@ -63,6 +64,16 @@
 #define	AMD_PMC_CORE_DEFAULT	6
 #define	AMD_PMC_CORE_MAX	16
 
+/*
+ * These are the PerfMonV2 global-control MSRs (Fam 19h Zen3+ / 1Ah).
+ * They control core counters only.  L3 and DF counters keep the
+ * classic per-counter path.
+ */
+#define	AMD_PMC_GLOBAL_STATUS		0xC0000300	/* RO  */
+#define	AMD_PMC_GLOBAL_CTL		0xC0000301	/* RW  */
+#define	AMD_PMC_GLOBAL_STATUS_CLR	0xC0000302	/* WO  */
+/* A future LBR v2 change adds GLOBAL_STATUS.LBRS_FROZEN (bit 58) here. */
+
 #define	AMD_PMC_COUNTERMASK	0xFF000000
 #define AMD_PMC_PRECISERETIRE	(1ULL << 43) /* Only valid for PERF_CTL2 */
 #define	AMD_PMC_HOST		(1ULL << 41)
diff --git a/sys/dev/hwpmc/hwpmc_mod.c b/sys/dev/hwpmc/hwpmc_mod.c
index 6c405d53486b..1a333e137d0a 100644
--- a/sys/dev/hwpmc/hwpmc_mod.c
+++ b/sys/dev/hwpmc/hwpmc_mod.c
@@ -248,6 +248,9 @@ static void	pmc_post_callchain_callback(void);
 static void	pmc_process_allproc(struct pmc *pm);
 static void	pmc_process_csw_in(struct thread *td);
 static void	pmc_process_csw_out(struct thread *td);
+static void	pmc_process_csw_out_prepare(int cpu);
+static void	pmc_process_csw_start_all(int cpu);
+static void	pmc_process_csw_stop_all(int cpu);
 static void	pmc_process_exec(struct thread *td,
     struct pmckern_procexec *pk);
 static void	pmc_process_exit(void *arg, struct proc *p);
@@ -1424,6 +1427,63 @@ pmc_process_exec(struct thread *td, struct pmckern_procexec *pk)
 		pmc_destroy_process_descriptor(pp);
 }
 
+/*
+ * Execute optional context-switch batch operations for all classes.
+ */
+static void
+pmc_process_csw_start_all(int cpu)
+{
+	struct pmc_classdep *pcd;
+	u_int class;
+
+	for (class = 0; class < md->pmd_nclass; class++) {
+		pcd = &md->pmd_classdep[class];
+		if (pcd->pcd_start_all != NULL)
+			(void)pcd->pcd_start_all(cpu);
+	}
+}
+
+static void
+pmc_process_csw_stop_all(int cpu)
+{
+	struct pmc_classdep *pcd;
+	u_int class;
+
+	pmc_process_csw_out_prepare(cpu);
+
+	for (class = 0; class < md->pmd_nclass; class++) {
+		pcd = &md->pmd_classdep[class];
+		if (pcd->pcd_stop_all != NULL)
+			(void)pcd->pcd_stop_all(cpu);
+	}
+}
+
+/*
+ * Mark virtual PMCs stopped, before you close hardware gates.
+ */
+static void
+pmc_process_csw_out_prepare(int cpu)
+{
+	struct pmc *pm;
+	struct pmc_classdep *pcd;
+	u_int class;
+	int adjri;
+
+	for (class = 0; class < md->pmd_nclass; class++) {
+		pcd = &md->pmd_classdep[class];
+		if (pcd->pcd_stop_all == NULL)
+			continue;
+		for (adjri = 0; adjri < pcd->pcd_num; adjri++) {
+			pm = NULL;
+			(void)pcd->pcd_get_config(cpu, adjri, &pm);
+			if (pm == NULL ||
+			    !PMC_IS_VIRTUAL_MODE(PMC_TO_MODE(pm)))
+				continue;
+			pm->pm_pcpu_state[cpu].pps_cpustate = 0;
+		}
+	}
+}
+
 /*
  * Thread context switch IN.
  */
@@ -1579,6 +1639,9 @@ pmc_process_csw_in(struct thread *td)
 	 */
 	(void)(*md->pmd_switch_in)(pc, pp);
 
+	/* Commit all class PMC start updates at one boundary. */
+	pmc_process_csw_start_all(cpu);
+
 	critical_exit();
 }
 
@@ -1700,6 +1763,9 @@ pmc_process_csw_out(struct thread *td)
 
 	pc = pmc_pcpu[cpu];
 
+	/* Close shared class gates before any PMC stop or read. */
+	pmc_process_csw_stop_all(cpu);
+
 	/*
 	 * When a PMC gets unlinked from a target PMC, it will
 	 * be removed from the target's pp_pmc[] array.
@@ -5230,6 +5296,9 @@ pmc_process_exit(void *arg __unused, struct proc *p)
 
 	PMCDBG2(PRC,EXT,2, "process-exit proc=%p pmc-process=%p", p, pp);
 
+	/* Run the context-switch-out steps for process exit. */
+	pmc_process_csw_stop_all(cpu);
+
 	/*
 	 * The exiting process could be the target of some PMCs which will be
 	 * running on currently executing CPU.
diff --git a/sys/sys/pmc.h b/sys/sys/pmc.h
index 284e82561de5..be932f4289b7 100644
--- a/sys/sys/pmc.h
+++ b/sys/sys/pmc.h
@@ -1054,6 +1054,11 @@ struct pmc_classdep {
 	/* starting and stopping PMCs */
 	int (*pcd_start_pmc)(int _cpu, int _ri, struct pmc *_pm);
 	int (*pcd_stop_pmc)(int _cpu, int _ri, struct pmc *_pm);
+	/*
+	 * Optional context-switch batch operations.
+	 */
+	int (*pcd_start_all)(int _cpu);
+	int (*pcd_stop_all)(int _cpu);
 
 	/* description */
 	int (*pcd_describe)(int _cpu, int _ri, struct pmc_info *_pi,