git: 5369f8e73eff - main - hwpmc: add MPERF/APERF MSR support for AMD/Intel CPUs

From: Mitchell Horne <mhorne_at_FreeBSD.org>
Date: Wed, 09 Sep 2026 14:43:02 UTC
The branch main has been updated by mhorne:

URL: https://cgit.FreeBSD.org/src/commit/?id=5369f8e73eff4772b133284012cb810e50d7c018

commit 5369f8e73eff4772b133284012cb810e50d7c018
Author:     Anderson Nascimento <anascime@amd.com>
AuthorDate: 2026-09-03 17:51:38 +0000
Commit:     Mitchell Horne <mhorne@FreeBSD.org>
CommitDate: 2026-09-09 14:42:58 +0000

    hwpmc: add MPERF/APERF MSR support for AMD/Intel CPUs
    
    Add support for reading the MPERF (MSR 0xE7) and APERF (MSR 0xE8)
    model-specific registers on AMD/Intel CPUs through hwpmc(4). These
    counters track maximum and actual performance frequency respectively,
    and are used to compute effective CPU frequency scaling independent
    of the nominal TSC rate. The name of the class was chosen as PERF
    because later support for other PERF MSRs can be added to the same
    class.
    
    Extend libpmc(3) to expose the AMD/Intel MPERF/APERF counters added
    to hwpmc(4) in the companion kernel change, so userland consumers
    (pmcstat(8), etc.) can allocate and read these events by name.
    
    Document the new PERF class and its MPERF/APERF counters in a new
    pmc.perf.3 manual page, describing their semantics and how to read
    them via pmc(3) and pmcstat(8).
    
    Bump PMC_VERSION_MINOR.
    
    Signed-off-by:  Anderson Nascimento <anascime@amd.com>
    Reviewed by:    Ali Mashtizadeh <ali@mashtizadeh.com>, mhorne
    Reviewed by:    ziaee (manpages)
    Sponsored by:   AMD
    Differential Revision:  https://reviews.freebsd.org/D58647
---
 lib/libpmc/Makefile          |   1 +
 lib/libpmc/libpmc.c          |  31 ++++
 lib/libpmc/pmc.3             |   8 +
 lib/libpmc/pmc.perf.3        | 132 ++++++++++++++++
 sys/amd64/include/pmc_mdep.h |   2 +
 sys/conf/files.x86           |   1 +
 sys/dev/hwpmc/hwpmc_amd.c    |  16 ++
 sys/dev/hwpmc/hwpmc_intel.c  |  18 +++
 sys/dev/hwpmc/hwpmc_perf.c   | 360 +++++++++++++++++++++++++++++++++++++++++++
 sys/dev/hwpmc/hwpmc_perf.h   |  28 ++++
 sys/dev/hwpmc/pmc_events.h   |  13 +-
 sys/modules/hwpmc/Makefile   |   1 +
 sys/sys/pmc.h                |   5 +-
 13 files changed, 613 insertions(+), 3 deletions(-)

diff --git a/lib/libpmc/Makefile b/lib/libpmc/Makefile
index d86191cc35cb..1540e9c8417c 100644
--- a/lib/libpmc/Makefile
+++ b/lib/libpmc/Makefile
@@ -77,6 +77,7 @@ MAN+=	pmc.iaf.3
 MAN+=	pmc.ibs.3
 MAN+=	pmc.ivybridge.3
 MAN+=	pmc.ivybridgexeon.3
+MAN+=	pmc.perf.3
 MAN+=	pmc.rapl.3
 MAN+=	pmc.sandybridge.3
 MAN+=	pmc.sandybridgeuc.3
diff --git a/lib/libpmc/libpmc.c b/lib/libpmc/libpmc.c
index 5fc4630540fc..41a75ac694de 100644
--- a/lib/libpmc/libpmc.c
+++ b/lib/libpmc/libpmc.c
@@ -60,6 +60,8 @@ static int tsc_allocate_pmc(enum pmc_event _pe, char *_ctrspec,
     struct pmc_op_pmcallocate *_pmc_config);
 static int rapl_allocate_pmc(enum pmc_event _pe, char *_ctrspec,
     struct pmc_op_pmcallocate *_pmc_config);
+static int perf_allocate_pmc(enum pmc_event _pe, char *_ctrspec,
+    struct pmc_op_pmcallocate *_pmc_config);
 #endif
 #if defined(__arm__)
 static int armv7_allocate_pmc(enum pmc_event _pe, char *_ctrspec,
@@ -139,6 +141,7 @@ struct pmc_class_descr {
 PMC_CLASSDEP_TABLE(iaf, IAF);
 PMC_CLASSDEP_TABLE(k8, K8);
 PMC_CLASSDEP_TABLE(ibs, IBS);
+PMC_CLASSDEP_TABLE(perf, PERF);
 PMC_CLASSDEP_TABLE(armv7, ARMV7);
 PMC_CLASSDEP_TABLE(armv8, ARMV8);
 PMC_CLASSDEP_TABLE(cmn600_pmu, CMN600_PMU);
@@ -216,6 +219,7 @@ PMC_CLASS_TABLE_DESC(k8, K8, k8, k8);
 PMC_CLASS_TABLE_DESC(ibs, IBS, ibs, ibs);
 PMC_CLASS_TABLE_DESC(tsc, TSC, tsc, tsc);
 PMC_CLASS_TABLE_DESC(rapl, RAPL, rapl, rapl);
+PMC_CLASS_TABLE_DESC(perf, PERF, perf, perf);
 #endif
 #if	defined(__arm__)
 PMC_CLASS_TABLE_DESC(cortex_a8, ARMV7, cortex_a8, armv7);
@@ -903,6 +907,22 @@ rapl_allocate_pmc(enum pmc_event pe, char *ctrspec,
 
 	return (0);
 }
+
+static int
+perf_allocate_pmc(enum pmc_event pe, char *ctrspec,
+    struct pmc_op_pmcallocate *pmc_config)
+{
+	if (pe < PMC_EV_PERF_FIRST || pe > PMC_EV_PERF_LAST)
+		return (-1);
+
+	/* PERF events must be unqualified. */
+	if (ctrspec != NULL && *ctrspec != '\0')
+		return (-1);
+
+	pmc_config->pm_caps |= PMC_CAP_READ;
+
+	return (0);
+}
 #endif
 
 static struct pmc_event_alias generic_aliases[] = {
@@ -1463,6 +1483,10 @@ pmc_event_names_of_class(enum pmc_class cl, const char ***eventnames,
 		ev = rapl_event_table;
 		count = PMC_EVENT_TABLE_SIZE(rapl);
 		break;
+	case PMC_CLASS_PERF:
+		ev = perf_event_table;
+		count = PMC_EVENT_TABLE_SIZE(perf);
+		break;
 	case PMC_CLASS_K8:
 		ev = k8_event_table;
 		count = PMC_EVENT_TABLE_SIZE(k8);
@@ -1675,6 +1699,10 @@ pmc_init(void)
 			pmc_class_table[n++] = &rapl_class_table_descr;
 			break;
 
+		case PMC_CLASS_PERF:
+			pmc_class_table[n++] = &perf_class_table_descr;
+			break;
+
 		case PMC_CLASS_K8:
 			pmc_class_table[n++] = &k8_class_table_descr;
 			break;
@@ -1950,6 +1978,9 @@ _pmc_name_of_event(enum pmc_event pe, enum pmc_cputype cpu)
 	} else if (pe >= PMC_EV_RAPL_FIRST && pe <= PMC_EV_RAPL_LAST) {
 		ev = rapl_event_table;
 		evfence = rapl_event_table + PMC_EVENT_TABLE_SIZE(rapl);
+	} else if (pe >= PMC_EV_PERF_FIRST && pe <= PMC_EV_PERF_LAST) {
+		ev = perf_event_table;
+		evfence = perf_event_table + PMC_EVENT_TABLE_SIZE(perf);
 	} else if ((int)pe >= PMC_EV_SOFT_FIRST && (int)pe <= PMC_EV_SOFT_LAST) {
 		ev = soft_event_table;
 		evfence = soft_event_table + soft_event_info.pm_nevent;
diff --git a/lib/libpmc/pmc.3 b/lib/libpmc/pmc.3
index b04c2d39cf69..bb051b87ec13 100644
--- a/lib/libpmc/pmc.3
+++ b/lib/libpmc/pmc.3
@@ -233,6 +233,12 @@ Family 10h and above.
 Programmable hardware counters present in
 .Tn "AMD Athlon64"
 CPUs.
+.It Li PMC_CLASS_PERF
+MPERF and APERF counters present in
+.Tn AMD
+and
+.Tn Intel
+CPUs.
 .It Li PMC_CLASS_RAPL
 RAPL energy counters present in
 .Tn AMD
@@ -505,6 +511,7 @@ following manual pages:
 .It Li PMC_CLASS_IBS    Ta Xr pmc.ibs 3
 .It Li PMC_CLASS_K8     Ta Xr pmc.amd 3
 .It Li PMC_CLASS_RAPL   Ta Xr pmc.rapl 3
+.It Li PMC_CLASS_PERF   Ta Xr pmc.perf 3
 .It Li PMC_CLASS_TSC    Ta Xr pmc.tsc 3
 .El
 .Ss Event Name Aliases
@@ -558,6 +565,7 @@ Doing otherwise is unsupported.
 .Xr pmc.ibs 3 ,
 .Xr pmc.ivybridge 3 ,
 .Xr pmc.ivybridgexeon 3 ,
+.Xr pmc.perf 3 ,
 .Xr pmc.rapl 3 ,
 .Xr pmc.sandybridge 3 ,
 .Xr pmc.sandybridgeuc 3 ,
diff --git a/lib/libpmc/pmc.perf.3 b/lib/libpmc/pmc.perf.3
new file mode 100644
index 000000000000..2404abd53555
--- /dev/null
+++ b/lib/libpmc/pmc.perf.3
@@ -0,0 +1,132 @@
+.\"
+.\" Copyright (c) 2026 Advanced Micro Devices, Inc.
+.\"
+.\" SPDX-License-Identifier: BSD-2-Clause
+.\"
+.Dd August 5, 2026
+.Dt PMC.PERF 3
+.Os
+.Sh NAME
+.Nm pmc.perf
+.Nd measurements using the MPERF/APERF effective frequency counters
+.Sh LIBRARY
+.Lb libpmc
+.Sh SYNOPSIS
+.In pmc.h
+.Sh DESCRIPTION
+AMD/Intel processors that implement the EffFreq
+.Pq Dq Effective Frequency Interface
+feature report two fixed-counters machine specific registers,
+.Va MPERF
+.Pq Dq Max Performance Frequency Clock Count
+and
+.Va APERF
+.Pq Dq Actual Performance Frequency Clock Count ,
+that together allow the average effective CPU frequency over an
+interval to be computed.
+.Va MPERF
+is incremented by hardware at the P0 frequency.
+The
+.Va APERF
+register increments in proportion to the actual number of core
+clock cycles.
+Both registers only increment while the core is in C0.
+The ratio of
+.Va APERF
+to
+.Va MPERF
+scaled by the processor's nominal frequency, yields the
+average effective frequency.
+.Pp
+The
+.Dv PMC_CLASS_PERF
+class exposes these two MSRs as read-only 64-bit counters that may
+only be allocated in system-wide counting mode
+.Pq Dv PMC_MODE_SC.
+Both counters expose the unmodified contents of the underlying MSRs.
+.Pp
+It reflects the free-running count of the underlying MSR since the
+processor was last reset, and continues to accumulate across
+allocate/release cycles.
+.Ss PMC Features
+AMD/Intel PERF supports the following capabilities.
+.Bl -column "PMC_CAP_THRESHOLD" "Support"
+.It Sy Capability Ta Sy Support
+.It Dv PMC_CAP_CASCADE Ta \&No
+.It Dv PMC_CAP_EDGE Ta \&No
+.It Dv PMC_CAP_INTERRUPT Ta \&No
+.It Dv PMC_CAP_INVERT Ta \&No
+.It Dv PMC_CAP_PRECISE Ta \&No
+.It Dv PMC_CAP_READ Ta Yes
+.It Dv PMC_CAP_SYSTEM Ta \&No
+.It Dv PMC_CAP_TAGGING Ta \&No
+.It Dv PMC_CAP_THRESHOLD Ta \&No
+.It Dv PMC_CAP_USER Ta \&No
+.It Dv PMC_CAP_WRITE Ta \&No
+.El
+.Pp
+By default AMD/Intel PERF enables the read flag.
+.Pp
+PERF events do not support further event qualifiers.
+.Ss Event Specifiers
+The following event names are supported:
+.Bl -tag -width "perf-mperf / mperf"
+.It Cm perf-mperf , Cm mperf
+The Max Performance Frequency Clock Count
+MSR of the core the counter is bound to.
+.It Cm perf-aperf , Cm aperf
+The Actual Performance Frequency Clock Count
+MSR of the core the counter is bound to.
+.El
+.Ss Counter Scope
+.Va MPERF
+and
+.Va APERF
+are per-core registers: each core has its own independent pair of
+MSRs, and a PMC of this class always reads the value local to the
+CPU it is bound to.
+Readings from different cores must not be combined or averaged
+without accounting for each core's independent P-state history.
+.Ss Availability
+Availability of the
+.Dv PMC_CLASS_PERF
+class depends on both of the following being true:
+.Bl -enum
+.It
+The processor advertises EffFreq
+.Pq Dq Effective Frequency Interface
+support in CPUID 6 leaf 0000_0006H bit 0 ECX.
+.It
+The kernel verified, at boot time, that the MSRs actually increment.
+Some virtualized environments advertise the CPUID bit without
+functional MSR support.
+.El
+.Pp
+If either condition does not hold, the
+.Dv PMC_CLASS_PERF
+class is not registered and does not appear in the output of
+.Xr pmccontrol 8
+or
+.Xr pmcstat 8 .
+.Sh EXAMPLES
+Sample both counters on CPU 0 for five seconds:
+.Bd -literal -offset indent
+pmcstat -c 0 -s perf-mperf -s perf-aperf -w 5
+.Ed
+.Sh SEE ALSO
+.Xr pmc 3 ,
+.Xr pmc.tsc 3 ,
+.Xr pmc_allocate 3 ,
+.Xr pmc_read 3 ,
+.Xr pmclog 3 ,
+.Xr hwpmc 4 ,
+.Xr pmcstat 8
+.Sh HISTORY
+The
+.Dv PMC_CLASS_PERF
+class first appeared in
+.Fx 16.0 .
+.Sh AUTHORS
+AMD/Intel PERF support and this manual page were written by
+.An Anderson Nascimento Aq Mt anascime@amd.com
+and sponsored by AMD, Inc.
diff --git a/sys/amd64/include/pmc_mdep.h b/sys/amd64/include/pmc_mdep.h
index 2adc16c05cdd..8eccff3dfa30 100644
--- a/sys/amd64/include/pmc_mdep.h
+++ b/sys/amd64/include/pmc_mdep.h
@@ -42,6 +42,7 @@ struct pmc_mdep;
 #include <dev/hwpmc/hwpmc_amd.h>
 #include <dev/hwpmc/hwpmc_core.h>
 #include <dev/hwpmc/hwpmc_ibs.h>
+#include <dev/hwpmc/hwpmc_perf.h>
 #include <dev/hwpmc/hwpmc_rapl.h>
 #include <dev/hwpmc/hwpmc_tsc.h>
 #include <dev/hwpmc/hwpmc_uncore.h>
@@ -66,6 +67,7 @@ struct pmc_mdep;
  * TSC		The timestamp counter
  * K8		AMD Athlon64 and Opteron PMCs in 64 bit mode.
  * IBS		AMD IBS
+ * PERF		AMD/Intel MPERF and APERF
  * IAP		Intel Core/Core2/Atom CPUs in 64 bits mode.
  * IAF		Intel fixed-function PMCs in Core2 and later CPUs.
  * UCP		Intel Uncore programmable PMCs.
diff --git a/sys/conf/files.x86 b/sys/conf/files.x86
index a5c85cd08955..bb52975598ab 100644
--- a/sys/conf/files.x86
+++ b/sys/conf/files.x86
@@ -121,6 +121,7 @@ dev/hptrr/hptrr_osm_bsd.c	optional	hptrr
 dev/hptrr/hptrr_config.c	optional	hptrr
 dev/hptrr/$M-elf.hptrr_lib.o	optional	hptrr
 dev/hwpmc/hwpmc_amd.c		optional	hwpmc
+dev/hwpmc/hwpmc_perf.c		optional	hwpmc
 dev/hwpmc/hwpmc_ibs.c		optional	hwpmc
 dev/hwpmc/hwpmc_intel.c		optional	hwpmc
 dev/hwpmc/hwpmc_core.c		optional	hwpmc
diff --git a/sys/dev/hwpmc/hwpmc_amd.c b/sys/dev/hwpmc/hwpmc_amd.c
index 5c4b14ad1003..c75bb9d0d14c 100644
--- a/sys/dev/hwpmc/hwpmc_amd.c
+++ b/sys/dev/hwpmc/hwpmc_amd.c
@@ -1144,6 +1144,17 @@ pmc_amd_initialize(void)
 		nclasses = 3;
 	}
 
+	/*
+	 * Detect support for MPERF and APERF MSRs. tsc_perf_stat is set by the
+	 * kernel's generic TSC initialization (start_TSC(), called at boot via
+	 * cpu_startup() -> startrtclock()), not by hwpmc's TSC PMC class. It is
+	 * set only after confirming both MSRs actually increment (some emulators
+	 * expose the CPUID bit without real MSR support).
+	 */
+	if ((cpu_power_ecx & CPUID_PERF_STAT) && (tsc_perf_stat == 1)) {
+		nclasses++;
+	}
+
 	pmc_mdep = pmc_mdep_alloc(nclasses);
 
 	ncpus = pmc_cpu_max();
@@ -1193,6 +1204,9 @@ pmc_amd_initialize(void)
 			goto error;
 	}
 
+	/* Initialize PERF class. */
+	pmc_perf_initialize(pmc_mdep, ncpus, nclasses - 1);
+
 	/* RAPL takes the reserved last slot; drop it if the probe fails. */
 	error = pmc_rapl_initialize(pmc_mdep, ncpus, pmc_mdep->pmd_nclass - 1);
 	if (error != 0)
@@ -1218,6 +1232,8 @@ pmc_amd_finalize(struct pmc_mdep *md)
 
 	pmc_tsc_finalize(md);
 
+	pmc_perf_finalize(md);
+
 	for (int i = 0; i < pmc_cpu_max(); i++)
 		KASSERT(amd_pcpu[i] == NULL,
 		    ("[amd,%d] non-null pcpu cpu %d", __LINE__, i));
diff --git a/sys/dev/hwpmc/hwpmc_intel.c b/sys/dev/hwpmc/hwpmc_intel.c
index f2cca80bff21..68f10173ddb4 100644
--- a/sys/dev/hwpmc/hwpmc_intel.c
+++ b/sys/dev/hwpmc/hwpmc_intel.c
@@ -280,6 +280,17 @@ pmc_intel_initialize(void)
 		return (NULL);
 	}
 
+	/*
+	 * Detect support for MPERF and APERF MSRs. tsc_perf_stat is set by the
+	 * kernel's generic TSC initialization (start_TSC(), called at boot via
+	 * cpu_startup() -> startrtclock()), not by hwpmc's TSC PMC class. It is
+	 * set only after confirming both MSRs actually increment (some emulators
+	 * expose the CPUID bit without real MSR support).
+	 */
+	if ((cpu_power_ecx & CPUID_PERF_STAT) && (tsc_perf_stat == 1)) {
+		nclasses++;
+	}
+
 	/* Reserve one extra class slot for the optional RAPL counters. */
 	nclasses++;
 
@@ -338,6 +349,11 @@ pmc_intel_initialize(void)
 		break;
 	}
 
+	if (error == 0) {
+		/* Initialize PERF class. */
+		pmc_perf_initialize(pmc_mdep, ncpus, nclasses - 1);
+	}
+
 	if (error == 0 &&
 	    pmc_rapl_initialize(pmc_mdep, ncpus, pmc_mdep->pmd_nclass - 1) != 0)
 		pmc_mdep->pmd_nclass--;
@@ -359,6 +375,8 @@ pmc_intel_finalize(struct pmc_mdep *md)
 
 	pmc_core_finalize(md);
 
+	pmc_perf_finalize(md);
+
 	/*
 	 * Uncore.
 	 */
diff --git a/sys/dev/hwpmc/hwpmc_perf.c b/sys/dev/hwpmc/hwpmc_perf.c
new file mode 100644
index 000000000000..c6f271947829
--- /dev/null
+++ b/sys/dev/hwpmc/hwpmc_perf.c
@@ -0,0 +1,360 @@
+/*
+ * Copyright (c) 2026 Advanced Micro Devices, Inc.
+ *
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * AMD/Intel MPERF and APERF MSRs counters exposed as an hwpmc(4) PMC class.
+ *
+ * Read-only, system-scope (PMC_MODE_SC), 64-bit counters reporting
+ * MPERF and APERF MSR values.
+ */
+
+#include <sys/param.h>
+#include <sys/pmc.h>
+#include <sys/pmckern.h>
+#include <sys/priv.h>
+
+#include <machine/specialreg.h>
+
+#define PERF_CAPS	PMC_CAP_READ
+
+struct perf_descr {
+	struct pmc_descr pm_descr;  /* "base class" */
+};
+
+static const struct perf_descr perf_pmcdesc[PERF_NPMCS] = {
+	{
+		.pm_descr = {
+			.pd_name  = "MPERF",
+			.pd_class = PMC_CLASS_PERF,
+			.pd_caps  = PERF_CAPS,
+			.pd_width = 64
+		},
+	},
+	{
+		.pm_descr = {
+			.pd_name  = "APERF",
+			.pd_class = PMC_CLASS_PERF,
+			.pd_caps  = PERF_CAPS,
+			.pd_width = 64
+		}
+	}
+};
+
+struct perf_cpu {
+	struct pmc_hw tc_hw[PERF_NPMCS];
+};
+
+static struct perf_cpu **perf_pcpu;
+static int perf_classindex;
+
+static int
+perf_allocate_pmc(int cpu __diagused, int ri __diagused,
+	    struct pmc *pm __unused, const struct pmc_op_pmcallocate *a)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS,
+	    ("[perf,%d] illegal row index %d", __LINE__, ri));
+
+	if (a->pm_class != PMC_CLASS_PERF)
+		return (EINVAL);
+
+	if ((a->pm_ev < PMC_EV_PERF_FIRST || a->pm_ev > PMC_EV_PERF_LAST) ||
+	    a->pm_mode != PMC_MODE_SC)
+		return (EINVAL);
+
+	/*
+	 * Allows the PERF class to be accessed only by privileged users.
+	 * Frequency is an indirect power proxy and could be abused by
+	 * attacks like Hertzbleed.
+	 */
+
+	if (priv_check(curthread, PRIV_PMC_SYSTEM) != 0)
+		return (EPERM);
+
+	if ((a->pm_caps & PERF_CAPS) == 0)
+		return (EINVAL);
+	if ((a->pm_caps & ~PERF_CAPS) != 0)
+		return (EPERM);
+
+	switch (ri) {
+	case PERF_MPERF:
+		if (a->pm_ev != PMC_EV_PERF_MPERF)
+			return (EINVAL);
+		break;
+	case PERF_APERF:
+		if (a->pm_ev != PMC_EV_PERF_APERF)
+			return (EINVAL);
+		break;
+	default:
+		return (EINVAL);
+	}
+
+	return (0);
+}
+
+static int
+perf_config_pmc(int cpu, int ri, struct pmc *pm)
+{
+	struct pmc_hw *phw;
+
+	PMCDBG3(MDP, CFG, 1, "cpu=%d ri=%d pm=%p", cpu, ri, pm);
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	phw = &perf_pcpu[cpu]->tc_hw[ri];
+
+	KASSERT(pm == NULL || phw->phw_pmc == NULL,
+	    ("[perf,%d] pm=%p phw->pm=%p hwpmc not unconfigured", __LINE__,
+	    pm, phw->phw_pmc));
+
+	phw->phw_pmc = pm;
+
+	return (0);
+}
+
+static int
+perf_describe(int cpu, int ri, struct pmc_info *pi, struct pmc **ppmc)
+{
+	const struct perf_descr *pd;
+	struct pmc_hw *phw;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	phw = &perf_pcpu[cpu]->tc_hw[ri];
+	pd  = &perf_pmcdesc[ri];
+
+	strlcpy(pi->pm_name, pd->pm_descr.pd_name, sizeof(pi->pm_name));
+	pi->pm_class = pd->pm_descr.pd_class;
+
+	if (phw->phw_state & PMC_PHW_FLAG_IS_ENABLED) {
+		pi->pm_enabled = TRUE;
+		*ppmc = phw->phw_pmc;
+	} else {
+		pi->pm_enabled = FALSE;
+		*ppmc = NULL;
+	}
+
+	return (0);
+}
+
+static int
+perf_get_config(int cpu, int ri, struct pmc **ppm)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	*ppm = perf_pcpu[cpu]->tc_hw[ri].phw_pmc;
+
+	return (0);
+}
+
+static int
+perf_get_msr(int ri __diagused, uint32_t *msr __unused)
+{
+	KASSERT(ri >= 0 && ri < PERF_NPMCS,
+	    ("[perf,%d] ri %d out of range", __LINE__, ri));
+
+	return (EINVAL);
+}
+
+static int
+perf_pcpu_fini(struct pmc_mdep *md, int cpu)
+{
+	int i, ri;
+	struct pmc_cpu *pc;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal cpu %d", __LINE__, cpu));
+	KASSERT(perf_pcpu[cpu] != NULL, ("[perf,%d] null pcpu", __LINE__));
+
+	free(perf_pcpu[cpu], M_PMC);
+	perf_pcpu[cpu] = NULL;
+
+	ri = md->pmd_classdep[perf_classindex].pcd_ri;
+	pc = pmc_pcpu[cpu];
+
+	for (i = 0; i < PERF_NPMCS; i++) {
+		pc->pc_hwpmcs[i + ri] = NULL;
+	}
+
+	return (0);
+}
+
+static int
+perf_pcpu_init(struct pmc_mdep *md, int cpu)
+{
+	int i, ri;
+	struct pmc_cpu *pc;
+	struct perf_cpu *perf_pc;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal cpu %d", __LINE__, cpu));
+	KASSERT(perf_pcpu, ("[perf,%d] null pcpu", __LINE__));
+	KASSERT(perf_pcpu[cpu] == NULL, ("[perf,%d] non-null per-cpu",
+	    __LINE__));
+
+	perf_pc = malloc(sizeof(struct perf_cpu), M_PMC, M_WAITOK | M_ZERO);
+
+	perf_pcpu[cpu] = perf_pc;
+
+	ri = md->pmd_classdep[perf_classindex].pcd_ri;
+
+	KASSERT(pmc_pcpu, ("[perf,%d] null generic pcpu", __LINE__));
+
+	pc = pmc_pcpu[cpu];
+
+	KASSERT(pc, ("[perf,%d] null generic per-cpu", __LINE__));
+
+	for (i = 0; i < PERF_NPMCS; i++) {
+		perf_pc->tc_hw[i].phw_state = PMC_PHW_FLAG_IS_ENABLED |
+			PMC_PHW_CPU_TO_STATE(cpu) | PMC_PHW_INDEX_TO_STATE(i) |
+			PMC_PHW_FLAG_IS_SHAREABLE;
+		pc->pc_hwpmcs[i + ri] = &perf_pc->tc_hw[i];
+	}
+
+	return (0);
+}
+
+static int
+perf_read_pmc(int cpu __diagused, int ri, struct pmc *pm, pmc_value_t *v)
+{
+	enum pmc_mode mode __diagused;
+
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal ri %d",
+	    __LINE__, ri));
+
+	mode = PMC_TO_MODE(pm);
+
+	KASSERT(mode == PMC_MODE_SC,
+		("[perf,%d] illegal pmc mode %d", __LINE__, mode));
+
+	PMCDBG1(MDP, REA, 1, "perf-read id=%d", ri);
+
+	switch (ri) {
+	case PERF_MPERF:
+		*v = rdmsr(MSR_MPERF);
+		break;
+	case PERF_APERF:
+		*v = rdmsr(MSR_APERF);
+		break;
+	default:
+		return (EINVAL);
+	}
+
+	return (0);
+}
+
+static int
+perf_write_pmc(int cpu __diagused, int ri __diagused, struct pmc *pm __unused,
+	    pmc_value_t v __unused)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	return (0);
+}
+
+static int
+perf_release_pmc(int cpu __diagused, int ri __diagused, struct pmc *pmc __unused)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS,
+	    ("[perf,%d] illegal row-index %d", __LINE__, ri));
+	KASSERT(perf_pcpu[cpu]->tc_hw[ri].phw_pmc == NULL,
+	   ("[perf,%d] PHW pmc non-NULL", __LINE__));
+
+	return (0);
+}
+
+static int
+perf_start_pmc(int cpu __diagused, int ri __diagused, struct pmc *pm __unused)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	return (0);
+}
+
+static int
+perf_stop_pmc(int cpu __diagused, int ri __diagused, struct pmc *pm __unused)
+{
+	KASSERT(cpu >= 0 && cpu < pmc_cpu_max(),
+	    ("[perf,%d] illegal CPU value %d", __LINE__, cpu));
+	KASSERT(ri >= 0 && ri < PERF_NPMCS, ("[perf,%d] illegal row-index %d",
+	    __LINE__, ri));
+
+	return (0);
+}
+
+int
+pmc_perf_initialize(struct pmc_mdep *md, int maxcpu, int classindex)
+{
+	struct pmc_classdep *pcd;
+
+	KASSERT(md != NULL, ("[perf,%d] md is NULL", __LINE__));
+	KASSERT(md->pmd_nclass >= 1, ("[perf,%d] dubious md->nclass %d",
+	    __LINE__, md->pmd_nclass));
+
+	if ((cpu_power_ecx & CPUID_PERF_STAT) && (tsc_perf_stat == 1)) {
+
+		perf_pcpu = malloc(sizeof(struct perf_cpu *) * maxcpu, M_PMC,
+			M_ZERO | M_WAITOK);
+
+		perf_classindex = classindex;
+		pcd = &md->pmd_classdep[classindex];
+
+		pcd->pcd_caps   = PMC_CAP_READ;
+		pcd->pcd_class  = PMC_CLASS_PERF;
+		pcd->pcd_num    = PERF_NPMCS;
+		pcd->pcd_ri	= md->pmd_npmc;
+		pcd->pcd_width  = 64;
+
+		pcd->pcd_allocate_pmc	= perf_allocate_pmc;
+		pcd->pcd_config_pmc	= perf_config_pmc;
+		pcd->pcd_describe	= perf_describe;
+		pcd->pcd_get_config	= perf_get_config;
+		pcd->pcd_get_msr	= perf_get_msr;
+		pcd->pcd_pcpu_init	= perf_pcpu_init;
+		pcd->pcd_pcpu_fini	= perf_pcpu_fini;
+		pcd->pcd_read_pmc	= perf_read_pmc;
+		pcd->pcd_write_pmc	= perf_write_pmc;
+		pcd->pcd_release_pmc	= perf_release_pmc;
+		pcd->pcd_start_pmc	= perf_start_pmc;
+		pcd->pcd_stop_pmc	= perf_stop_pmc;
+
+		md->pmd_npmc += PERF_NPMCS;
+	}
+
+	return (0);
+}
+
+void pmc_perf_finalize(struct pmc_mdep *md)
+{
+	PMCDBG0(MDP, INI, 1, "perf-finalize");
+
+	if (perf_pcpu != NULL) {
+		for (int i = 0; i < pmc_cpu_max(); i++)
+			KASSERT(perf_pcpu[i] == NULL,
+			    ("[perf,%d] non-null pcpu cpu %d", __LINE__, i));
+
+		free(perf_pcpu, M_PMC);
+		perf_pcpu = NULL;
+	}
+}
diff --git a/sys/dev/hwpmc/hwpmc_perf.h b/sys/dev/hwpmc/hwpmc_perf.h
new file mode 100644
index 000000000000..855c91d290f9
--- /dev/null
+++ b/sys/dev/hwpmc/hwpmc_perf.h
@@ -0,0 +1,28 @@
+/*
+ * Copyright (c) 2026 Advanced Micro Devices, Inc.
+ *
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * AMD/Intel MPERF and APERF MSRs counters exposed as an hwpmc(4) PMC class.
+ */
+
+#ifdef  _KERNEL
+
+/*
+ * A row per MSR: MPERF and APERF.
+ */
+
+#define PERF_NPMCS 2
+#define PERF_MPERF 0
+#define PERF_APERF 1
+
+extern int      tsc_perf_stat;
+extern  u_int   cpu_power_ecx;
+
+/*
+ * Prototypes.
+ */
+
+int     pmc_perf_initialize(struct pmc_mdep *_md, int maxcpu, int classindex);
+void    pmc_perf_finalize(struct pmc_mdep *_md);
+#endif  /* _KERNEL */
diff --git a/sys/dev/hwpmc/pmc_events.h b/sys/dev/hwpmc/pmc_events.h
index f21abd0b77a7..442c1e0a0fe0 100644
--- a/sys/dev/hwpmc/pmc_events.h
+++ b/sys/dev/hwpmc/pmc_events.h
@@ -62,6 +62,14 @@ __PMC_EV_ALIAS("cycles",	TSC_TSC)
 #define	PMC_EV_RAPL_FIRST	PMC_EV_RAPL_ENERGY_PKG
 #define	PMC_EV_RAPL_LAST	PMC_EV_RAPL_ENERGY_DRAM
 
+/* MPERF / APERF MSRs */
+#define __PMC_EV_PERF()				\
+	__PMC_EV(PERF, MPERF)			\
+	__PMC_EV(PERF, APERF)
+
+#define		PMC_EV_PERF_FIRST	PMC_EV_PERF_MPERF
+#define		PMC_EV_PERF_LAST	PMC_EV_PERF_APERF
+
 /*
  * Software events are dynamically defined.
  */
@@ -2438,6 +2446,7 @@ __PMC_EV_ALIAS("unhalted-reference-cycles", IAF_CPU_CLK_UNHALTED_REF)
  * 0x14520	0x0080		ARM DMC-620 clk events
  * 0x14600	0x0100		ARM CMN-600 events
  * 0x14700	0x0100		AMD/Intel RAPL energy events
+ * 0x14800	0x0100		AMD/Intel PERF MSRs events
  * 0x20000	0x1000		Software events
  */
 #define	__PMC_EVENTS()					\
@@ -2466,7 +2475,9 @@ __PMC_EV_ALIAS("unhalted-reference-cycles", IAF_CPU_CLK_UNHALTED_REF)
 	__PMC_EV_BLOCK(CMN600_PMU,	0x14600)	\
 	__PMC_EV_CMN600_PMU()				\
 	__PMC_EV_BLOCK(RAPL,		0x14700)	\
-	__PMC_EV_RAPL()
+	__PMC_EV_RAPL()					\
+	__PMC_EV_BLOCK(PERF,		0x14800)	\
+	__PMC_EV_PERF()
 
 #define	PMC_EVENT_FIRST	PMC_EV_TSC_TSC
 #define	PMC_EVENT_LAST	PMC_EV_SOFT_LAST
diff --git a/sys/modules/hwpmc/Makefile b/sys/modules/hwpmc/Makefile
index 39ce7be9884e..e6953db370a9 100644
--- a/sys/modules/hwpmc/Makefile
+++ b/sys/modules/hwpmc/Makefile
@@ -24,6 +24,7 @@ SRCS+=	hwpmc_amd.c \
 	hwpmc_core.c \
 	hwpmc_ibs.c \
 	hwpmc_intel.c \
+	hwpmc_perf.c \
 	hwpmc_rapl.c \
 	hwpmc_tsc.c \
 	hwpmc_uncore.c \
diff --git a/sys/sys/pmc.h b/sys/sys/pmc.h
index dc2e9cd9109a..284e82561de5 100644
--- a/sys/sys/pmc.h
+++ b/sys/sys/pmc.h
@@ -60,7 +60,7 @@
  * The patch version is incremented for every bug fix.
  */
 #define	PMC_VERSION_MAJOR	0x0A
-#define	PMC_VERSION_MINOR	0x02
+#define	PMC_VERSION_MINOR	0x03
 #define	PMC_VERSION_PATCH	0x0000
 
 #define	PMC_VERSION		(PMC_VERSION_MAJOR << 24 |		\
@@ -157,7 +157,8 @@ enum pmc_cputype {
     __PMC_CLASS(DMC620_PMU_CD2,	0x16,	"ARM DMC620 Memory Controller PMU CLKDIV2") \
     __PMC_CLASS(DMC620_PMU_C,	0x17,	"ARM DMC620 Memory Controller PMU CLK")	\
     __PMC_CLASS(CMN600_PMU,	0x18,	"Arm CoreLink CMN600 Coherent Mesh Network PMU") \
-    __PMC_CLASS(RAPL,		0x19,	"AMD/Intel RAPL energy counters")
+    __PMC_CLASS(RAPL,		0x19,	"AMD/Intel RAPL energy counters")	\
+    __PMC_CLASS(PERF,		0x1A,	"AMD/Intel PERF MSRs")
 
 enum pmc_class {
 #undef  __PMC_CLASS