summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-09-27 08:15:58 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-09-27 08:15:58 -0700
commit5ccda18d1ba2ecf436102a211baf1fbd5d81703f (patch)
treeadc457d112f1f87ef0fe8c97f9ad7ccf54c3de0d
parentfd179f8a05be3ccae366b9b96e176b51fbe54aab (diff)
parent24b620729e53d978b3e425f55bc66efd3bab1f59 (diff)
downloadlinux-5ccda18d1ba2ecf436102a211baf1fbd5d81703f.tar.gz
linux-5ccda18d1ba2ecf436102a211baf1fbd5d81703f.zip
Merge tag 'perf-urgent-2026-09-27' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip
Pull perf events fixes from Ingo Molnar: - Fixes for KVM guest PEBS virtualization (Sean Christopherson) - Fixes for various Intel PMUs related to PEBS data-source (Dapeng Mi) - Fix Intel Panther Cove event scheduling constraints (Dapeng Mi) - Fix Intel DMR/NVL OMR extra registers event scheduling (Dapeng Mi) - Rename two confusingly named PMU attributes (Dapeng Mi) - Fix a refcount leak in attach_perf_ctx_data() (Namhyung Kim) - Fix NULL pointer dereference crash in __perf_pmu_sched_task() (Puranjay Mohan) - Fix CPU-wide event scheduling (Puranjay Mohan) - Fix x86 LBR branch entry generation (Puranjay Mohan) * tag 'perf-urgent-2026-09-27' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip: perf/core: Fill branch entries with a single assignment perf/core: Run sched_task() for PMUs with only CPU-wide events perf/core: Fix NULL pmu_ctx passed to pmu->sched_task() perf/core: Fix a refcount leak in attach_perf_ctx_data() perf/x86/intel: Rename NVL offcore_rsp attribute to offmodule_rsp perf/x86/intel: Rename DMR offcore_rsp attribute to offmodule_rsp perf/x86/intel: Fix precise OMR event scheduling for DMR/NVL perf/x86/intel: Constrain Panther Cove UOPS_DISPATCHED events to PMCs 0-3 perf/x86/intel: Delete dead NVL PEBS data-source initcall perf/x86/intel: Fix Panther Cove PEBS data-source snoop states perf/x86/intel: Remove incorrect Panther Cove PEBS data-source constraints perf/x86/intel: Remove incorrect LionCove PEBS data-source constraints perf/x86/intel: Update arw_latency_data() mem-op direction handling perf/x86/intel: Fix DKT PEBS load/store direction for latency events, to fix sample classification perf/x86/intel: Fix CMT PEBS load/store direction for latency events, to fix sample classification perf/x86/intel: Fix GRT PEBS load/store direction for latency events, to fix sample classification perf/x86/intel: Make @data a mandatory param for intel_guest_get_msrs() perf/x86/intel: Don't pointlessly context switch DS_AREA (and PEBS config) if PEBS is unused perf/x86/intel: Don't write PEBS_ENABLED on host<=>guest xfers if CPU has PEBS isolation, to fix stuck PEBS_ENABLED perf/x86/intel: Ensure KVM guest PEBS path doesn't set unwanted PERF_GLOBAL_CTRL bits
-rw-r--r--arch/x86/events/amd/brs.c9
-rw-r--r--arch/x86/events/amd/lbr.c16
-rw-r--r--arch/x86/events/intel/core.c131
-rw-r--r--arch/x86/events/intel/ds.c74
-rw-r--r--arch/x86/events/intel/lbr.c65
-rw-r--r--arch/x86/events/perf_event.h4
-rw-r--r--drivers/perf/arm_brbe.c2
-rw-r--r--include/linux/perf_event.h17
-rw-r--r--kernel/events/core.c17
9 files changed, 196 insertions, 139 deletions
diff --git a/arch/x86/events/amd/brs.c b/arch/x86/events/amd/brs.c
index dc564688f3d7..54b13faba116 100644
--- a/arch/x86/events/amd/brs.c
+++ b/arch/x86/events/amd/brs.c
@@ -343,11 +343,10 @@ void amd_brs_drain(void)
if (!amd_brs_match_plm(event, from, to))
continue;
- perf_clear_branch_entry_bitfields(br+nr);
-
- br[nr].from = from;
- br[nr].to = to;
-
+ br[nr] = (struct perf_branch_entry){
+ .from = from,
+ .to = to,
+ };
nr++;
}
empty:
diff --git a/arch/x86/events/amd/lbr.c b/arch/x86/events/amd/lbr.c
index 9d9c961989d5..a55646fcb846 100644
--- a/arch/x86/events/amd/lbr.c
+++ b/arch/x86/events/amd/lbr.c
@@ -184,13 +184,6 @@ void amd_pmu_lbr_read(void)
entry.to.split.reserved)
continue;
- perf_clear_branch_entry_bitfields(br + out);
-
- br[out].from = sign_ext_branch_ip(entry.from.split.ip);
- br[out].to = sign_ext_branch_ip(entry.to.split.ip);
- br[out].mispred = entry.from.split.mispredict;
- br[out].predicted = !br[out].mispred;
-
/*
* Set branch speculation information using the status of
* the valid and spec bits.
@@ -208,7 +201,14 @@ void amd_pmu_lbr_read(void)
* speculative and took the correct path
*/
idx = (entry.to.split.valid << 1) | entry.to.split.spec;
- br[out].spec = lbr_spec_map[idx];
+
+ br[out] = (struct perf_branch_entry){
+ .from = sign_ext_branch_ip(entry.from.split.ip),
+ .to = sign_ext_branch_ip(entry.to.split.ip),
+ .mispred = entry.from.split.mispredict,
+ .predicted = !entry.from.split.mispredict,
+ .spec = lbr_spec_map[idx],
+ };
out++;
}
diff --git a/arch/x86/events/intel/core.c b/arch/x86/events/intel/core.c
index 1ac2ca35db53..3ef80882843e 100644
--- a/arch/x86/events/intel/core.c
+++ b/arch/x86/events/intel/core.c
@@ -506,6 +506,8 @@ static struct event_constraint intel_pnc_event_constraints[] = {
INTEL_EVENT_CONSTRAINT(0xce, 0x1),
INTEL_UEVENT_CONSTRAINT(0x01b1, 0x8),
+ INTEL_UEVENT_CONSTRAINT(0x01b2, 0xf),
+ INTEL_UEVENT_CONSTRAINT(0x02b2, 0xf),
INTEL_UEVENT_CONSTRAINT(0x0847, 0xf),
INTEL_UEVENT_CONSTRAINT(0x0446, 0xf),
INTEL_UEVENT_CONSTRAINT(0x0846, 0xf),
@@ -520,6 +522,14 @@ static struct extra_reg intel_pnc_extra_regs[] __read_mostly = {
INTEL_UEVENT_EXTRA_REG(0x022a, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1),
INTEL_UEVENT_EXTRA_REG(0x042a, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2),
INTEL_UEVENT_EXTRA_REG(0x082a, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3),
+ INTEL_UEVENT_EXTRA_REG(0x014f, MSR_OMR_0, 0x40ffffff0000ffffull, OMR_0),
+ INTEL_UEVENT_EXTRA_REG(0x024f, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1),
+ INTEL_UEVENT_EXTRA_REG(0x044f, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2),
+ INTEL_UEVENT_EXTRA_REG(0x084f, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3),
+ INTEL_UEVENT_EXTRA_REG(0x01d6, MSR_OMR_0, 0x40ffffff0000ffffull, OMR_0),
+ INTEL_UEVENT_EXTRA_REG(0x02d6, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1),
+ INTEL_UEVENT_EXTRA_REG(0x04d6, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2),
+ INTEL_UEVENT_EXTRA_REG(0x08d6, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3),
INTEL_UEVENT_PEBS_LDLAT_EXTRA_REG(0x01cd),
INTEL_UEVENT_EXTRA_REG(0x02c6, MSR_PEBS_FRONTEND, 0x9, FE),
INTEL_UEVENT_EXTRA_REG(0x03c6, MSR_PEBS_FRONTEND, 0x7fff1f, FE),
@@ -5292,12 +5302,15 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
struct kvm_pmu *kvm_pmu = (struct kvm_pmu *)data;
u64 intel_ctrl = hybrid(cpuc->pmu, intel_ctrl);
u64 pebs_mask = cpuc->pebs_enabled & x86_pmu.pebs_capable;
- int global_ctrl, pebs_enable;
+ u64 guest_pebs_mask;
+ int global_ctrl;
/*
* In addition to obeying exclude_guest/exclude_host, remove bits being
* used for PEBS when running a guest, because PEBS writes to virtual
- * addresses (not physical addresses).
+ * addresses (not physical addresses). If the guest wants to utilize
+ * PEBS, and PEBS can be safely enabled in the guest, bits for the guest's
+ * PEBS-enabled counters will be OR'd back in as appropriate.
*/
*nr = 0;
global_ctrl = (*nr)++;
@@ -5327,41 +5340,68 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
return arr;
}
- if (!kvm_pmu || !x86_pmu.pebs_ept)
+ /*
+ * If the CPU doesn't support PEBS in the guest, then there's nothing
+ * more to do as disabling PMCs via PERF_GLOBAL_CTRL is sufficient on
+ * CPUs with guest/host isolation.
+ */
+ if (!x86_pmu.pebs_ept)
return arr;
+ /*
+ * Restrict guest PEBS events to counters that (a) perf supports, (b)
+ * the guest wants to use for PEBS, (c) are not excluded from counting
+ * in the guest, and (d) _are_ excluded from counting in the host.
+ */
+ guest_pebs_mask = pebs_mask & intel_ctrl & kvm_pmu->pebs_enable &
+ ~cpuc->intel_ctrl_host_mask &
+ cpuc->intel_ctrl_guest_mask;
+
+ /*
+ * Disable counters where the guest PMC is different than the host PMC
+ * being used on behalf of the guest, as the PEBS record includes
+ * PERF_GLOBAL_STATUS, i.e. the guest will see overflow status for the
+ * wrong counter(s).
+ */
+ guest_pebs_mask &= ~kvm_pmu->host_cross_mapped_mask;
+
+ /*
+ * FIXME: Allow guest and host usage of PEBS events to co-exist instead
+ * of disabling guest PEBS entirely if the host is using PEBS.
+ * What exactly goes wrong if guest and host are using PEBS is
+ * unknown.
+ */
+ if (pebs_mask & ~cpuc->intel_ctrl_guest_mask)
+ guest_pebs_mask = 0;
+
+ /*
+ * Context switch DS_AREA and PEBS_DATA_CFG if and only if PEBS will be
+ * active in the guest; if no records will be generated while the guest
+ * is running, then simply keep the host values resident in hardware.
+ */
arr[(*nr)++] = (struct perf_guest_switch_msr){
.msr = MSR_IA32_DS_AREA,
.host = (unsigned long)cpuc->ds,
- .guest = kvm_pmu->ds_area,
+ .guest = guest_pebs_mask ? kvm_pmu->ds_area : (unsigned long)cpuc->ds,
};
if (x86_pmu.intel_cap.pebs_baseline) {
arr[(*nr)++] = (struct perf_guest_switch_msr){
.msr = MSR_PEBS_DATA_CFG,
.host = cpuc->active_pebs_data_cfg,
- .guest = kvm_pmu->pebs_data_cfg,
+ .guest = guest_pebs_mask ? kvm_pmu->pebs_data_cfg :
+ cpuc->active_pebs_data_cfg,
};
}
- pebs_enable = (*nr)++;
- arr[pebs_enable] = (struct perf_guest_switch_msr){
- .msr = MSR_IA32_PEBS_ENABLE,
- .host = cpuc->pebs_enabled & ~cpuc->intel_ctrl_guest_mask,
- .guest = pebs_mask & ~cpuc->intel_ctrl_host_mask & kvm_pmu->pebs_enable,
- };
-
- if (arr[pebs_enable].host) {
- /* Disable guest PEBS if host PEBS is enabled. */
- arr[pebs_enable].guest = 0;
- } else {
- /* Disable guest PEBS thoroughly for cross-mapped PEBS counters. */
- arr[pebs_enable].guest &= ~kvm_pmu->host_cross_mapped_mask;
- arr[global_ctrl].guest &= ~kvm_pmu->host_cross_mapped_mask;
- /* Set hw GLOBAL_CTRL bits for PEBS counter when it runs for guest */
- arr[global_ctrl].guest |= arr[pebs_enable].guest;
- }
-
+ /*
+ * Do NOT mess with PEBS_ENABLED. As above, disabling counters via
+ * PERF_GLOBAL_CTRL is sufficient, and loading a stale PEBS_ENABLED,
+ * e.g. on VM-Exit, can put the system in a bad state. Simply enable
+ * counters in PERF_GLOBAL_CTRL, as perf load PEBS_ENABLED with the
+ * full value, i.e. perf *also* relies on PERF_GLOBAL_CTRL.
+ */
+ arr[global_ctrl].guest |= guest_pebs_mask;
return arr;
}
@@ -6561,6 +6601,8 @@ static void intel_pmu_filter(struct pmu *pmu, int cpu, bool *ret)
PMU_FORMAT_ATTR(offcore_rsp, "config1:0-63");
+PMU_FORMAT_ATTR(offmodule_rsp, "config1:0-63");
+
PMU_FORMAT_ATTR(ldlat, "config1:0-15");
PMU_FORMAT_ATTR(frontend, "config1:0-23");
@@ -6609,6 +6651,20 @@ static struct attribute *skl_format_attr[] = {
NULL,
};
+static struct attribute *pnc_format_attr_rtm[] = {
+ &format_attr_in_tx.attr,
+ &format_attr_in_tx_cp.attr,
+ &format_attr_offmodule_rsp.attr,
+ &format_attr_ldlat.attr,
+ NULL
+};
+
+static struct attribute *pnc_format_attr[] = {
+ &format_attr_offmodule_rsp.attr,
+ &format_attr_ldlat.attr,
+ NULL
+};
+
static __initconst const struct x86_pmu core_pmu = {
.name = "core",
.handle_irq = x86_pmu_handle_irq,
@@ -7474,6 +7530,7 @@ static struct attribute *adl_hybrid_tsx_attrs[] = {
FORMAT_ATTR_HYBRID(in_tx, hybrid_big);
FORMAT_ATTR_HYBRID(in_tx_cp, hybrid_big);
FORMAT_ATTR_HYBRID(offcore_rsp, hybrid_big_small_tiny);
+FORMAT_ATTR_HYBRID(offmodule_rsp, hybrid_big_small_tiny);
FORMAT_ATTR_HYBRID(ldlat, hybrid_big_small_tiny);
FORMAT_ATTR_HYBRID(frontend, hybrid_big);
@@ -7512,6 +7569,23 @@ static struct attribute *mtl_hybrid_extra_attr[] = {
NULL
};
+static struct attribute *nvl_hybrid_extra_attr_rtm[] = {
+ ADL_HYBRID_RTM_FORMAT_ATTR,
+ FORMAT_HYBRID_PTR(offmodule_rsp),
+ FORMAT_HYBRID_PTR(ldlat),
+ FORMAT_HYBRID_PTR(frontend),
+ FORMAT_HYBRID_PTR(snoop_rsp),
+ NULL
+};
+
+static struct attribute *nvl_hybrid_extra_attr[] = {
+ FORMAT_HYBRID_PTR(offmodule_rsp),
+ FORMAT_HYBRID_PTR(ldlat),
+ FORMAT_HYBRID_PTR(frontend),
+ FORMAT_HYBRID_PTR(snoop_rsp),
+ NULL
+};
+
static bool is_attr_for_this_pmu(struct kobject *kobj, struct attribute *attr)
{
struct device *dev = kobj_to_dev(kobj);
@@ -8549,6 +8623,8 @@ __init int intel_pmu_init(void)
case INTEL_DIAMONDRAPIDS_X:
intel_pmu_init_pnc(NULL);
x86_pmu.pebs_latency_data = pnc_latency_data;
+ extra_attr = boot_cpu_has(X86_FEATURE_RTM) ?
+ pnc_format_attr_rtm : pnc_format_attr;
pr_cont("Panthercove events, ");
name = "panthercove";
@@ -8557,13 +8633,12 @@ __init int intel_pmu_init(void)
glc_common:
intel_pmu_init_glc(NULL);
intel_pmu_pebs_data_source_skl(true);
-
+ extra_attr = boot_cpu_has(X86_FEATURE_RTM) ?
+ hsw_format_attr : nhm_format_attr;
glc_base:
x86_pmu.pebs_ept = 1;
x86_pmu.hw_config = hsw_hw_config;
x86_pmu.get_event_constraints = glc_get_event_constraints;
- extra_attr = boot_cpu_has(X86_FEATURE_RTM) ?
- hsw_format_attr : nhm_format_attr;
extra_skl_attr = skl_format_attr;
mem_attr = glc_events_attrs;
td_attr = glc_td_events_attrs;
@@ -8785,7 +8860,7 @@ __init int intel_pmu_init(void)
mem_attr = mtl_hybrid_mem_attrs;
tsx_attr = adl_hybrid_tsx_attrs;
extra_attr = boot_cpu_has(X86_FEATURE_RTM) ?
- mtl_hybrid_extra_attr_rtm : mtl_hybrid_extra_attr;
+ nvl_hybrid_extra_attr_rtm : nvl_hybrid_extra_attr;
/* Initialize big core specific PerfMon capabilities.*/
pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX];
@@ -8794,8 +8869,6 @@ __init int intel_pmu_init(void)
/* Initialize Atom core specific PerfMon capabilities.*/
pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX];
intel_pmu_init_arw(&pmu->pmu);
-
- intel_pmu_pebs_data_source_lnl();
break;
default:
diff --git a/arch/x86/events/intel/ds.c b/arch/x86/events/intel/ds.c
index b98029b44052..d0329a7eb8a5 100644
--- a/arch/x86/events/intel/ds.c
+++ b/arch/x86/events/intel/ds.c
@@ -277,8 +277,8 @@ static u64 pnc_pebs_l2_hit_data_source[PNC_PEBS_DATA_SOURCE_MAX] = {
0, /* 0x06: Reserved */
OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, HIT), /* 0x07: L2 Hit Snoop HIT */
OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, HITM), /* 0x08: L2 Hit Snoop Hit Modified */
- OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, MISS), /* 0x09: Prefetch Promotion */
- OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, MISS), /* 0x0a: Cross Core Prefetch Promotion */
+ OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, NONE), /* 0x09: Prefetch Promotion */
+ OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, NONE), /* 0x0a: Cross Core Prefetch Promotion */
0, /* 0x0b: Reserved */
0, /* 0x0c: Reserved */
0, /* 0x0d: Reserved */
@@ -455,6 +455,7 @@ static inline void pebs_set_tlb_lock(u64 *val, bool tlb, bool lock)
static u64 __grt_latency_data(struct perf_event *event, u64 status,
u8 dse, bool tlb, bool lock, bool blk)
{
+ union perf_mem_data_src src;
u64 val;
WARN_ON_ONCE(is_hybrid() &&
@@ -470,7 +471,16 @@ static u64 __grt_latency_data(struct perf_event *event, u64 status,
else
val |= P(BLK, NA);
- return val;
+ src.val = val;
+
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
+ src.mem_op = P(OP, LOAD);
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
+ src.mem_op = P(OP, STORE);
+
+ return src.val;
}
u64 grt_latency_data(struct perf_event *event, u64 status)
@@ -528,7 +538,11 @@ static u64 arw_latency_data(struct perf_event *event, u64 status)
val |= P(BLK, NA);
src.val = val;
- if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW)
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
+ src.mem_op = P(OP, LOAD);
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
src.mem_op = P(OP, STORE);
return src.val;
@@ -563,7 +577,11 @@ static u64 lnc_latency_data(struct perf_event *event, u64 status)
val |= P(BLK, NA);
src.val = val;
- if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW)
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
+ src.mem_op = P(OP, LOAD);
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
src.mem_op = P(OP, STORE);
return src.val;
@@ -621,7 +639,11 @@ u64 pnc_latency_data(struct perf_event *event, u64 status)
val |= P(BLK, NA);
src.val = val;
- if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW)
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW))
+ src.mem_op = P(OP, LOAD);
+ if (event->hw.flags &
+ (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW))
src.mem_op = P(OP, STORE);
return src.val;
@@ -1291,22 +1313,22 @@ struct event_constraint intel_glm_pebs_event_constraints[] = {
struct event_constraint intel_grt_pebs_event_constraints[] = {
/* Allow all events as PEBS with no flags */
- INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0x3),
- INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0x3f),
+ INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3),
+ INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0x3f),
EVENT_CONSTRAINT_END
};
struct event_constraint intel_cmt_pebs_event_constraints[] = {
/* Allow all events as PEBS with no flags */
- INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0x3),
- INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0xff),
+ INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3),
+ INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff),
EVENT_CONSTRAINT_END
};
struct event_constraint intel_dkt_pebs_event_constraints[] = {
/* Allow all events as PEBS with no flags */
- INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0xff),
- INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0xff),
+ INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0xff),
+ INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff),
EVENT_CONSTRAINT_END
};
@@ -1496,24 +1518,8 @@ struct event_constraint intel_lnc_pebs_event_constraints[] = {
INTEL_FLAGS_UEVENT_CONSTRAINT(0x012a, 0x1), /* OCR.* events */
INTEL_FLAGS_UEVENT_CONSTRAINT(0x012b, 0x1), /* OCR.* events */
- INTEL_FLAGS_UEVENT_CONSTRAINT(0x04a4, 0x1), /* TOPDOWN.BAD_SPEC_SLOTS */
- INTEL_FLAGS_UEVENT_CONSTRAINT(0x08a4, 0x1), /* TOPDOWN.BR_MISPREDICT_SLOTS */
- INTEL_FLAGS_UEVENT_CONSTRAINT(0x10a4, 0x8), /* TOPDOWN.MEMORY_BOUND_SLOTS */
-
INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0x3fc),
INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3),
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_STORES */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_INST_RETIRED.LOCK_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_INST_RETIRED.SPLIT_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_INST_RETIRED.SPLIT_STORES */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_INST_RETIRED.ALL_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_INST_RETIRED.ALL_STORES */
- INTEL_FLAGS_UEVENT_CONSTRAINT(0x87d0, 0x3ff), /* MEM_INST_RETIRED.ANY */
-
- INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf),
-
- INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf),
/*
* Everything else is handled by PMU_FL_PEBS_ALL, because we
@@ -1526,18 +1532,6 @@ struct event_constraint intel_lnc_pebs_event_constraints[] = {
struct event_constraint intel_pnc_pebs_event_constraints[] = {
INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0xfc),
INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3),
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_STORES */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_INST_RETIRED.LOCK_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_INST_RETIRED.SPLIT_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_INST_RETIRED.SPLIT_STORES */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_INST_RETIRED.ALL_LOADS */
- INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_INST_RETIRED.ALL_STORES */
-
- INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf),
-
- INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf),
- INTEL_FLAGS_EVENT_CONSTRAINT(0xd6, 0xf),
/*
* Everything else is handled by PMU_FL_PEBS_ALL, because we
diff --git a/arch/x86/events/intel/lbr.c b/arch/x86/events/intel/lbr.c
index cbe5c762008d..22e2a06d5786 100644
--- a/arch/x86/events/intel/lbr.c
+++ b/arch/x86/events/intel/lbr.c
@@ -756,10 +756,10 @@ void intel_pmu_lbr_read_32(struct cpu_hw_events *cpuc)
rdmsrq(x86_pmu.lbr_from + lbr_idx, msr_lastbranch.lbr);
- perf_clear_branch_entry_bitfields(br);
-
- br->from = msr_lastbranch.from;
- br->to = msr_lastbranch.to;
+ *br = (struct perf_branch_entry){
+ .from = msr_lastbranch.from,
+ .to = msr_lastbranch.to,
+ };
br++;
}
cpuc->lbr_stack.nr = i;
@@ -847,14 +847,15 @@ void intel_pmu_lbr_read_64(struct cpu_hw_events *cpuc)
if (abort && x86_pmu.lbr_double_abort && out > 0)
out--;
- perf_clear_branch_entry_bitfields(br+out);
- br[out].from = from;
- br[out].to = to;
- br[out].mispred = mis;
- br[out].predicted = pred;
- br[out].in_tx = in_tx;
- br[out].abort = abort;
- br[out].cycles = cycles;
+ br[out] = (struct perf_branch_entry){
+ .from = from,
+ .to = to,
+ .mispred = mis,
+ .predicted = pred,
+ .in_tx = in_tx,
+ .abort = abort,
+ .cycles = cycles,
+ };
out++;
}
cpuc->lbr_stack.nr = out;
@@ -905,6 +906,7 @@ static void intel_pmu_store_lbr(struct cpu_hw_events *cpuc,
struct perf_branch_entry *e;
struct lbr_entry *lbr;
u64 from, to, info;
+ bool mispred;
int i;
for (i = 0; i < x86_pmu.lbr_nr; i++) {
@@ -921,24 +923,27 @@ static void intel_pmu_store_lbr(struct cpu_hw_events *cpuc,
to = rdlbr_to(i, lbr);
info = rdlbr_info(i, lbr);
- perf_clear_branch_entry_bitfields(e);
-
- e->from = from;
- e->to = to;
- e->mispred = get_lbr_mispred(info);
- e->predicted = !e->mispred;
- e->in_tx = !!(info & LBR_INFO_IN_TX);
- e->abort = !!(info & LBR_INFO_ABORT);
- e->cycles = get_lbr_cycles(info);
- e->type = get_lbr_br_type(info);
-
- /*
- * Leverage the reserved field of cpuc->lbr_entries[i] to
- * temporarily store the branch counters information.
- * The later code will decide what content can be disclosed
- * to the perf tool. Pleae see intel_pmu_lbr_counters_reorder().
- */
- e->reserved = (info >> LBR_INFO_BR_CNTR_OFFSET) & LBR_INFO_BR_CNTR_FULL_MASK;
+ mispred = get_lbr_mispred(info);
+
+ *e = (struct perf_branch_entry){
+ .from = from,
+ .to = to,
+ .mispred = mispred,
+ .predicted = !mispred,
+ .in_tx = !!(info & LBR_INFO_IN_TX),
+ .abort = !!(info & LBR_INFO_ABORT),
+ .cycles = get_lbr_cycles(info),
+ .type = get_lbr_br_type(info),
+ /*
+ * Leverage the reserved field of
+ * cpuc->lbr_entries[i] to temporarily store the
+ * branch counters information. The later code will
+ * decide what content can be disclosed to the perf
+ * tool. Pleae see intel_pmu_lbr_counters_reorder().
+ */
+ .reserved = (info >> LBR_INFO_BR_CNTR_OFFSET) &
+ LBR_INFO_BR_CNTR_FULL_MASK,
+ };
}
cpuc->lbr_stack.nr = i;
diff --git a/arch/x86/events/perf_event.h b/arch/x86/events/perf_event.h
index 4680cba91340..fab9da78a5c7 100644
--- a/arch/x86/events/perf_event.h
+++ b/arch/x86/events/perf_event.h
@@ -517,10 +517,6 @@ struct cpu_hw_events {
__EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \
HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_ST)
-#define INTEL_HYBRID_LAT_CONSTRAINT(c, n) \
- __EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \
- HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_LAT_HYBRID)
-
#define INTEL_HYBRID_LDLAT_CONSTRAINT(c, n) \
__EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \
HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_LAT_HYBRID|PERF_X86_EVENT_PEBS_LD_HSW)
diff --git a/drivers/perf/arm_brbe.c b/drivers/perf/arm_brbe.c
index ba554e0c846c..254be4da8ae2 100644
--- a/drivers/perf/arm_brbe.c
+++ b/drivers/perf/arm_brbe.c
@@ -604,7 +604,7 @@ static bool perf_entry_from_brbe_regset(int index, struct perf_branch_entry *ent
return false;
brbinf = bregs.brbinf;
- perf_clear_branch_entry_bitfields(entry);
+ *entry = (struct perf_branch_entry){ };
if (brbe_record_is_complete(brbinf)) {
entry->from = bregs.brbsrc;
entry->to = bregs.brbtgt;
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
index 5842552294c1..915c6fd3f084 100644
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -1467,23 +1467,6 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data,
return size;
}
-/*
- * Clear all bitfields in the perf_branch_entry.
- * The to and from fields are not cleared because they are
- * systematically modified by caller.
- */
-static inline void perf_clear_branch_entry_bitfields(struct perf_branch_entry *br)
-{
- br->mispred = 0;
- br->predicted = 0;
- br->in_tx = 0;
- br->abort = 0;
- br->cycles = 0;
- br->type = 0;
- br->spec = PERF_BR_SPEC_NA;
- br->reserved = 0;
-}
-
extern void perf_output_sample(struct perf_output_handle *handle,
struct perf_event_header *header,
struct perf_sample_data *data,
diff --git a/kernel/events/core.c b/kernel/events/core.c
index db7b76d6b68a..634d2ccbab82 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx,
list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) {
cpc = this_cpc(pmu_ctx->pmu);
+ if (cpc->task_epc != pmu_ctx)
+ continue;
+
if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task)
pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in);
}
@@ -3914,7 +3917,7 @@ static void __perf_pmu_sched_task(struct perf_cpu_pmu_context *cpc,
perf_ctx_lock(cpuctx, cpuctx->task_ctx);
perf_pmu_disable(pmu);
- pmu->sched_task(cpc->task_epc, task, sched_in);
+ pmu->sched_task(&cpc->epc, task, sched_in);
perf_pmu_enable(pmu);
perf_ctx_unlock(cpuctx, cpuctx->task_ctx);
@@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev,
struct task_struct *next,
bool sched_in)
{
- struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
struct perf_cpu_pmu_context *cpc, *cpc2;
- /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */
- if (prev == next || cpuctx->task_ctx)
+ if (prev == next)
return;
- list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry)
+ list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) {
+ if (cpc->task_epc)
+ continue;
+
__perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in);
+ }
}
static void perf_event_switch(struct task_struct *task,
@@ -5454,6 +5459,8 @@ attach_task_ctx_data(struct task_struct *task, struct kmem_cache *ctx_cache,
}
if (refcount_inc_not_zero(&old->refcount)) {
+ if (global)
+ old->global = true;
free_perf_ctx_data(cd); /* unused */
return 0;
}