diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-09-27 08:15:58 -0700 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-09-27 08:15:58 -0700 |
| commit | 5ccda18d1ba2ecf436102a211baf1fbd5d81703f (patch) | |
| tree | adc457d112f1f87ef0fe8c97f9ad7ccf54c3de0d | |
| parent | fd179f8a05be3ccae366b9b96e176b51fbe54aab (diff) | |
| parent | 24b620729e53d978b3e425f55bc66efd3bab1f59 (diff) | |
| download | linux-5ccda18d1ba2ecf436102a211baf1fbd5d81703f.tar.gz linux-5ccda18d1ba2ecf436102a211baf1fbd5d81703f.zip | |
Merge tag 'perf-urgent-2026-09-27' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip
Pull perf events fixes from Ingo Molnar:
- Fixes for KVM guest PEBS virtualization (Sean Christopherson)
- Fixes for various Intel PMUs related to PEBS data-source (Dapeng Mi)
- Fix Intel Panther Cove event scheduling constraints (Dapeng Mi)
- Fix Intel DMR/NVL OMR extra registers event scheduling (Dapeng Mi)
- Rename two confusingly named PMU attributes (Dapeng Mi)
- Fix a refcount leak in attach_perf_ctx_data() (Namhyung Kim)
- Fix NULL pointer dereference crash in __perf_pmu_sched_task()
(Puranjay Mohan)
- Fix CPU-wide event scheduling (Puranjay Mohan)
- Fix x86 LBR branch entry generation (Puranjay Mohan)
* tag 'perf-urgent-2026-09-27' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip:
perf/core: Fill branch entries with a single assignment
perf/core: Run sched_task() for PMUs with only CPU-wide events
perf/core: Fix NULL pmu_ctx passed to pmu->sched_task()
perf/core: Fix a refcount leak in attach_perf_ctx_data()
perf/x86/intel: Rename NVL offcore_rsp attribute to offmodule_rsp
perf/x86/intel: Rename DMR offcore_rsp attribute to offmodule_rsp
perf/x86/intel: Fix precise OMR event scheduling for DMR/NVL
perf/x86/intel: Constrain Panther Cove UOPS_DISPATCHED events to PMCs 0-3
perf/x86/intel: Delete dead NVL PEBS data-source initcall
perf/x86/intel: Fix Panther Cove PEBS data-source snoop states
perf/x86/intel: Remove incorrect Panther Cove PEBS data-source constraints
perf/x86/intel: Remove incorrect LionCove PEBS data-source constraints
perf/x86/intel: Update arw_latency_data() mem-op direction handling
perf/x86/intel: Fix DKT PEBS load/store direction for latency events, to fix sample classification
perf/x86/intel: Fix CMT PEBS load/store direction for latency events, to fix sample classification
perf/x86/intel: Fix GRT PEBS load/store direction for latency events, to fix sample classification
perf/x86/intel: Make @data a mandatory param for intel_guest_get_msrs()
perf/x86/intel: Don't pointlessly context switch DS_AREA (and PEBS config) if PEBS is unused
perf/x86/intel: Don't write PEBS_ENABLED on host<=>guest xfers if CPU has PEBS isolation, to fix stuck PEBS_ENABLED
perf/x86/intel: Ensure KVM guest PEBS path doesn't set unwanted PERF_GLOBAL_CTRL bits
| -rw-r--r-- | arch/x86/events/amd/brs.c | 9 | ||||
| -rw-r--r-- | arch/x86/events/amd/lbr.c | 16 | ||||
| -rw-r--r-- | arch/x86/events/intel/core.c | 131 | ||||
| -rw-r--r-- | arch/x86/events/intel/ds.c | 74 | ||||
| -rw-r--r-- | arch/x86/events/intel/lbr.c | 65 | ||||
| -rw-r--r-- | arch/x86/events/perf_event.h | 4 | ||||
| -rw-r--r-- | drivers/perf/arm_brbe.c | 2 | ||||
| -rw-r--r-- | include/linux/perf_event.h | 17 | ||||
| -rw-r--r-- | kernel/events/core.c | 17 |
9 files changed, 196 insertions, 139 deletions
diff --git a/arch/x86/events/amd/brs.c b/arch/x86/events/amd/brs.c index dc564688f3d7..54b13faba116 100644 --- a/arch/x86/events/amd/brs.c +++ b/arch/x86/events/amd/brs.c @@ -343,11 +343,10 @@ void amd_brs_drain(void) if (!amd_brs_match_plm(event, from, to)) continue; - perf_clear_branch_entry_bitfields(br+nr); - - br[nr].from = from; - br[nr].to = to; - + br[nr] = (struct perf_branch_entry){ + .from = from, + .to = to, + }; nr++; } empty: diff --git a/arch/x86/events/amd/lbr.c b/arch/x86/events/amd/lbr.c index 9d9c961989d5..a55646fcb846 100644 --- a/arch/x86/events/amd/lbr.c +++ b/arch/x86/events/amd/lbr.c @@ -184,13 +184,6 @@ void amd_pmu_lbr_read(void) entry.to.split.reserved) continue; - perf_clear_branch_entry_bitfields(br + out); - - br[out].from = sign_ext_branch_ip(entry.from.split.ip); - br[out].to = sign_ext_branch_ip(entry.to.split.ip); - br[out].mispred = entry.from.split.mispredict; - br[out].predicted = !br[out].mispred; - /* * Set branch speculation information using the status of * the valid and spec bits. @@ -208,7 +201,14 @@ void amd_pmu_lbr_read(void) * speculative and took the correct path */ idx = (entry.to.split.valid << 1) | entry.to.split.spec; - br[out].spec = lbr_spec_map[idx]; + + br[out] = (struct perf_branch_entry){ + .from = sign_ext_branch_ip(entry.from.split.ip), + .to = sign_ext_branch_ip(entry.to.split.ip), + .mispred = entry.from.split.mispredict, + .predicted = !entry.from.split.mispredict, + .spec = lbr_spec_map[idx], + }; out++; } diff --git a/arch/x86/events/intel/core.c b/arch/x86/events/intel/core.c index 1ac2ca35db53..3ef80882843e 100644 --- a/arch/x86/events/intel/core.c +++ b/arch/x86/events/intel/core.c @@ -506,6 +506,8 @@ static struct event_constraint intel_pnc_event_constraints[] = { INTEL_EVENT_CONSTRAINT(0xce, 0x1), INTEL_UEVENT_CONSTRAINT(0x01b1, 0x8), + INTEL_UEVENT_CONSTRAINT(0x01b2, 0xf), + INTEL_UEVENT_CONSTRAINT(0x02b2, 0xf), INTEL_UEVENT_CONSTRAINT(0x0847, 0xf), INTEL_UEVENT_CONSTRAINT(0x0446, 0xf), INTEL_UEVENT_CONSTRAINT(0x0846, 0xf), @@ -520,6 +522,14 @@ static struct extra_reg intel_pnc_extra_regs[] __read_mostly = { INTEL_UEVENT_EXTRA_REG(0x022a, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1), INTEL_UEVENT_EXTRA_REG(0x042a, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2), INTEL_UEVENT_EXTRA_REG(0x082a, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3), + INTEL_UEVENT_EXTRA_REG(0x014f, MSR_OMR_0, 0x40ffffff0000ffffull, OMR_0), + INTEL_UEVENT_EXTRA_REG(0x024f, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1), + INTEL_UEVENT_EXTRA_REG(0x044f, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2), + INTEL_UEVENT_EXTRA_REG(0x084f, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3), + INTEL_UEVENT_EXTRA_REG(0x01d6, MSR_OMR_0, 0x40ffffff0000ffffull, OMR_0), + INTEL_UEVENT_EXTRA_REG(0x02d6, MSR_OMR_1, 0x40ffffff0000ffffull, OMR_1), + INTEL_UEVENT_EXTRA_REG(0x04d6, MSR_OMR_2, 0x40ffffff0000ffffull, OMR_2), + INTEL_UEVENT_EXTRA_REG(0x08d6, MSR_OMR_3, 0x40ffffff0000ffffull, OMR_3), INTEL_UEVENT_PEBS_LDLAT_EXTRA_REG(0x01cd), INTEL_UEVENT_EXTRA_REG(0x02c6, MSR_PEBS_FRONTEND, 0x9, FE), INTEL_UEVENT_EXTRA_REG(0x03c6, MSR_PEBS_FRONTEND, 0x7fff1f, FE), @@ -5292,12 +5302,15 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) struct kvm_pmu *kvm_pmu = (struct kvm_pmu *)data; u64 intel_ctrl = hybrid(cpuc->pmu, intel_ctrl); u64 pebs_mask = cpuc->pebs_enabled & x86_pmu.pebs_capable; - int global_ctrl, pebs_enable; + u64 guest_pebs_mask; + int global_ctrl; /* * In addition to obeying exclude_guest/exclude_host, remove bits being * used for PEBS when running a guest, because PEBS writes to virtual - * addresses (not physical addresses). + * addresses (not physical addresses). If the guest wants to utilize + * PEBS, and PEBS can be safely enabled in the guest, bits for the guest's + * PEBS-enabled counters will be OR'd back in as appropriate. */ *nr = 0; global_ctrl = (*nr)++; @@ -5327,41 +5340,68 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) return arr; } - if (!kvm_pmu || !x86_pmu.pebs_ept) + /* + * If the CPU doesn't support PEBS in the guest, then there's nothing + * more to do as disabling PMCs via PERF_GLOBAL_CTRL is sufficient on + * CPUs with guest/host isolation. + */ + if (!x86_pmu.pebs_ept) return arr; + /* + * Restrict guest PEBS events to counters that (a) perf supports, (b) + * the guest wants to use for PEBS, (c) are not excluded from counting + * in the guest, and (d) _are_ excluded from counting in the host. + */ + guest_pebs_mask = pebs_mask & intel_ctrl & kvm_pmu->pebs_enable & + ~cpuc->intel_ctrl_host_mask & + cpuc->intel_ctrl_guest_mask; + + /* + * Disable counters where the guest PMC is different than the host PMC + * being used on behalf of the guest, as the PEBS record includes + * PERF_GLOBAL_STATUS, i.e. the guest will see overflow status for the + * wrong counter(s). + */ + guest_pebs_mask &= ~kvm_pmu->host_cross_mapped_mask; + + /* + * FIXME: Allow guest and host usage of PEBS events to co-exist instead + * of disabling guest PEBS entirely if the host is using PEBS. + * What exactly goes wrong if guest and host are using PEBS is + * unknown. + */ + if (pebs_mask & ~cpuc->intel_ctrl_guest_mask) + guest_pebs_mask = 0; + + /* + * Context switch DS_AREA and PEBS_DATA_CFG if and only if PEBS will be + * active in the guest; if no records will be generated while the guest + * is running, then simply keep the host values resident in hardware. + */ arr[(*nr)++] = (struct perf_guest_switch_msr){ .msr = MSR_IA32_DS_AREA, .host = (unsigned long)cpuc->ds, - .guest = kvm_pmu->ds_area, + .guest = guest_pebs_mask ? kvm_pmu->ds_area : (unsigned long)cpuc->ds, }; if (x86_pmu.intel_cap.pebs_baseline) { arr[(*nr)++] = (struct perf_guest_switch_msr){ .msr = MSR_PEBS_DATA_CFG, .host = cpuc->active_pebs_data_cfg, - .guest = kvm_pmu->pebs_data_cfg, + .guest = guest_pebs_mask ? kvm_pmu->pebs_data_cfg : + cpuc->active_pebs_data_cfg, }; } - pebs_enable = (*nr)++; - arr[pebs_enable] = (struct perf_guest_switch_msr){ - .msr = MSR_IA32_PEBS_ENABLE, - .host = cpuc->pebs_enabled & ~cpuc->intel_ctrl_guest_mask, - .guest = pebs_mask & ~cpuc->intel_ctrl_host_mask & kvm_pmu->pebs_enable, - }; - - if (arr[pebs_enable].host) { - /* Disable guest PEBS if host PEBS is enabled. */ - arr[pebs_enable].guest = 0; - } else { - /* Disable guest PEBS thoroughly for cross-mapped PEBS counters. */ - arr[pebs_enable].guest &= ~kvm_pmu->host_cross_mapped_mask; - arr[global_ctrl].guest &= ~kvm_pmu->host_cross_mapped_mask; - /* Set hw GLOBAL_CTRL bits for PEBS counter when it runs for guest */ - arr[global_ctrl].guest |= arr[pebs_enable].guest; - } - + /* + * Do NOT mess with PEBS_ENABLED. As above, disabling counters via + * PERF_GLOBAL_CTRL is sufficient, and loading a stale PEBS_ENABLED, + * e.g. on VM-Exit, can put the system in a bad state. Simply enable + * counters in PERF_GLOBAL_CTRL, as perf load PEBS_ENABLED with the + * full value, i.e. perf *also* relies on PERF_GLOBAL_CTRL. + */ + arr[global_ctrl].guest |= guest_pebs_mask; return arr; } @@ -6561,6 +6601,8 @@ static void intel_pmu_filter(struct pmu *pmu, int cpu, bool *ret) PMU_FORMAT_ATTR(offcore_rsp, "config1:0-63"); +PMU_FORMAT_ATTR(offmodule_rsp, "config1:0-63"); + PMU_FORMAT_ATTR(ldlat, "config1:0-15"); PMU_FORMAT_ATTR(frontend, "config1:0-23"); @@ -6609,6 +6651,20 @@ static struct attribute *skl_format_attr[] = { NULL, }; +static struct attribute *pnc_format_attr_rtm[] = { + &format_attr_in_tx.attr, + &format_attr_in_tx_cp.attr, + &format_attr_offmodule_rsp.attr, + &format_attr_ldlat.attr, + NULL +}; + +static struct attribute *pnc_format_attr[] = { + &format_attr_offmodule_rsp.attr, + &format_attr_ldlat.attr, + NULL +}; + static __initconst const struct x86_pmu core_pmu = { .name = "core", .handle_irq = x86_pmu_handle_irq, @@ -7474,6 +7530,7 @@ static struct attribute *adl_hybrid_tsx_attrs[] = { FORMAT_ATTR_HYBRID(in_tx, hybrid_big); FORMAT_ATTR_HYBRID(in_tx_cp, hybrid_big); FORMAT_ATTR_HYBRID(offcore_rsp, hybrid_big_small_tiny); +FORMAT_ATTR_HYBRID(offmodule_rsp, hybrid_big_small_tiny); FORMAT_ATTR_HYBRID(ldlat, hybrid_big_small_tiny); FORMAT_ATTR_HYBRID(frontend, hybrid_big); @@ -7512,6 +7569,23 @@ static struct attribute *mtl_hybrid_extra_attr[] = { NULL }; +static struct attribute *nvl_hybrid_extra_attr_rtm[] = { + ADL_HYBRID_RTM_FORMAT_ATTR, + FORMAT_HYBRID_PTR(offmodule_rsp), + FORMAT_HYBRID_PTR(ldlat), + FORMAT_HYBRID_PTR(frontend), + FORMAT_HYBRID_PTR(snoop_rsp), + NULL +}; + +static struct attribute *nvl_hybrid_extra_attr[] = { + FORMAT_HYBRID_PTR(offmodule_rsp), + FORMAT_HYBRID_PTR(ldlat), + FORMAT_HYBRID_PTR(frontend), + FORMAT_HYBRID_PTR(snoop_rsp), + NULL +}; + static bool is_attr_for_this_pmu(struct kobject *kobj, struct attribute *attr) { struct device *dev = kobj_to_dev(kobj); @@ -8549,6 +8623,8 @@ __init int intel_pmu_init(void) case INTEL_DIAMONDRAPIDS_X: intel_pmu_init_pnc(NULL); x86_pmu.pebs_latency_data = pnc_latency_data; + extra_attr = boot_cpu_has(X86_FEATURE_RTM) ? + pnc_format_attr_rtm : pnc_format_attr; pr_cont("Panthercove events, "); name = "panthercove"; @@ -8557,13 +8633,12 @@ __init int intel_pmu_init(void) glc_common: intel_pmu_init_glc(NULL); intel_pmu_pebs_data_source_skl(true); - + extra_attr = boot_cpu_has(X86_FEATURE_RTM) ? + hsw_format_attr : nhm_format_attr; glc_base: x86_pmu.pebs_ept = 1; x86_pmu.hw_config = hsw_hw_config; x86_pmu.get_event_constraints = glc_get_event_constraints; - extra_attr = boot_cpu_has(X86_FEATURE_RTM) ? - hsw_format_attr : nhm_format_attr; extra_skl_attr = skl_format_attr; mem_attr = glc_events_attrs; td_attr = glc_td_events_attrs; @@ -8785,7 +8860,7 @@ __init int intel_pmu_init(void) mem_attr = mtl_hybrid_mem_attrs; tsx_attr = adl_hybrid_tsx_attrs; extra_attr = boot_cpu_has(X86_FEATURE_RTM) ? - mtl_hybrid_extra_attr_rtm : mtl_hybrid_extra_attr; + nvl_hybrid_extra_attr_rtm : nvl_hybrid_extra_attr; /* Initialize big core specific PerfMon capabilities.*/ pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX]; @@ -8794,8 +8869,6 @@ __init int intel_pmu_init(void) /* Initialize Atom core specific PerfMon capabilities.*/ pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX]; intel_pmu_init_arw(&pmu->pmu); - - intel_pmu_pebs_data_source_lnl(); break; default: diff --git a/arch/x86/events/intel/ds.c b/arch/x86/events/intel/ds.c index b98029b44052..d0329a7eb8a5 100644 --- a/arch/x86/events/intel/ds.c +++ b/arch/x86/events/intel/ds.c @@ -277,8 +277,8 @@ static u64 pnc_pebs_l2_hit_data_source[PNC_PEBS_DATA_SOURCE_MAX] = { 0, /* 0x06: Reserved */ OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, HIT), /* 0x07: L2 Hit Snoop HIT */ OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, HITM), /* 0x08: L2 Hit Snoop Hit Modified */ - OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, MISS), /* 0x09: Prefetch Promotion */ - OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, MISS), /* 0x0a: Cross Core Prefetch Promotion */ + OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, NONE), /* 0x09: Prefetch Promotion */ + OP_LH | P(LVL, L2) | LEVEL(L2) | P(SNOOP, NONE), /* 0x0a: Cross Core Prefetch Promotion */ 0, /* 0x0b: Reserved */ 0, /* 0x0c: Reserved */ 0, /* 0x0d: Reserved */ @@ -455,6 +455,7 @@ static inline void pebs_set_tlb_lock(u64 *val, bool tlb, bool lock) static u64 __grt_latency_data(struct perf_event *event, u64 status, u8 dse, bool tlb, bool lock, bool blk) { + union perf_mem_data_src src; u64 val; WARN_ON_ONCE(is_hybrid() && @@ -470,7 +471,16 @@ static u64 __grt_latency_data(struct perf_event *event, u64 status, else val |= P(BLK, NA); - return val; + src.val = val; + + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW)) + src.mem_op = P(OP, LOAD); + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW)) + src.mem_op = P(OP, STORE); + + return src.val; } u64 grt_latency_data(struct perf_event *event, u64 status) @@ -528,7 +538,11 @@ static u64 arw_latency_data(struct perf_event *event, u64 status) val |= P(BLK, NA); src.val = val; - if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW) + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW)) + src.mem_op = P(OP, LOAD); + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW)) src.mem_op = P(OP, STORE); return src.val; @@ -563,7 +577,11 @@ static u64 lnc_latency_data(struct perf_event *event, u64 status) val |= P(BLK, NA); src.val = val; - if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW) + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW)) + src.mem_op = P(OP, LOAD); + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW)) src.mem_op = P(OP, STORE); return src.val; @@ -621,7 +639,11 @@ u64 pnc_latency_data(struct perf_event *event, u64 status) val |= P(BLK, NA); src.val = val; - if (event->hw.flags & PERF_X86_EVENT_PEBS_ST_HSW) + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_LDLAT | PERF_X86_EVENT_PEBS_LD_HSW)) + src.mem_op = P(OP, LOAD); + if (event->hw.flags & + (PERF_X86_EVENT_PEBS_STLAT | PERF_X86_EVENT_PEBS_ST_HSW)) src.mem_op = P(OP, STORE); return src.val; @@ -1291,22 +1313,22 @@ struct event_constraint intel_glm_pebs_event_constraints[] = { struct event_constraint intel_grt_pebs_event_constraints[] = { /* Allow all events as PEBS with no flags */ - INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0x3), - INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0x3f), + INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3), + INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0x3f), EVENT_CONSTRAINT_END }; struct event_constraint intel_cmt_pebs_event_constraints[] = { /* Allow all events as PEBS with no flags */ - INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0x3), - INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0xff), + INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0x3), + INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff), EVENT_CONSTRAINT_END }; struct event_constraint intel_dkt_pebs_event_constraints[] = { /* Allow all events as PEBS with no flags */ - INTEL_HYBRID_LAT_CONSTRAINT(0x5d0, 0xff), - INTEL_HYBRID_LAT_CONSTRAINT(0x6d0, 0xff), + INTEL_HYBRID_LDLAT_CONSTRAINT(0x5d0, 0xff), + INTEL_HYBRID_STLAT_CONSTRAINT(0x6d0, 0xff), EVENT_CONSTRAINT_END }; @@ -1496,24 +1518,8 @@ struct event_constraint intel_lnc_pebs_event_constraints[] = { INTEL_FLAGS_UEVENT_CONSTRAINT(0x012a, 0x1), /* OCR.* events */ INTEL_FLAGS_UEVENT_CONSTRAINT(0x012b, 0x1), /* OCR.* events */ - INTEL_FLAGS_UEVENT_CONSTRAINT(0x04a4, 0x1), /* TOPDOWN.BAD_SPEC_SLOTS */ - INTEL_FLAGS_UEVENT_CONSTRAINT(0x08a4, 0x1), /* TOPDOWN.BR_MISPREDICT_SLOTS */ - INTEL_FLAGS_UEVENT_CONSTRAINT(0x10a4, 0x8), /* TOPDOWN.MEMORY_BOUND_SLOTS */ - INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0x3fc), INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3), - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_STORES */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_INST_RETIRED.LOCK_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_INST_RETIRED.SPLIT_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_INST_RETIRED.SPLIT_STORES */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_INST_RETIRED.ALL_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_INST_RETIRED.ALL_STORES */ - INTEL_FLAGS_UEVENT_CONSTRAINT(0x87d0, 0x3ff), /* MEM_INST_RETIRED.ANY */ - - INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf), - - INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf), /* * Everything else is handled by PMU_FL_PEBS_ALL, because we @@ -1526,18 +1532,6 @@ struct event_constraint intel_lnc_pebs_event_constraints[] = { struct event_constraint intel_pnc_pebs_event_constraints[] = { INTEL_HYBRID_LDLAT_CONSTRAINT(0x1cd, 0xfc), INTEL_HYBRID_STLAT_CONSTRAINT(0x2cd, 0x3), - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x11d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x12d0, 0xf), /* MEM_INST_RETIRED.STLB_MISS_STORES */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x21d0, 0xf), /* MEM_INST_RETIRED.LOCK_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x41d0, 0xf), /* MEM_INST_RETIRED.SPLIT_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x42d0, 0xf), /* MEM_INST_RETIRED.SPLIT_STORES */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_LD(0x81d0, 0xf), /* MEM_INST_RETIRED.ALL_LOADS */ - INTEL_FLAGS_UEVENT_CONSTRAINT_DATALA_ST(0x82d0, 0xf), /* MEM_INST_RETIRED.ALL_STORES */ - - INTEL_FLAGS_EVENT_CONSTRAINT_DATALA_LD_RANGE(0xd1, 0xd4, 0xf), - - INTEL_FLAGS_EVENT_CONSTRAINT(0xd0, 0xf), - INTEL_FLAGS_EVENT_CONSTRAINT(0xd6, 0xf), /* * Everything else is handled by PMU_FL_PEBS_ALL, because we diff --git a/arch/x86/events/intel/lbr.c b/arch/x86/events/intel/lbr.c index cbe5c762008d..22e2a06d5786 100644 --- a/arch/x86/events/intel/lbr.c +++ b/arch/x86/events/intel/lbr.c @@ -756,10 +756,10 @@ void intel_pmu_lbr_read_32(struct cpu_hw_events *cpuc) rdmsrq(x86_pmu.lbr_from + lbr_idx, msr_lastbranch.lbr); - perf_clear_branch_entry_bitfields(br); - - br->from = msr_lastbranch.from; - br->to = msr_lastbranch.to; + *br = (struct perf_branch_entry){ + .from = msr_lastbranch.from, + .to = msr_lastbranch.to, + }; br++; } cpuc->lbr_stack.nr = i; @@ -847,14 +847,15 @@ void intel_pmu_lbr_read_64(struct cpu_hw_events *cpuc) if (abort && x86_pmu.lbr_double_abort && out > 0) out--; - perf_clear_branch_entry_bitfields(br+out); - br[out].from = from; - br[out].to = to; - br[out].mispred = mis; - br[out].predicted = pred; - br[out].in_tx = in_tx; - br[out].abort = abort; - br[out].cycles = cycles; + br[out] = (struct perf_branch_entry){ + .from = from, + .to = to, + .mispred = mis, + .predicted = pred, + .in_tx = in_tx, + .abort = abort, + .cycles = cycles, + }; out++; } cpuc->lbr_stack.nr = out; @@ -905,6 +906,7 @@ static void intel_pmu_store_lbr(struct cpu_hw_events *cpuc, struct perf_branch_entry *e; struct lbr_entry *lbr; u64 from, to, info; + bool mispred; int i; for (i = 0; i < x86_pmu.lbr_nr; i++) { @@ -921,24 +923,27 @@ static void intel_pmu_store_lbr(struct cpu_hw_events *cpuc, to = rdlbr_to(i, lbr); info = rdlbr_info(i, lbr); - perf_clear_branch_entry_bitfields(e); - - e->from = from; - e->to = to; - e->mispred = get_lbr_mispred(info); - e->predicted = !e->mispred; - e->in_tx = !!(info & LBR_INFO_IN_TX); - e->abort = !!(info & LBR_INFO_ABORT); - e->cycles = get_lbr_cycles(info); - e->type = get_lbr_br_type(info); - - /* - * Leverage the reserved field of cpuc->lbr_entries[i] to - * temporarily store the branch counters information. - * The later code will decide what content can be disclosed - * to the perf tool. Pleae see intel_pmu_lbr_counters_reorder(). - */ - e->reserved = (info >> LBR_INFO_BR_CNTR_OFFSET) & LBR_INFO_BR_CNTR_FULL_MASK; + mispred = get_lbr_mispred(info); + + *e = (struct perf_branch_entry){ + .from = from, + .to = to, + .mispred = mispred, + .predicted = !mispred, + .in_tx = !!(info & LBR_INFO_IN_TX), + .abort = !!(info & LBR_INFO_ABORT), + .cycles = get_lbr_cycles(info), + .type = get_lbr_br_type(info), + /* + * Leverage the reserved field of + * cpuc->lbr_entries[i] to temporarily store the + * branch counters information. The later code will + * decide what content can be disclosed to the perf + * tool. Pleae see intel_pmu_lbr_counters_reorder(). + */ + .reserved = (info >> LBR_INFO_BR_CNTR_OFFSET) & + LBR_INFO_BR_CNTR_FULL_MASK, + }; } cpuc->lbr_stack.nr = i; diff --git a/arch/x86/events/perf_event.h b/arch/x86/events/perf_event.h index 4680cba91340..fab9da78a5c7 100644 --- a/arch/x86/events/perf_event.h +++ b/arch/x86/events/perf_event.h @@ -517,10 +517,6 @@ struct cpu_hw_events { __EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \ HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_ST) -#define INTEL_HYBRID_LAT_CONSTRAINT(c, n) \ - __EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \ - HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_LAT_HYBRID) - #define INTEL_HYBRID_LDLAT_CONSTRAINT(c, n) \ __EVENT_CONSTRAINT(c, n, INTEL_ARCH_EVENT_MASK|X86_ALL_EVENT_FLAGS, \ HWEIGHT(n), 0, PERF_X86_EVENT_PEBS_LAT_HYBRID|PERF_X86_EVENT_PEBS_LD_HSW) diff --git a/drivers/perf/arm_brbe.c b/drivers/perf/arm_brbe.c index ba554e0c846c..254be4da8ae2 100644 --- a/drivers/perf/arm_brbe.c +++ b/drivers/perf/arm_brbe.c @@ -604,7 +604,7 @@ static bool perf_entry_from_brbe_regset(int index, struct perf_branch_entry *ent return false; brbinf = bregs.brbinf; - perf_clear_branch_entry_bitfields(entry); + *entry = (struct perf_branch_entry){ }; if (brbe_record_is_complete(brbinf)) { entry->from = bregs.brbsrc; entry->to = bregs.brbtgt; diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h index 5842552294c1..915c6fd3f084 100644 --- a/include/linux/perf_event.h +++ b/include/linux/perf_event.h @@ -1467,23 +1467,6 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data, return size; } -/* - * Clear all bitfields in the perf_branch_entry. - * The to and from fields are not cleared because they are - * systematically modified by caller. - */ -static inline void perf_clear_branch_entry_bitfields(struct perf_branch_entry *br) -{ - br->mispred = 0; - br->predicted = 0; - br->in_tx = 0; - br->abort = 0; - br->cycles = 0; - br->type = 0; - br->spec = PERF_BR_SPEC_NA; - br->reserved = 0; -} - extern void perf_output_sample(struct perf_output_handle *handle, struct perf_event_header *header, struct perf_sample_data *data, diff --git a/kernel/events/core.c b/kernel/events/core.c index db7b76d6b68a..634d2ccbab82 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx, list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) { cpc = this_cpc(pmu_ctx->pmu); + if (cpc->task_epc != pmu_ctx) + continue; + if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task) pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in); } @@ -3914,7 +3917,7 @@ static void __perf_pmu_sched_task(struct perf_cpu_pmu_context *cpc, perf_ctx_lock(cpuctx, cpuctx->task_ctx); perf_pmu_disable(pmu); - pmu->sched_task(cpc->task_epc, task, sched_in); + pmu->sched_task(&cpc->epc, task, sched_in); perf_pmu_enable(pmu); perf_ctx_unlock(cpuctx, cpuctx->task_ctx); @@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev, struct task_struct *next, bool sched_in) { - struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context); struct perf_cpu_pmu_context *cpc, *cpc2; - /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */ - if (prev == next || cpuctx->task_ctx) + if (prev == next) return; - list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) + list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) { + if (cpc->task_epc) + continue; + __perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in); + } } static void perf_event_switch(struct task_struct *task, @@ -5454,6 +5459,8 @@ attach_task_ctx_data(struct task_struct *task, struct kmem_cache *ctx_cache, } if (refcount_inc_not_zero(&old->refcount)) { + if (global) + old->global = true; free_perf_ctx_data(cd); /* unused */ return 0; } |
