diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-30 13:15:48 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-30 13:15:48 +0100 |
| commit | 24bf019cbe7e44d1e933480933b8410886bf4c60 (patch) | |
| tree | 68ba5a2536df9ec6d93b3f524b60b53011988d08 | |
| parent | d366f5b1dbc9ab26c4575690274dd8f6412805b0 (diff) | |
| parent | 1aeb52f7869a680c042fc9ae806281e8f60469f7 (diff) | |
| download | linux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.tar.gz linux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.zip | |
Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git
# Conflicts:
# Documentation/scheduler/index.rst
# arch/arm64/configs/defconfig
342 files changed, 10161 insertions, 2240 deletions
diff --git a/Documentation/ABI/testing/sysfs-devices-system-cpu b/Documentation/ABI/testing/sysfs-devices-system-cpu index 73c8c204820d..545687ebadcc 100644 --- a/Documentation/ABI/testing/sysfs-devices-system-cpu +++ b/Documentation/ABI/testing/sysfs-devices-system-cpu @@ -693,15 +693,21 @@ Description: Umwait control Low order two bits must be zero. What: /sys/devices/system/cpu/sev + /sys/devices/system/cpu/sev/sev_status /sys/devices/system/cpu/sev/vmpl Date: May 2024 Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org> Description: Secure Encrypted Virtualization (SEV) information - This directory is only present when running as an SEV-SNP guest. + This directory is only present when running as an SEV guest. + + sev_status: Reports the value of the SEV_STATUS MSR which + enumerates the enabled features of an SEV + environment. vmpl: Reports the Virtual Machine Privilege Level (VMPL) at which - the SEV-SNP guest is running. + the SEV-SNP guest is running. This file is only present + when running as an SEV-SNP guest. What: /sys/devices/system/cpu/svm @@ -810,3 +816,17 @@ Date: Nov 2022 Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org> Description: (RO) the list of CPUs that can be brought online. + +What: /sys/devices/system/cpu/preferred +Date: Sep 2026 +Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org> +Description: + (RO) the list of preferred CPUs applicable in + paravirtualized environments. + + The steal governor driver dynamically adjusts this mask + based on observed steal time. Scheduling tasks on + CPUs outside of this list may lead to performance + degradations due to underlying physical CPU contention. + + See Documentation/scheduler/sched-paravirt.rst for more details. diff --git a/Documentation/ABI/testing/sysfs-platform-ts5500 b/Documentation/ABI/testing/sysfs-platform-ts5500 deleted file mode 100644 index e685957caa12..000000000000 --- a/Documentation/ABI/testing/sysfs-platform-ts5500 +++ /dev/null @@ -1,54 +0,0 @@ -What: /sys/devices/platform/ts5500/adc -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Indicates the presence of an A/D Converter. If it is present, - it will display "1", otherwise "0". - -What: /sys/devices/platform/ts5500/ereset -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Indicates the presence of an external reset. If it is present, - it will display "1", otherwise "0". - -What: /sys/devices/platform/ts5500/id -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Product ID of the TS board. TS-5500 ID is 0x60. - -What: /sys/devices/platform/ts5500/jumpers -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Bitfield showing the jumpers' state. If a jumper is present, - the corresponding bit is set. For instance, 0x0e means jumpers - 2, 3 and 4 are set. - -What: /sys/devices/platform/ts5500/name -Date: July 2014 -KernelVersion: 3.16 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Model name of the TS board, e.g. "TS-5500". - -What: /sys/devices/platform/ts5500/rs485 -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Indicates the presence of the RS485 option. If it is present, - it will display "1", otherwise "0". - -What: /sys/devices/platform/ts5500/sram -Date: January 2013 -KernelVersion: 3.7 -Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -Description: - Indicates the presence of the SRAM option. If it is present, - it will display "1", otherwise "0". diff --git a/Documentation/arch/arm64/silicon-errata.rst b/Documentation/arch/arm64/silicon-errata.rst index ac3248b9f2f3..25afa79266a4 100644 --- a/Documentation/arch/arm64/silicon-errata.rst +++ b/Documentation/arch/arm64/silicon-errata.rst @@ -53,6 +53,9 @@ stable kernels. | Allwinner | A64/R18 | UNKNOWN1 | SUN50I_ERRATUM_UNKNOWN1 | +----------------+-----------------+-----------------+-----------------------------+ +----------------+-----------------+-----------------+-----------------------------+ +| Altera | SoCFPGA Agilex5 | 2.1.23 |ALTERA_ERRATUM_AGILEX5_2_1_23| ++----------------+-----------------+-----------------+-----------------------------+ ++----------------+-----------------+-----------------+-----------------------------+ | Ampere | AmpereOne | AC03_CPU_38 | AMPERE_ERRATUM_AC03_CPU_38 | +----------------+-----------------+-----------------+-----------------------------+ | Ampere | AmpereOne | AC03_CPU_57 | N/A | diff --git a/Documentation/arch/x86/tdx.rst b/Documentation/arch/x86/tdx.rst index 3303499ad4c6..a36ea2bd4301 100644 --- a/Documentation/arch/x86/tdx.rst +++ b/Documentation/arch/x86/tdx.rst @@ -200,6 +200,27 @@ reflects the TCB of the currently running TDX module and therefore changes after an update. By contrast, TEE_TCB_SVN reflects the TCB at TD launch time and is not affected. +Dynamic PAMT +------------ + +The Physical Address Metadata Table (PAMT) is metadata in which the TDX +module keeps data about each physical page (think struct page). Space +for it is allocated by the VMM, consumes up to about 0.4% of system +memory and needs to be supplied to the TDX module when the TDX module is +first loaded. + +Dynamic PAMT is an add-on feature that allows a VMM to dynamically +allocate the part of the PAMT which tracks 4KB pages. This reduces the +amount of memory that TDX consumes while TDs are not in use. + +When Dynamic PAMT is in use, dmesg shows it like:: + + [..] virt/tdx: Enable Dynamic PAMT + [..] virt/tdx: 10092 KB allocated for PAMT + [..] virt/tdx: TDX-Module initialized + +Dynamic PAMT is enabled automatically if supported. + TDX Interaction to Other Kernel Components ------------------------------------------ diff --git a/Documentation/arch/x86/xstate.rst b/Documentation/arch/x86/xstate.rst index cec05ac464c1..e2944f744255 100644 --- a/Documentation/arch/x86/xstate.rst +++ b/Documentation/arch/x86/xstate.rst @@ -172,3 +172,57 @@ are extended to control the guest permission: Note that some VMMs may have already established a set of supported state components. These options are not presumed to support any particular VMM. + +Signal Frame Layout and Portability +----------------------------------- + +The signal frame is designed to be self-describing and portable. This is +especially important for checkpoint/restore tools like CRIU, which may restore +a process on a different host than where it was checkpointed. A signal frame +created on a machine with fewer CPU features can be successfully restored on a +machine with more CPU features, but not vice-versa. + +Signal Frame Software Reserved Bytes +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +On CPUs supporting XSAVE, bytes 464..511 in the 512-byte FXSAVE/FXRSTOR frame +are reserved for software use and contain ``struct _fpx_sw_bytes`` (defined in +``<uapi/asm/sigcontext.h>``):: + + struct _fpx_sw_bytes { + __u32 magic1; + __u32 extended_size; + __u64 xfeatures; + __u32 xstate_size; + __u32 padding[7]; + }; + +- ``magic1``: Set to ``FP_XSTATE_MAGIC1`` (``0x46505853U``) if an extended + xstate context is present; 0 for a legacy frame. +- ``extended_size``: The total size allocated on the stack for the frame, + measured from the ``fpstate`` pointer. In 32-bit signal frames, this also + includes the 112-byte legacy FPU state prefix of ``struct _fpstate_32``. +- ``xfeatures``: The mask of xstate features saved in the frame. +- ``xstate_size``: The actual size of the xstate context for the enabled + features (including the 512-byte FXSAVE area and the 64-byte XSAVE header). + +The kernel uses ``xstate_size`` in conjunction with the pointer to the xstate +context to locate the ``FP_XSTATE_MAGIC2`` (``0x46505845U``) marker right after +the xstate context (at ``xstate_context + xstate_size``). In 64-bit signal frames, +the ``fpstate`` pointer points directly to the xstate context. In 32-bit signal +frames (including 32-bit compat tasks on 64-bit kernels), the ``fpstate`` +pointer points to ``struct _fpstate_32``, which contains the 112-byte legacy +FPU state followed by the 512-byte FXSR state (and any extended xstate). Since +there is no standalone UAPI structure defined for just the 112-byte legacy +state, the xstate context starts at ``fpstate + 112`` (and ``extended_size`` +spans the entire allocation from ``fpstate``). + +Portability Constraints +^^^^^^^^^^^^^^^^^^^^^^^ + +Signal frame portability is constrained by the architectural XSAVE layout. +Restoration is supported only if the destination host supports all features +present in the frame and uses matching component offsets and sizes for them. +While layout compatibility is generally maintained across CPUs from the same +vendor, differences can occur across vendors or if the XSAVE space of a +deprecated feature (e.g. MPX) is repurposed for a newer feature (e.g. APX). diff --git a/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml b/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml index 518fbf6b2761..be4dc95e3916 100644 --- a/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml +++ b/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml @@ -59,6 +59,7 @@ properties: - qcom,sm8650-pdc - qcom,sm8750-pdc - qcom,x1e80100-pdc + - qcom,x1p42100-pdc - const: qcom,pdc reg: diff --git a/Documentation/driver-api/index.rst b/Documentation/driver-api/index.rst index 6601a258690f..26b7638a327d 100644 --- a/Documentation/driver-api/index.rst +++ b/Documentation/driver-api/index.rst @@ -139,6 +139,7 @@ Subsystem-specific APIs sm501 soundwire/index spi + steal-governor surface_aggregator/index switchtec sync_file diff --git a/Documentation/driver-api/steal-governor.rst b/Documentation/driver-api/steal-governor.rst new file mode 100644 index 000000000000..3817eedb38d7 --- /dev/null +++ b/Documentation/driver-api/steal-governor.rst @@ -0,0 +1,151 @@ +.. SPDX-License-Identifier: GPL-2.0 + +Steal Governor +============== + +:Author: Shrikanth Hegde <sshegde@linux.ibm.com> + +Introduction +============ + +The steal governor is aimed at mitigating the Noisy Neighbour problem +which occurs in paravirtualized environments with CPU overcommit. +The performance of a workload running in one VM gets degraded by +the activity of other VMs on the same host. As a result, all VMs +collectively make slower forward progress. + +In such systems, high utilization in all VMs causes the hypervisor to +frequently preempt vCPUs. This vCPU preemption is expensive. +To mitigate this, the kernel aims to restrict workloads to a subset of +Preferred CPUs to reduce physical CPU contention. +A detailed explanation of Preferred CPUs is available in +``Documentation/scheduler/sched-paravirt.rst``. + +The steal governor selects ``CONFIG_PREFERRED_CPU=y`` which enables the +scheduler core infrastructure to move the tasks to Preferred CPUs where +possible. The driver controls the policy decisions regarding the state of +preferred CPUs. That is, this driver decides which CPUs are preferred +and which CPUs are non-preferred. + +The driver code is available at ``drivers/virt/steal_governor.c``. + +Core idea +========= + +steal time is an indication available today in Guest which shows contention +for underlying physical CPU. Use it as a hint in the guest to fold the +workload to a reduced set of vCPUs. When there is contention, steal time +will show up in all the guests. When each guest honors the hint and folds +the workload to a smaller set of vCPUs (Preferred CPUs), it reduces the +contention and thereby reduces vCPU preemption. +This is achieved without any cross-guest communication. + +Steal governor driver effectively does: + +1. Periodically computes the steal ratio using accumulated steal time + across possible CPUs, normalized by the number of active CPUs. + +2. If steal ratio is greater than high threshold, reduce the number of + preferred CPUs by 1 core. Ensure at least one core is left always. + Skip changing the state of offline CPUs in that core. + +3. If steal ratio is less than or equal to low threshold, increase the + number of preferred CPUs by 1 core. If preferred is same as active, + nothing to be done. Skip changing the state of offline CPUs. + This helps to handle cases where few CPUs are offline in a core and + those offline CPUs will not be marked as preferred. + +4. Ensure preferred CPUs is always subset of active CPUs. + On feature disable it is same as active CPUs. + +This feature works best only when all the VMs enable the feature as +it is a co-operative scheme. If a specific VM doesn't enable this feature +it may end up with more CPUs than others, still should lead to better +performance when seen from system view. Those who enable this driver must +ensure it is enabled in all VMs. + +Note that this driver is strictly intended for actual guests; for example, +loading this module in a privileged VM like Xen Dom0 is blocked. + +Workload considerations +======================= + +The steal governor is useful for workloads where vCPU preemption has +costs beyond the lost CPU time, such as lock-holder preemption, critical +sections, communicating threads, and cache or TLB disruption. + +Pure CPU-time workloads with independent workers may not benefit and +could see a small regression due to additional guest scheduling overhead. + +Module Parameters +================= + +interval_ms +----------- + +How often steal governor checks for steal time. +Default: 1000 i.e. 1 second. Value should be in between 100ms to 100sec. + +This controls how fast steal governor driver reacts to changes to the +contention of physical CPUs. Since it does a fair amount of work, setting +too low may have overhead. Setting it too high might render it ineffective. + +low_threshold +------------- + +lower threshold value in percentage * 100. +Default: 200, i.e. 2% steal is considered as low threshold. +Can't be higher than high_threshold. + +This determines what values should be considered as nil/no steal values. +When steal governor sees steal ratio is less than or equal to this value, +it will increase the preferred CPUs by 1 core. +Using zero might cause oscillations. + +high_threshold +-------------- + +higher threshold value in percentage * 100 +Default: 500, i.e. 5% steal is considered as high threshold. +Can't be lower than low_threshold. Must be less than 10000. + +This determines what values should be considered as high steal values. +When steal governor sees steal ratio is higher than this value, it will +reduce the preferred CPUs by 1 core. + +Limitations of default values +----------------------------- + +Because of the vast diversity in VM configurations and different +architectures, the default thresholds may not be optimal for all systems. +Users may need to tune these parameters based on the system under +test to achieve the best results. + +The governor sums the steal time across all possible CPUs, which ensures +the accumulated steal time remains a monotonically increasing value. +However, to calculate the effective steal ratio, it divides this sum +by the number of active CPUs. Because only active CPUs contribute to +the steal time delta, this prevents threshold dilution on sparsely +populated systems. + +The driver reduces/increases preferred CPUs by core-level. This could provide +faster convergence for hypervisors such as powerVM. But on KVM and Xen +convergence could be slower depending on the configuration. +Using a smaller interval_ms could help one to expedite it. + +Reasons for CONFIG_STEAL_GOVERNOR=m +=================================== + +Selecting this driver makes CONFIG_PREFERRED_CPU=y. That makes configs +driven by user preference. Though one can have CONFIG_STEAL_GOVERNOR=y, +It is recommended to build CONFIG_STEAL_GOVERNOR=m due to below reasons: + +1. Doing periodic work has additional overheads. Enabling this driver + in systems where steal time cannot happen is of no use. There is no + benefit with additional overheads in such systems. + +2. This works well when all VMs work in a co-operative manner. When an + administrative user enables it in one VM, he/she will likely enable + it in all VMs. + +3. User can tweak the module parameters by reloading the module. diff --git a/Documentation/filesystems/resctrl.rst b/Documentation/filesystems/resctrl.rst index e4b66af55ffb..b52795e03303 100644 --- a/Documentation/filesystems/resctrl.rst +++ b/Documentation/filesystems/resctrl.rst @@ -236,12 +236,11 @@ with respect to allocation: user can request. "bandwidth_gran": - The granularity in which the memory bandwidth - percentage is allocated. The allocated - b/w percentage is rounded off to the next - control step available on the hardware. The - available bandwidth control steps are: - min_bandwidth + N * bandwidth_gran. + The approximate granularity in which the memory bandwidth + percentage is allocated. The allocated bandwidth percentage is + rounded up or down to the closest control step available on the + hardware. The available hardware steps are no larger than this + value. "delay_linear": Indicates if the delay scale is linear or @@ -643,7 +642,7 @@ When monitoring is enabled all MON groups will also contain: during execution of instructions summed across all logical CPUs on a package for the current monitoring group. - "activity" also reports a floating point value (in Farads). This provides + "activity" also reports a floating point value (in nanofarads). This provides an estimate of work done independent of the frequency that the CPUs used for execution. @@ -881,8 +880,10 @@ The minimum bandwidth percentage value for each cpu model is predefined and can be looked up through "info/MB/min_bandwidth". The bandwidth granularity that is allocated is also dependent on the cpu model and can be looked up at "info/MB/bandwidth_gran". The available bandwidth -control steps are: min_bw + N * bw_gran. Intermediate values are rounded -to the next control step available on the hardware. +control steps are, approximately, min_bw + N * bw_gran. The steps may +appear irregular due to rounding to an exact percentage: bw_gran is the +maximum interval between the percentage values corresponding to any two +adjacent steps in the hardware. The bandwidth throttling is a core specific mechanism on some of Intel SKUs. Using a high bandwidth and a low bandwidth setting on two threads diff --git a/Documentation/scheduler/index.rst b/Documentation/scheduler/index.rst index d6d75421756a..791178ac8bec 100644 --- a/Documentation/scheduler/index.rst +++ b/Documentation/scheduler/index.rst @@ -23,6 +23,7 @@ Scheduler sched-stats sched-ext sched-debug + sched-paravirt sched-preemption text_files diff --git a/Documentation/scheduler/sched-paravirt.rst b/Documentation/scheduler/sched-paravirt.rst new file mode 100644 index 000000000000..311cdbf2722b --- /dev/null +++ b/Documentation/scheduler/sched-paravirt.rst @@ -0,0 +1,67 @@ +.. SPDX-License-Identifier: GPL-2.0 +.. _sched-paravirt: + +Preferred CPUs +============== + +In paravirtualized environments CPU overcommit is a common scenario. +i.e. the sum of virtual CPUs (vCPUs) of all VMs is greater than number of +physical CPUs (pCPUs). Under such conditions when all or many VMs have +high utilization, hypervisor won't be able to satisfy the CPU requirement +and has to context switch within or across VMs. The hypervisor needs to +preempt one vCPU to run another. This is called vCPU preemption. +This is more expensive compared to task context switch within a vCPU, since +hypervisor lacks vCPU context and could preempt a critical section which +slows forward progress. + +In such cases it is better that combined vCPU demand from all VMs is reduced +by not using some of the vCPUs in each VM. vCPUs where workload can be safely +scheduled which won't increase any contention for pCPU are called +"Preferred CPUs". + +One of the main design constructs is that preferred CPUs are always +a subset of active CPUs. In most cases preferred CPUs will be same as +active CPUs. When there is pCPU contention, Preferred CPUs will reduce +based on the steal time. When the pCPU contention goes away as indicated +by steal time, Preferred CPUs could become same as active CPUs again. +The policy decisions are to be taken by driver. +For example, steal_governor. Look at its documentation for more +details. (``drivers/virt/steal_governor.c``) + +Scheduling decisions such as wakeup, pushing the task etc, need this +CPU state info. This is maintained in ``cpu_preferred_mask``. +vCPUs which are not in ``cpu_preferred_mask`` should be treated as vCPUs which +should not be used at this moment provided it doesn't break user affinity. + +This is achieved by: + +1. Selecting a preferred CPU at wakeup using fallback mechanism. +2. Pushing the task away from non-preferred CPU at tick. +3. Selecting only preferred CPUs for load balance. + +``/sys/devices/system/cpu/preferred`` prints the current ``cpu_preferred_mask`` +in cpulist format. + +Notes: + +1. This feature is available under ``CONFIG_PREFERRED_CPU``. Driver which + makes decisions should enable it. For example, steal_governor driver + (``CONFIG_STEAL_GOVERNOR``). On enabling the driver, CPU preferred state + can change based on steal time. Without the driver, preferred CPUs is + same as active CPUs. + +2. This feature works for the FAIR class only. + +3. A pinned task, which can't be moved to preferred CPUs will continue + to run based on its affinity. But no load balancing happens if it is affined + only on non-preferred CPUs. + +4. Decision to change the preferred CPU state is driven by the kernel. + Hence it shouldn't break user affinities. One of the main reasons why + CPU hotplug or Isolated cpuset partitions was not a solution. + +5. This feature works best only when all the Guest VMs enable the feature as + it is a co-operative scheme. If a specific VM doesn't enable this feature + it may end up with more CPUs than others, still should lead to better + performance when seen from system view. + Users who enable this driver must ensure it is enabled in all Guest VMs. diff --git a/MAINTAINERS b/MAINTAINERS index 11e412d802b1..d4a65bd29a4a 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -26514,6 +26514,15 @@ F: rust/helpers/jump_label.c F: rust/kernel/generated_arch_static_branch_asm.rs.S F: rust/kernel/jump_label.rs +STEAL GOVERNOR DRIVER +M: Shrikanth Hegde <sshegde@linux.ibm.com> +R: Yury Norov <yury.norov@gmail.com> +L: linux-kernel@vger.kernel.org +S: Maintained +T: git git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git sched/core +F: Documentation/driver-api/steal-governor.rst +F: drivers/virt/steal_governor.c + STI AUDIO (ASoC) DRIVERS M: Arnaud Pouliquen <arnaud.pouliquen@foss.st.com> L: linux-sound@vger.kernel.org @@ -27104,11 +27113,6 @@ S: Maintained F: Documentation/process/contribution-maturity-model.rst F: Documentation/process/researcher-guidelines.rst -TECHNOLOGIC SYSTEMS TS-5500 PLATFORM SUPPORT -M: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com> -S: Maintained -F: arch/x86/platform/ts5500/ - TECHNOTREND USB IR RECEIVER M: Sean Young <sean@mess.org> L: linux-media@vger.kernel.org diff --git a/arch/Kconfig b/arch/Kconfig index a438eda84cd4..1163736de1f9 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -1696,44 +1696,6 @@ config HAVE_STATIC_CALL_INLINE depends on HAVE_STATIC_CALL select OBJTOOL -config HAVE_PREEMPT_DYNAMIC - bool - -config HAVE_PREEMPT_DYNAMIC_CALL - bool - depends on HAVE_STATIC_CALL - select HAVE_PREEMPT_DYNAMIC - help - An architecture should select this if it can handle the preemption - model being selected at boot time using static calls. - - Where an architecture selects HAVE_STATIC_CALL_INLINE, any call to a - preemption function will be patched directly. - - Where an architecture does not select HAVE_STATIC_CALL_INLINE, any - call to a preemption function will go through a trampoline, and the - trampoline will be patched. - - It is strongly advised to support inline static call to avoid any - overhead. - -config HAVE_PREEMPT_DYNAMIC_KEY - bool - depends on HAVE_ARCH_JUMP_LABEL - select HAVE_PREEMPT_DYNAMIC - help - An architecture should select this if it can handle the preemption - model being selected at boot time using static keys. - - Each preemption function will be given an early return based on a - static key. This should have slightly lower overhead than non-inline - static calls, as this effectively inlines each trampoline into the - start of its callee. This may avoid redundant work, and may - integrate better with CFI schemes. - - This will have greater overhead than using inline static calls as - the call to the preemption function cannot be entirely elided. - config ARCH_WANT_LD_ORPHAN_WARN bool help diff --git a/arch/arm/kernel/perf_regs.c b/arch/arm/kernel/perf_regs.c index 0529f90395c9..838d701adf4d 100644 --- a/arch/arm/kernel/perf_regs.c +++ b/arch/arm/kernel/perf_regs.c @@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1ULL << PERF_REG_ARM_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; @@ -31,9 +31,3 @@ u64 perf_reg_abi(struct task_struct *task) return PERF_SAMPLE_REGS_ABI_32; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index 88de433a78ef..1ec35aada772 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -217,7 +217,6 @@ config ARM64 select HAVE_PERF_EVENTS_NMI if ARM64_PSEUDO_NMI select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP - select HAVE_PREEMPT_DYNAMIC_KEY select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RELIABLE_STACKTRACE select HAVE_POSIX_CPU_TIMERS_TASK_WORK @@ -1422,6 +1421,15 @@ config NVIDIA_OLYMPUS_1027_ERRATUM If unsure, say Y. +config ALTERA_ERRATUM_AGILEX5_2_1_23 + bool + help + The Altera SoCFPGA Agilex5 GIC600 SoC integration has ACE-lite + addressing limited to the first 32bit of physical address space, + so the ITS cannot access memory above 4GB. + + Selected by ARCH_INTEL_SOCFPGA, as all Agilex5 devices are affected. + config RENESAS_ERRATUM_GEN4GICITS1 bool "Renesas R-Car Gen4: GIC600 can not access physical addresses above 4 GiB" default y diff --git a/arch/arm64/Kconfig.platforms b/arch/arm64/Kconfig.platforms index 962e67fcb5fb..c52dc955df5f 100644 --- a/arch/arm64/Kconfig.platforms +++ b/arch/arm64/Kconfig.platforms @@ -370,6 +370,7 @@ config ARCH_SEATTLE config ARCH_INTEL_SOCFPGA bool "Intel's SoCFPGA ARMv8 Families" + select ALTERA_ERRATUM_AGILEX5_2_1_23 help This enables support for Intel's SoCFPGA ARMv8 families: Stratix 10 (ex. Altera), Stratix10 Software Virtual Platform, diff --git a/arch/arm64/boot/dts/qcom/purwa.dtsi b/arch/arm64/boot/dts/qcom/purwa.dtsi index c698e6cb2543..4348dd3d1dc5 100644 --- a/arch/arm64/boot/dts/qcom/purwa.dtsi +++ b/arch/arm64/boot/dts/qcom/purwa.dtsi @@ -224,6 +224,11 @@ compatible = "qcom,x1p42100-qmp-gen4x4-pcie-phy"; }; +/* X1P42100 PDC is same as X1E80100, but without hardware register bug */ +&pdc { + compatible = "qcom,x1p42100-pdc", "qcom,pdc"; +}; + &qfprom { gpu_speed_bin: gpu-speed-bin@119 { reg = <0x119 0x2>; diff --git a/arch/arm64/configs/defconfig b/arch/arm64/configs/defconfig index cbb43e5890bc..fce418fe6ff6 100644 --- a/arch/arm64/configs/defconfig +++ b/arch/arm64/configs/defconfig @@ -1638,8 +1638,6 @@ CONFIG_PWM_VISCONTI=m CONFIG_PWM_XILINX=m CONFIG_SL28CPLD_INTC=y CONFIG_XILINX_INTC=y -CONFIG_QCOM_PDC=y -CONFIG_QCOM_MPM=y CONFIG_TI_SCI_INTR_IRQCHIP=m CONFIG_TI_SCI_INTA_IRQCHIP=m CONFIG_RESET_GPIO=m diff --git a/arch/arm64/include/asm/preempt.h b/arch/arm64/include/asm/preempt.h index 3c85cacc1d19..75a592b7f47f 100644 --- a/arch/arm64/include/asm/preempt.h +++ b/arch/arm64/include/asm/preempt.h @@ -97,19 +97,9 @@ static inline bool __preempt_count_dec_and_test(void) void preempt_schedule(void); void preempt_schedule_notrace(void); -#ifdef CONFIG_PREEMPT_DYNAMIC - -void dynamic_preempt_schedule(void); -#define __preempt_schedule() dynamic_preempt_schedule() -void dynamic_preempt_schedule_notrace(void); -#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace() - -#else /* CONFIG_PREEMPT_DYNAMIC */ - #define __preempt_schedule() preempt_schedule() #define __preempt_schedule_notrace() preempt_schedule_notrace() -#endif /* CONFIG_PREEMPT_DYNAMIC */ #endif /* CONFIG_PREEMPTION */ #endif /* __ASM_PREEMPT_H */ diff --git a/arch/arm64/kernel/paravirt.c b/arch/arm64/kernel/paravirt.c index 572efb96b23f..30bf61d031eb 100644 --- a/arch/arm64/kernel/paravirt.c +++ b/arch/arm64/kernel/paravirt.c @@ -157,9 +157,9 @@ int __init pv_time_init(void) static_call_update(pv_steal_clock, para_steal_clock); - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); pr_info("using stolen time PV\n"); diff --git a/arch/arm64/kernel/perf_regs.c b/arch/arm64/kernel/perf_regs.c index b4eece3eb17d..71a3e0238de4 100644 --- a/arch/arm64/kernel/perf_regs.c +++ b/arch/arm64/kernel/perf_regs.c @@ -77,7 +77,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1ULL << PERF_REG_ARM64_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { u64 reserved_mask = REG_RESERVED; @@ -98,9 +98,3 @@ u64 perf_reg_abi(struct task_struct *task) return PERF_SAMPLE_REGS_ABI_64; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/csky/kernel/perf_regs.c b/arch/csky/kernel/perf_regs.c index 09b7f88a2d6a..c932a96afc56 100644 --- a/arch/csky/kernel/perf_regs.c +++ b/arch/csky/kernel/perf_regs.c @@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1ULL << PERF_REG_CSKY_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; @@ -31,9 +31,3 @@ u64 perf_reg_abi(struct task_struct *task) return PERF_SAMPLE_REGS_ABI_32; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index 0bd8503fd5c4..0aad0004b010 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -169,7 +169,6 @@ config LOONGARCH select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select HAVE_PREEMPT_DYNAMIC_KEY select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RELIABLE_STACKTRACE if UNWINDER_ORC select HAVE_RETHOOK diff --git a/arch/loongarch/kernel/paravirt.c b/arch/loongarch/kernel/paravirt.c index 10821cce554c..e8965a3f8082 100644 --- a/arch/loongarch/kernel/paravirt.c +++ b/arch/loongarch/kernel/paravirt.c @@ -308,10 +308,10 @@ int __init pv_time_init(void) static_call_update(pv_steal_clock, paravt_steal_clock); - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); #endif if (static_key_enabled(&virt_preempt_key)) diff --git a/arch/loongarch/kernel/perf_regs.c b/arch/loongarch/kernel/perf_regs.c index 263ac4ab5af6..164514f40ae0 100644 --- a/arch/loongarch/kernel/perf_regs.c +++ b/arch/loongarch/kernel/perf_regs.c @@ -25,7 +25,7 @@ u64 perf_reg_abi(struct task_struct *tsk) } #endif /* CONFIG_32BIT */ -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask) return -EINVAL; @@ -45,9 +45,3 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) return regs->regs[idx]; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/mips/kernel/perf_regs.c b/arch/mips/kernel/perf_regs.c index e686780d1647..00a5201dbd5d 100644 --- a/arch/mips/kernel/perf_regs.c +++ b/arch/mips/kernel/perf_regs.c @@ -28,7 +28,7 @@ u64 perf_reg_abi(struct task_struct *tsk) } #endif /* CONFIG_32BIT */ -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask) return -EINVAL; @@ -60,9 +60,3 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) return (s64)v; /* Sign extend if 32-bit. */ } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/parisc/kernel/perf_regs.c b/arch/parisc/kernel/perf_regs.c index 10a1a5f06a18..4f21aab5405c 100644 --- a/arch/parisc/kernel/perf_regs.c +++ b/arch/parisc/kernel/perf_regs.c @@ -34,7 +34,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1ULL << PERF_REG_PARISC_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; @@ -53,9 +53,3 @@ u64 perf_reg_abi(struct task_struct *task) return PERF_SAMPLE_REGS_ABI_64; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 3f59b201b62f..c3b6cd655606 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -278,7 +278,6 @@ config PPC select HAVE_PERF_EVENTS_NMI if PPC64 select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP - select HAVE_PREEMPT_DYNAMIC_KEY select HAVE_POSIX_CPU_TIMERS_TASK_WORK select HAVE_RETHOOK if KPROBES select HAVE_REGS_AND_STACK_ACCESS_API diff --git a/arch/powerpc/perf/perf_regs.c b/arch/powerpc/perf/perf_regs.c index 350dccb0143c..a01d8a903640 100644 --- a/arch/powerpc/perf/perf_regs.c +++ b/arch/powerpc/perf/perf_regs.c @@ -125,7 +125,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) return regs_get_register(regs, pt_regs_offset[idx]); } -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; diff --git a/arch/powerpc/platforms/pseries/setup.c b/arch/powerpc/platforms/pseries/setup.c index f29e7547c995..c181df4dec70 100644 --- a/arch/powerpc/platforms/pseries/setup.c +++ b/arch/powerpc/platforms/pseries/setup.c @@ -876,9 +876,9 @@ static void __init pSeries_setup_arch(void) static_branch_enable(&shared_processor); pv_spinlocks_init(); #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); #endif } diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index f409f264d8c4..0ccb72101d28 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -194,7 +194,6 @@ config RISCV select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select HAVE_PREEMPT_DYNAMIC_KEY select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RETHOOK select HAVE_RSEQ diff --git a/arch/riscv/include/asm/smp.h b/arch/riscv/include/asm/smp.h index 0ecc67641b09..bed39fff1f8a 100644 --- a/arch/riscv/include/asm/smp.h +++ b/arch/riscv/include/asm/smp.h @@ -15,6 +15,18 @@ struct seq_file; extern unsigned long boot_cpu_hartid; +enum ipi_message_type { + IPI_RESCHEDULE, + IPI_CALL_FUNC, + IPI_CPU_STOP, + IPI_CPU_CRASH_STOP, + IPI_IRQ_WORK, + IPI_TIMER, + IPI_CPU_BACKTRACE, + IPI_KGDB_ROUNDUP, + IPI_MAX +}; + #ifdef CONFIG_SMP #include <linux/jump_label.h> diff --git a/arch/riscv/kernel/paravirt.c b/arch/riscv/kernel/paravirt.c index 5f56be79cd06..9c13a6f1ea2a 100644 --- a/arch/riscv/kernel/paravirt.c +++ b/arch/riscv/kernel/paravirt.c @@ -116,9 +116,9 @@ int __init pv_time_init(void) static_call_update(pv_steal_clock, pv_time_steal_clock); - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); pr_info("Computing paravirt steal-time\n"); diff --git a/arch/riscv/kernel/perf_regs.c b/arch/riscv/kernel/perf_regs.c index fd304a248de6..1ecc8760b88b 100644 --- a/arch/riscv/kernel/perf_regs.c +++ b/arch/riscv/kernel/perf_regs.c @@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1ULL << PERF_REG_RISCV_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; @@ -35,9 +35,3 @@ u64 perf_reg_abi(struct task_struct *task) #endif } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} diff --git a/arch/riscv/kernel/sbi-ipi.c b/arch/riscv/kernel/sbi-ipi.c index 0cc5559c08d8..eeec178a9b95 100644 --- a/arch/riscv/kernel/sbi-ipi.c +++ b/arch/riscv/kernel/sbi-ipi.c @@ -57,7 +57,7 @@ void __init sbi_ipi_init(void) return; } - virq = ipi_mux_create(BITS_PER_BYTE, sbi_send_ipi); + virq = ipi_mux_create(IPI_MAX, sbi_send_ipi); if (virq <= 0) { pr_err("unable to create muxed IPIs\n"); irq_dispose_mapping(sbi_ipi_virq); @@ -75,7 +75,7 @@ void __init sbi_ipi_init(void) "irqchip/sbi-ipi:starting", sbi_ipi_starting_cpu, NULL); - riscv_ipi_set_virq_range(virq, BITS_PER_BYTE); + riscv_ipi_set_virq_range(virq, IPI_MAX); pr_info("providing IPIs using SBI IPI extension\n"); /* diff --git a/arch/riscv/kernel/smp.c b/arch/riscv/kernel/smp.c index fa66f9c97d74..8930b62b15e7 100644 --- a/arch/riscv/kernel/smp.c +++ b/arch/riscv/kernel/smp.c @@ -28,18 +28,6 @@ #include <asm/cacheflush.h> #include <asm/cpu_ops.h> -enum ipi_message_type { - IPI_RESCHEDULE, - IPI_CALL_FUNC, - IPI_CPU_STOP, - IPI_CPU_CRASH_STOP, - IPI_IRQ_WORK, - IPI_TIMER, - IPI_CPU_BACKTRACE, - IPI_KGDB_ROUNDUP, - IPI_MAX -}; - static const char * const ipi_names[] = { [IPI_RESCHEDULE] = "Rescheduling interrupts", [IPI_CALL_FUNC] = "Function call interrupts", diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 55b074e748f8..2b2b51224d5d 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -245,7 +245,6 @@ config S390 select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select HAVE_PREEMPT_DYNAMIC_KEY select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RELIABLE_STACKTRACE select HAVE_RETHOOK diff --git a/arch/s390/include/asm/preempt.h b/arch/s390/include/asm/preempt.h index 5560d5fca2a3..60c6f019ec45 100644 --- a/arch/s390/include/asm/preempt.h +++ b/arch/s390/include/asm/preempt.h @@ -155,20 +155,9 @@ static __always_inline int __preempt_count_sub_return(int val) void preempt_schedule(void); void preempt_schedule_notrace(void); -#ifdef CONFIG_PREEMPT_DYNAMIC - -void dynamic_preempt_schedule(void); -void dynamic_preempt_schedule_notrace(void); -#define __preempt_schedule() dynamic_preempt_schedule() -#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace() - -#else /* CONFIG_PREEMPT_DYNAMIC */ - #define __preempt_schedule() preempt_schedule() #define __preempt_schedule_notrace() preempt_schedule_notrace() -#endif /* CONFIG_PREEMPT_DYNAMIC */ - #endif /* CONFIG_PREEMPTION */ #endif /* __ASM_PREEMPT_H */ diff --git a/arch/s390/kernel/hiperdispatch.c b/arch/s390/kernel/hiperdispatch.c index 8494823559b1..1397dab8a677 100644 --- a/arch/s390/kernel/hiperdispatch.c +++ b/arch/s390/kernel/hiperdispatch.c @@ -207,16 +207,12 @@ static unsigned long hd_calculate_steal_percentage(void) { unsigned long time_delta, steal_delta, steal, percentage; static ktime_t prev; - int cpus, cpu; + int cpus; ktime_t now; - cpus = 0; - steal = 0; percentage = 0; - for_each_cpu(cpu, &hd_vmvl_cpumask) { - steal += kcpustat_cpu(cpu).cpustat[CPUTIME_STEAL]; - cpus++; - } + steal = kcpustat_field_total(CPUTIME_STEAL, &hd_vmvl_cpumask); + cpus = cpumask_weight(&hd_vmvl_cpumask); /* * If there is no vertical medium and low CPUs steal time * is 0 as vertical high CPUs shouldn't experience steal time. diff --git a/arch/s390/kernel/perf_regs.c b/arch/s390/kernel/perf_regs.c index 7b305f1456f8..6496fd23c540 100644 --- a/arch/s390/kernel/perf_regs.c +++ b/arch/s390/kernel/perf_regs.c @@ -34,7 +34,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) #define REG_RESERVED (~((1UL << PERF_REG_S390_MAX) - 1)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { if (!mask || mask & REG_RESERVED) return -EINVAL; diff --git a/arch/um/kernel/um_arch.c b/arch/um/kernel/um_arch.c index 3dbe3acc4833..c3f83e805392 100644 --- a/arch/um/kernel/um_arch.c +++ b/arch/um/kernel/um_arch.c @@ -269,12 +269,88 @@ unsigned long brk_start; #define MIN_VMALLOC (32 * 1024 * 1024) +static u64 __init read_xcr0(void) +{ + u32 a, b, c, d; + + asm volatile("cpuid" + : "=a"(a), "=b"(b), "=c"(c), "=d"(d) + : "a"(0), "c"(0)); + if (a >= 1) { /* max_leaf >= 1 */ + asm volatile("cpuid" + : "=a"(a), "=b"(b), "=c"(c), "=d"(d) + : "a"(1), "c"(0)); + if (c & (1 << 27)) { /* XSAVE enabled by OS */ + asm volatile("xgetbv" : "=d"(d), "=a"(a) : "c"(0)); + return ((u64)d << 32) | a; + } + } + return 0; +} + +static void __init validate_and_set_cpu_cap(int cap, u64 xcr0) +{ + /* + * Check for missing xstate features right away, so that there's no + * perceived need for all optimized code in the kernel to do so. + */ + switch (cap) { + case X86_FEATURE_AVX: + case X86_FEATURE_AVX2: + case X86_FEATURE_AVX_VNNI: + case X86_FEATURE_FMA: + case X86_FEATURE_VAES: + case X86_FEATURE_VPCLMULQDQ: + if ((xcr0 & 0x7) != 0x7) { + static bool warned; + + if (!warned) { + os_warn("Disabling AVX support due to missing xstate features\n"); + warned = true; + } + return; + } + break; + case X86_FEATURE_AVX512F: + case X86_FEATURE_AVX512BW: + case X86_FEATURE_AVX512CD: + case X86_FEATURE_AVX512DQ: + case X86_FEATURE_AVX512ER: + case X86_FEATURE_AVX512IFMA: + case X86_FEATURE_AVX512PF: + case X86_FEATURE_AVX512VBMI: + case X86_FEATURE_AVX512VL: + case X86_FEATURE_AVX512_4FMAPS: + case X86_FEATURE_AVX512_4VNNIW: + case X86_FEATURE_AVX512_BF16: + case X86_FEATURE_AVX512_BITALG: + case X86_FEATURE_AVX512_FP16: + case X86_FEATURE_AVX512_VBMI2: + case X86_FEATURE_AVX512_VNNI: + case X86_FEATURE_AVX512_VP2INTERSECT: + case X86_FEATURE_AVX512_VPOPCNTDQ: + if ((xcr0 & 0xe7) != 0xe7) { + static bool warned; + + if (!warned) { + os_warn("Disabling AVX-512 support due to missing xstate features\n"); + warned = true; + } + return; + } + break; + } + set_cpu_cap(&boot_cpu_data, cap); +} + static void __init parse_host_cpu_flags(char *line) { + u64 xcr0 = read_xcr0(); int i; + for (i = 0; i < 32*NCAPINTS; i++) { if ((x86_cap_flags[i] != NULL) && strstr(line, x86_cap_flags[i])) - set_cpu_cap(&boot_cpu_data, i); + validate_and_set_cpu_cap(i, xcr0); } } diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index a8781db3be70..58dce0b66179 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -295,7 +295,6 @@ config X86 select HAVE_STACK_VALIDATION if HAVE_OBJTOOL select HAVE_STATIC_CALL select HAVE_STATIC_CALL_INLINE if HAVE_OBJTOOL - select HAVE_PREEMPT_DYNAMIC_CALL select HAVE_RSEQ select HAVE_RUST if X86_64 select HAVE_SYSCALL_TRACEPOINTS @@ -3073,15 +3072,6 @@ config GEOS help This option enables system support for the Traverse Technologies GEOS. -config TS5500 - bool "Technologic Systems TS-5500 platform support" - depends on MELAN - select CHECK_SIGNATURE - select NEW_LEDS - select LEDS_CLASS - help - This option enables system support for the Technologic Systems TS-5500. - endif # X86_32 config AMD_NB diff --git a/arch/x86/boot/compressed/error.c b/arch/x86/boot/compressed/error.c index 19a8251de506..ce5ed7d8265e 100644 --- a/arch/x86/boot/compressed/error.c +++ b/arch/x86/boot/compressed/error.c @@ -22,22 +22,3 @@ void error(char *m) while (1) asm("hlt"); } - -/* EFI libstub provides vsnprintf() */ -#ifdef CONFIG_EFI_STUB -void panic(const char *fmt, ...) -{ - static char buf[1024]; - va_list args; - int len; - - va_start(args, fmt); - len = vsnprintf(buf, sizeof(buf), fmt, args); - va_end(args); - - if (len && buf[len - 1] == '\n') - buf[len - 1] = '\0'; - - error(buf); -} -#endif diff --git a/arch/x86/boot/compressed/error.h b/arch/x86/boot/compressed/error.h index 31f9e080d61a..87062dea9a20 100644 --- a/arch/x86/boot/compressed/error.h +++ b/arch/x86/boot/compressed/error.h @@ -6,6 +6,5 @@ void warn(const char *m); void error(char *m) __noreturn; -void panic(const char *fmt, ...) __noreturn __cold; #endif /* BOOT_COMPRESSED_ERROR_H */ diff --git a/arch/x86/boot/compressed/mem.c b/arch/x86/boot/compressed/mem.c index 0e9f84ab4bdc..1721af3a8039 100644 --- a/arch/x86/boot/compressed/mem.c +++ b/arch/x86/boot/compressed/mem.c @@ -2,48 +2,6 @@ #include "error.h" #include "misc.h" -#include "tdx.h" -#include "sev.h" -#include <asm/shared/tdx.h> - -/* - * accept_memory() and process_unaccepted_memory() called from EFI stub which - * runs before decompressor and its early_tdx_detect(). - * - * Enumerate TDX directly from the early users. - */ -static bool early_is_tdx_guest(void) -{ - static bool once; - static bool is_tdx; - - if (!IS_ENABLED(CONFIG_INTEL_TDX_GUEST)) - return false; - - if (!once) { - u32 eax, sig[3]; - - cpuid_count(TDX_CPUID_LEAF_ID, 0, &eax, - &sig[0], &sig[2], &sig[1]); - is_tdx = !memcmp(TDX_IDENT, sig, sizeof(sig)); - once = true; - } - - return is_tdx; -} - -void arch_accept_memory(phys_addr_t start, phys_addr_t end) -{ - /* Platform-specific memory-acceptance call goes here */ - if (early_is_tdx_guest()) { - if (!tdx_accept_memory(start, end)) - panic("TDX: Failed to accept memory\n"); - } else if (early_is_sevsnp_guest()) { - snp_accept_memory(start, end); - } else { - error("Cannot accept memory: unknown platform\n"); - } -} bool init_unaccepted_memory(void) { diff --git a/arch/x86/boot/compressed/sev.h b/arch/x86/boot/compressed/sev.h index 22637b416b46..62e50c2e71ed 100644 --- a/arch/x86/boot/compressed/sev.h +++ b/arch/x86/boot/compressed/sev.h @@ -14,7 +14,6 @@ void snp_accept_memory(phys_addr_t start, phys_addr_t end); u64 sev_get_status(void); -bool early_is_sevsnp_guest(void); static inline u64 sev_es_rd_ghcb_msr(void) { @@ -37,7 +36,6 @@ static inline void sev_es_wr_ghcb_msr(u64 val) static inline void snp_accept_memory(phys_addr_t start, phys_addr_t end) { } static inline u64 sev_get_status(void) { return 0; } -static inline bool early_is_sevsnp_guest(void) { return false; } #endif diff --git a/arch/x86/boot/compressed/tdx-shared.c b/arch/x86/boot/compressed/tdx-shared.c index 5ac43762fe13..dc38047647cc 100644 --- a/arch/x86/boot/compressed/tdx-shared.c +++ b/arch/x86/boot/compressed/tdx-shared.c @@ -1,2 +1,4 @@ +#define __NO_FORTIFY + #include "error.h" #include "../../coco/tdx/tdx-shared.c" diff --git a/arch/x86/boot/early_serial_console.c b/arch/x86/boot/early_serial_console.c index 5b83beab89e1..39fcd551fc81 100644 --- a/arch/x86/boot/early_serial_console.c +++ b/arch/x86/boot/early_serial_console.c @@ -22,6 +22,7 @@ #define DLH 1 /* Divisor latch High */ #define DEFAULT_BAUD 9600 +#define BASE_BAUD (1843200 / 16) static void early_serial_init(int port, int baud) { @@ -33,7 +34,7 @@ static void early_serial_init(int port, int baud) outb(0, port + FCR); /* no fifo */ outb(0x3, port + MCR); /* DTR + RTS */ - divisor = 115200 / baud; + divisor = BASE_BAUD / baud; c = inb(port + LCR); outb(c | DLAB, port + LCR); outb(divisor & 0xff, port + DLL); @@ -74,16 +75,13 @@ static void parse_earlyprintk(void) else pos = e - arg; } else if (!strncmp(arg + pos, "ttyS", 4)) { - static const int bases[] = { 0x3f8, 0x2f8 }; - int idx = 0; - /* += strlen("ttyS"); */ pos += 4; if (arg[pos++] == '1') - idx = 1; - - port = bases[idx]; + port = 0x2f8; /* ttyS1 */ + else + port = DEFAULT_SERIAL_PORT; } if (arg[pos] == ',') @@ -98,7 +96,6 @@ static void parse_earlyprintk(void) early_serial_init(port, baud); } -#define BASE_BAUD (1843200/16) static unsigned int probe_baud(int port) { unsigned char lcr, dll, dlh; diff --git a/arch/x86/boot/string.c b/arch/x86/boot/string.c index 1632d40e1f54..be454a686422 100644 --- a/arch/x86/boot/string.c +++ b/arch/x86/boot/string.c @@ -15,6 +15,7 @@ #include <linux/errno.h> #include <linux/limits.h> #include <asm/asm.h> +#include <asm/shared/string.h> #include "ctype.h" #include "string.h" @@ -31,17 +32,7 @@ int memcmp(const void *s1, const void *s2, size_t len) { - bool diff; - - /* - * Make sure ZF is properly set in the len==0 case because in it, - * RCX==0 and the REPE; CMPSB won't get executed. - */ - asm volatile("test %3, %3\n\t" - "repe cmpsb" - : "=@ccnz" (diff), "+D" (s1), "+S" (s2), "+c" (len) - : : "cc", "memory"); - return diff; + return __inline_memcmp(s1, s2, len); } /* diff --git a/arch/x86/coco/sev/core.c b/arch/x86/coco/sev/core.c index cc292d7c6fd1..eb2e853e950a 100644 --- a/arch/x86/coco/sev/core.c +++ b/arch/x86/coco/sev/core.c @@ -1431,15 +1431,22 @@ static ssize_t vmpl_show(struct kobject *kobj, return sysfs_emit(buf, "%d\n", snp_vmpl); } +static ssize_t sev_status_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + return sysfs_emit(buf, "0x%llx\n", sev_status); +} + static struct kobj_attribute vmpl_attr = __ATTR_RO(vmpl); +static struct kobj_attribute sev_status_attr = __ATTR_RO(sev_status); -static struct attribute *vmpl_attrs[] = { - &vmpl_attr.attr, +static struct attribute *sev_status_attrs[] = { + &sev_status_attr.attr, NULL }; static struct attribute_group sev_attr_group = { - .attrs = vmpl_attrs, + .attrs = sev_status_attrs, }; static int __init sev_sysfs_init(void) @@ -1448,7 +1455,7 @@ static int __init sev_sysfs_init(void) struct device *dev_root; int ret; - if (!cc_platform_has(CC_ATTR_GUEST_SEV_SNP)) + if (!(sev_status & MSR_AMD64_SEV_ENABLED)) return -ENODEV; dev_root = bus_get_dev_root(&cpu_subsys); @@ -1463,7 +1470,20 @@ static int __init sev_sysfs_init(void) ret = sysfs_create_group(sev_kobj, &sev_attr_group); if (ret) - kobject_put(sev_kobj); + goto drop_kobj; + + if (sev_status & MSR_AMD64_SEV_SNP_ENABLED) { + ret = sysfs_add_file_to_group(sev_kobj, &vmpl_attr.attr, NULL); + if (ret) + goto drop_sysfs; + } + + return 0; + +drop_sysfs: + sysfs_remove_group(sev_kobj, &sev_attr_group); +drop_kobj: + kobject_put(sev_kobj); return ret; } diff --git a/arch/x86/coco/tdx/tdx-shared.c b/arch/x86/coco/tdx/tdx-shared.c index 1655aa56a0a5..29661d8dfe8e 100644 --- a/arch/x86/coco/tdx/tdx-shared.c +++ b/arch/x86/coco/tdx/tdx-shared.c @@ -89,3 +89,34 @@ noinstr u64 __tdx_hypercall(struct tdx_module_args *args) /* TDVMCALL leaf return code is in R10 */ return args->r10; } + +void __noreturn tdx_panic(const char *msg) +{ + struct tdx_module_args args = { + .r10 = TDX_HYPERCALL_STANDARD, + .r11 = TDVMCALL_REPORT_FATAL_ERROR, + .r12 = 0, /* Error code: 0 is Panic */ + }; + /* Define register order according to the GHCI */ + struct { u64 r14, r15, rbx, rdi, rsi, r8, r9, rdx; } message = {}; + + /* VMM assumes '\0' in byte 65, if the message took all 64 bytes */ + memcpy(&message, msg, strnlen(msg, sizeof(message))); + + args.r8 = message.r8; + args.r9 = message.r9; + args.r14 = message.r14; + args.r15 = message.r15; + args.rdi = message.rdi; + args.rsi = message.rsi; + args.rbx = message.rbx; + args.rdx = message.rdx; + + /* + * This hypercall should never return and it is not safe + * to keep the guest running. Call it forever if it + * happens to return. + */ + while (1) + __tdx_hypercall(&args); +} diff --git a/arch/x86/coco/tdx/tdx.c b/arch/x86/coco/tdx/tdx.c index f904a636d449..ad0131813a30 100644 --- a/arch/x86/coco/tdx/tdx.c +++ b/arch/x86/coco/tdx/tdx.c @@ -139,7 +139,7 @@ int tdx_mcall_get_report0(u8 *reportdata, u8 *tdreport) return 0; } -EXPORT_SYMBOL_GPL(tdx_mcall_get_report0); +EXPORT_SYMBOL_FOR_MODULES(tdx_mcall_get_report0, "tdx-guest"); /** * tdx_mcall_extend_rtmr() - Wrapper to extend RTMR registers using @@ -175,7 +175,7 @@ int tdx_mcall_extend_rtmr(u8 index, u8 *data) return 0; } -EXPORT_SYMBOL_GPL(tdx_mcall_extend_rtmr); +EXPORT_SYMBOL_FOR_MODULES(tdx_mcall_extend_rtmr, "tdx-guest"); /** * tdx_hcall_get_quote() - Wrapper to request TD Quote using GetQuote @@ -196,42 +196,7 @@ u64 tdx_hcall_get_quote(u8 *buf, size_t size) /* Since buf is a shared memory, set the shared (decrypted) bits */ return _tdx_hypercall(TDVMCALL_GET_QUOTE, cc_mkdec(virt_to_phys(buf)), size, 0, 0); } -EXPORT_SYMBOL_GPL(tdx_hcall_get_quote); - -static void __noreturn tdx_panic(const char *msg) -{ - struct tdx_module_args args = { - .r10 = TDX_HYPERCALL_STANDARD, - .r11 = TDVMCALL_REPORT_FATAL_ERROR, - .r12 = 0, /* Error code: 0 is Panic */ - }; - union { - /* Define register order according to the GHCI */ - struct { u64 r14, r15, rbx, rdi, rsi, r8, r9, rdx; }; - - char bytes[64] __nonstring; - } message; - - /* VMM assumes '\0' in byte 65, if the message took all 64 bytes */ - strtomem_pad(message.bytes, msg, '\0'); - - args.r8 = message.r8; - args.r9 = message.r9; - args.r14 = message.r14; - args.r15 = message.r15; - args.rdi = message.rdi; - args.rsi = message.rsi; - args.rbx = message.rbx; - args.rdx = message.rdx; - - /* - * This hypercall should never return and it is not safe - * to keep the guest running. Call it forever if it - * happens to return. - */ - while (1) - __tdx_hypercall(&args); -} +EXPORT_SYMBOL_FOR_MODULES(tdx_hcall_get_quote, "tdx-guest"); /* * The kernel cannot handle #VEs when accessing normal kernel memory. Ensure diff --git a/arch/x86/crypto/aegis128-aesni-glue.c b/arch/x86/crypto/aegis128-aesni-glue.c index f1adfba1a76e..09fc0b15b0e9 100644 --- a/arch/x86/crypto/aegis128-aesni-glue.c +++ b/arch/x86/crypto/aegis128-aesni-glue.c @@ -265,8 +265,7 @@ static struct aead_alg crypto_aegis128_aesni_alg = { static int __init crypto_aegis128_aesni_module_init(void) { if (!boot_cpu_has(X86_FEATURE_XMM4_1) || - !boot_cpu_has(X86_FEATURE_AES) || - !cpu_has_xfeatures(XFEATURE_MASK_SSE, NULL)) + !boot_cpu_has(X86_FEATURE_AES)) return -ENODEV; return crypto_register_aead(&crypto_aegis128_aesni_alg); diff --git a/arch/x86/crypto/aesni-intel_glue.c b/arch/x86/crypto/aesni-intel_glue.c index 1903ed8bbe3b..5135bd4e6ea6 100644 --- a/arch/x86/crypto/aesni-intel_glue.c +++ b/arch/x86/crypto/aesni-intel_glue.c @@ -1548,8 +1548,7 @@ static int __init register_avx_algs(void) if (!boot_cpu_has(X86_FEATURE_AVX2) || !boot_cpu_has(X86_FEATURE_VAES) || !boot_cpu_has(X86_FEATURE_VPCLMULQDQ) || - !boot_cpu_has(X86_FEATURE_PCLMULQDQ) || - !cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) + !boot_cpu_has(X86_FEATURE_PCLMULQDQ)) return 0; err = crypto_register_skciphers(skcipher_algs_vaes_avx2, ARRAY_SIZE(skcipher_algs_vaes_avx2)); @@ -1562,9 +1561,7 @@ static int __init register_avx_algs(void) if (!boot_cpu_has(X86_FEATURE_AVX512BW) || !boot_cpu_has(X86_FEATURE_AVX512VL) || - !boot_cpu_has(X86_FEATURE_BMI2) || - !cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM | - XFEATURE_MASK_AVX512, NULL)) + !boot_cpu_has(X86_FEATURE_BMI2)) return 0; if (boot_cpu_has(X86_FEATURE_PREFER_YMM)) { diff --git a/arch/x86/crypto/aria_aesni_avx2_glue.c b/arch/x86/crypto/aria_aesni_avx2_glue.c index 1487a49bfbac..371be2fb6469 100644 --- a/arch/x86/crypto/aria_aesni_avx2_glue.c +++ b/arch/x86/crypto/aria_aesni_avx2_glue.c @@ -195,22 +195,13 @@ static struct skcipher_alg aria_algs[] = { static int __init aria_avx2_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || !boot_cpu_has(X86_FEATURE_AVX2) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX2 or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - if (boot_cpu_has(X86_FEATURE_GFNI)) { aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way; aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way; diff --git a/arch/x86/crypto/aria_aesni_avx_glue.c b/arch/x86/crypto/aria_aesni_avx_glue.c index e4e3d78915a5..d23fc91c0ebd 100644 --- a/arch/x86/crypto/aria_aesni_avx_glue.c +++ b/arch/x86/crypto/aria_aesni_avx_glue.c @@ -182,21 +182,12 @@ static struct skcipher_alg aria_algs[] = { static int __init aria_avx_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - if (boot_cpu_has(X86_FEATURE_GFNI)) { aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way; aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way; diff --git a/arch/x86/crypto/aria_gfni_avx512_glue.c b/arch/x86/crypto/aria_gfni_avx512_glue.c index 363cbf4399cc..e05bbeb22d4a 100644 --- a/arch/x86/crypto/aria_gfni_avx512_glue.c +++ b/arch/x86/crypto/aria_gfni_avx512_glue.c @@ -196,24 +196,15 @@ static struct skcipher_alg aria_algs[] = { static int __init aria_avx512_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || !boot_cpu_has(X86_FEATURE_AVX2) || !boot_cpu_has(X86_FEATURE_AVX512F) || !boot_cpu_has(X86_FEATURE_AVX512VL) || - !boot_cpu_has(X86_FEATURE_GFNI) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_GFNI)) { pr_info("AVX512/GFNI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM | - XFEATURE_MASK_AVX512, &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way; aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way; aria_ops.aria_ctr_crypt_16way = aria_aesni_avx_gfni_ctr_crypt_16way; diff --git a/arch/x86/crypto/camellia_aesni_avx2_glue.c b/arch/x86/crypto/camellia_aesni_avx2_glue.c index 2d2f4e16537c..073fa3bb8388 100644 --- a/arch/x86/crypto/camellia_aesni_avx2_glue.c +++ b/arch/x86/crypto/camellia_aesni_avx2_glue.c @@ -97,22 +97,13 @@ static struct skcipher_alg camellia_algs[] = { static int __init camellia_aesni_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || !boot_cpu_has(X86_FEATURE_AVX2) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX2 or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - return crypto_register_skciphers(camellia_algs, ARRAY_SIZE(camellia_algs)); } diff --git a/arch/x86/crypto/camellia_aesni_avx_glue.c b/arch/x86/crypto/camellia_aesni_avx_glue.c index 5c321f255eb7..872e5e07220f 100644 --- a/arch/x86/crypto/camellia_aesni_avx_glue.c +++ b/arch/x86/crypto/camellia_aesni_avx_glue.c @@ -98,21 +98,12 @@ static struct skcipher_alg camellia_algs[] = { static int __init camellia_aesni_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - return crypto_register_skciphers(camellia_algs, ARRAY_SIZE(camellia_algs)); } diff --git a/arch/x86/crypto/cast5_avx_glue.c b/arch/x86/crypto/cast5_avx_glue.c index 3aca04d43b34..5de35e863370 100644 --- a/arch/x86/crypto/cast5_avx_glue.c +++ b/arch/x86/crypto/cast5_avx_glue.c @@ -92,11 +92,8 @@ static struct skcipher_alg cast5_algs[] = { static int __init cast5_init(void) { - const char *feature_name; - - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); + if (!boot_cpu_has(X86_FEATURE_AVX)) { + pr_info("AVX instructions are not detected.\n"); return -ENODEV; } diff --git a/arch/x86/crypto/cast6_avx_glue.c b/arch/x86/crypto/cast6_avx_glue.c index c4dd28c30303..3d7ea48007bc 100644 --- a/arch/x86/crypto/cast6_avx_glue.c +++ b/arch/x86/crypto/cast6_avx_glue.c @@ -92,11 +92,8 @@ static struct skcipher_alg cast6_algs[] = { static int __init cast6_init(void) { - const char *feature_name; - - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); + if (!boot_cpu_has(X86_FEATURE_AVX)) { + pr_info("AVX instructions are not detected.\n"); return -ENODEV; } diff --git a/arch/x86/crypto/serpent_avx2_glue.c b/arch/x86/crypto/serpent_avx2_glue.c index f5f2121b7956..72a9e2b306d6 100644 --- a/arch/x86/crypto/serpent_avx2_glue.c +++ b/arch/x86/crypto/serpent_avx2_glue.c @@ -93,17 +93,10 @@ static struct skcipher_alg serpent_algs[] = { static int __init serpent_avx2_init(void) { - const char *feature_name; - - if (!boot_cpu_has(X86_FEATURE_AVX2) || !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + if (!boot_cpu_has(X86_FEATURE_AVX2)) { pr_info("AVX2 instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } return crypto_register_skciphers(serpent_algs, ARRAY_SIZE(serpent_algs)); diff --git a/arch/x86/crypto/serpent_avx_glue.c b/arch/x86/crypto/serpent_avx_glue.c index 9c8b3a335d5c..42c4e1569674 100644 --- a/arch/x86/crypto/serpent_avx_glue.c +++ b/arch/x86/crypto/serpent_avx_glue.c @@ -100,11 +100,8 @@ static struct skcipher_alg serpent_algs[] = { static int __init serpent_init(void) { - const char *feature_name; - - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); + if (!boot_cpu_has(X86_FEATURE_AVX)) { + pr_info("AVX instructions are not detected.\n"); return -ENODEV; } diff --git a/arch/x86/crypto/sm4_aesni_avx2_glue.c b/arch/x86/crypto/sm4_aesni_avx2_glue.c index fec0ab7a63dd..eef73894e777 100644 --- a/arch/x86/crypto/sm4_aesni_avx2_glue.c +++ b/arch/x86/crypto/sm4_aesni_avx2_glue.c @@ -98,22 +98,13 @@ static struct skcipher_alg sm4_aesni_avx2_skciphers[] = { static int __init sm4_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || !boot_cpu_has(X86_FEATURE_AVX2) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX2 or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - return crypto_register_skciphers(sm4_aesni_avx2_skciphers, ARRAY_SIZE(sm4_aesni_avx2_skciphers)); } diff --git a/arch/x86/crypto/sm4_aesni_avx_glue.c b/arch/x86/crypto/sm4_aesni_avx_glue.c index 88caf418a06f..ed383da5ff46 100644 --- a/arch/x86/crypto/sm4_aesni_avx_glue.c +++ b/arch/x86/crypto/sm4_aesni_avx_glue.c @@ -314,21 +314,12 @@ static struct skcipher_alg sm4_aesni_avx_skciphers[] = { static int __init sm4_init(void) { - const char *feature_name; - if (!boot_cpu_has(X86_FEATURE_AVX) || - !boot_cpu_has(X86_FEATURE_AES) || - !boot_cpu_has(X86_FEATURE_OSXSAVE)) { + !boot_cpu_has(X86_FEATURE_AES)) { pr_info("AVX or AES-NI instructions are not detected.\n"); return -ENODEV; } - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); - return -ENODEV; - } - return crypto_register_skciphers(sm4_aesni_avx_skciphers, ARRAY_SIZE(sm4_aesni_avx_skciphers)); } diff --git a/arch/x86/crypto/twofish_avx_glue.c b/arch/x86/crypto/twofish_avx_glue.c index 9e20db013750..985bc54a2340 100644 --- a/arch/x86/crypto/twofish_avx_glue.c +++ b/arch/x86/crypto/twofish_avx_glue.c @@ -102,10 +102,8 @@ static struct skcipher_alg twofish_algs[] = { static int __init twofish_init(void) { - const char *feature_name; - - if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, &feature_name)) { - pr_info("CPU feature '%s' is not supported.\n", feature_name); + if (!boot_cpu_has(X86_FEATURE_AVX)) { + pr_info("AVX instructions are not detected.\n"); return -ENODEV; } diff --git a/arch/x86/events/amd/uncore.c b/arch/x86/events/amd/uncore.c index 7181973b5b12..7aa5a5b652a2 100644 --- a/arch/x86/events/amd/uncore.c +++ b/arch/x86/events/amd/uncore.c @@ -39,11 +39,11 @@ static int pmu_version; struct amd_uncore_ctx { int refcnt; int cpu; - struct perf_event **events; unsigned long active_mask[BITS_TO_LONGS(NUM_COUNTERS_MAX)]; int nr_active; struct hrtimer hrtimer; u64 hrtimer_duration; + struct perf_event *events[]; }; struct amd_uncore_pmu { @@ -206,17 +206,15 @@ static int amd_uncore_add(struct perf_event *event, int flags) struct amd_uncore_ctx *ctx = *per_cpu_ptr(pmu->ctx, event->cpu); struct hw_perf_event *hwc = &event->hw; - /* are we already assigned? */ + /* + * Perf serializes ->add() and ->del() for an event. A successful + * ->add() records the claimed slot in hwc->idx before returning, and + * ->del() clears that slot before resetting hwc->idx. Therefore, an + * existing assignment must be at hwc->idx. + */ if (hwc->idx != -1 && ctx->events[hwc->idx] == event) goto out; - for (i = 0; i < pmu->num_counters; i++) { - if (ctx->events[i] == event) { - hwc->idx = i; - goto out; - } - } - /* if not, take the first available counter */ hwc->idx = -1; for (i = 0; i < pmu->num_counters; i++) { @@ -248,19 +246,15 @@ out: static void amd_uncore_del(struct perf_event *event, int flags) { - int i; struct amd_uncore_pmu *pmu = event_to_amd_uncore_pmu(event); struct amd_uncore_ctx *ctx = *per_cpu_ptr(pmu->ctx, event->cpu); struct hw_perf_event *hwc = &event->hw; + struct perf_event *old = event; event->pmu->stop(event, PERF_EF_UPDATE); - for (i = 0; i < pmu->num_counters; i++) { - struct perf_event *tmp = event; - - if (try_cmpxchg(&ctx->events[i], &tmp, NULL)) - break; - } + /* ->del() follows a successful ->add(), so hwc->idx owns this slot. */ + WARN_ON_ONCE(!try_cmpxchg(&ctx->events[hwc->idx], &old, NULL)); hwc->idx = -1; } @@ -519,10 +513,8 @@ static void amd_uncore_ctx_free(struct amd_uncore *uncore, unsigned int cpu) if (cpu == ctx->cpu) cpumask_clear_cpu(cpu, &pmu->active_mask); - if (!--ctx->refcnt) { - kfree(ctx->events); + if (!--ctx->refcnt) kfree(ctx); - } *per_cpu_ptr(pmu->ctx, cpu) = NULL; } @@ -567,18 +559,11 @@ static int amd_uncore_ctx_init(struct amd_uncore *uncore, unsigned int cpu) /* Allocate context if sibling does not exist */ if (!curr) { node = cpu_to_node(cpu); - curr = kzalloc_node(sizeof(*curr), GFP_KERNEL, node); + curr = kzalloc_node(struct_size(curr, events, pmu->num_counters), GFP_KERNEL, node); if (!curr) goto fail; curr->cpu = cpu; - curr->events = kzalloc_node(sizeof(*curr->events) * - pmu->num_counters, - GFP_KERNEL, node); - if (!curr->events) { - kfree(curr); - goto fail; - } amd_uncore_init_hrtimer(curr); curr->hrtimer_duration = (u64)update_interval * NSEC_PER_MSEC; diff --git a/arch/x86/events/core.c b/arch/x86/events/core.c index 8b3ea0adb965..9b1a9032e36d 100644 --- a/arch/x86/events/core.c +++ b/arch/x86/events/core.c @@ -93,7 +93,7 @@ DEFINE_STATIC_CALL_NULL(x86_pmu_stop_scheduling, *x86_pmu.stop_scheduling); DEFINE_STATIC_CALL_NULL(x86_pmu_sched_task, *x86_pmu.sched_task); -DEFINE_STATIC_CALL_NULL(x86_pmu_drain_pebs, *x86_pmu.drain_pebs); +DEFINE_STATIC_CALL_RET0(x86_pmu_drain_pebs, *x86_pmu.drain_pebs); DEFINE_STATIC_CALL_NULL(x86_pmu_pebs_aliases, *x86_pmu.pebs_aliases); DEFINE_STATIC_CALL_NULL(x86_pmu_filter, *x86_pmu.filter); @@ -408,6 +408,53 @@ set_ext_hw_attr(struct hw_perf_event *hwc, struct perf_event *event) return x86_pmu_extra_regs(val, event); } +static DEFINE_PER_CPU(struct xregs_state *, ext_regs_buf); + +static void release_ext_regs_buffers(void) +{ + int cpu; + + if (!x86_pmu.ext_regs_mask) + return; + + for_each_possible_cpu(cpu) { + kfree(per_cpu(ext_regs_buf, cpu)); + per_cpu(ext_regs_buf, cpu) = NULL; + } +} + +static void reserve_ext_regs_buffers(void) +{ + bool compacted = cpu_feature_enabled(X86_FEATURE_XCOMPACTED); + unsigned int size; + int cpu; + + if (!x86_pmu.ext_regs_mask) + return; + + /* Add 64 bytes to satisfy the XSAVE area's 64-byte alignment. */ + size = xstate_calculate_size(x86_pmu.ext_regs_mask, compacted) + 64; + + for_each_possible_cpu(cpu) { + per_cpu(ext_regs_buf, cpu) = kzalloc_node(size, GFP_KERNEL, + cpu_to_node(cpu)); + if (WARN_ON_ONCE(!per_cpu(ext_regs_buf, cpu))) + goto err; + } + + return; + +err: + release_ext_regs_buffers(); +} + +static inline struct xregs_state *get_ext_regs_buf(int cpu) +{ + void *buf = per_cpu(ext_regs_buf, cpu); + + return buf ? PTR_ALIGN(buf, 64) : NULL; +} + int x86_reserve_hardware(void) { int err = 0; @@ -420,6 +467,7 @@ int x86_reserve_hardware(void) } else { reserve_ds_buffers(); reserve_lbr_buffers(); + reserve_ext_regs_buffers(); } } if (!err) @@ -436,6 +484,7 @@ void x86_release_hardware(void) release_pmc_hardware(); release_ds_buffers(); release_lbr_buffers(); + release_ext_regs_buffers(); mutex_unlock(&pmc_reserve_mutex); } } @@ -583,6 +632,79 @@ int x86_pmu_max_precise(struct pmu *pmu) return precise; } +static int pebs_simd_regs_validate(struct perf_event *event) +{ + u64 caps = hybrid(event->pmu, arch_pebs_cap).caps; + + if (event_needs_xmm(event) && + !x86_pmu.arch_pebs && !x86_pmu.intel_cap.pebs_baseline) + return -EINVAL; + if (event_needs_xmm(event) && + x86_pmu.arch_pebs && !(caps & ARCH_PEBS_VECR_XMM)) + return -EINVAL; + + if (event_needs_ssp(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_GPR))) + return -EINVAL; + if (event_needs_ymm(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_YMMH))) + return -EINVAL; + if (event_needs_egprs(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_EGPRS))) + return -EINVAL; + if (event_needs_opmask(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_OPMASK))) + return -EINVAL; + if (event_needs_low16_zmm(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_ZMMH))) + return -EINVAL; + if (event_needs_high16_zmm(event) && + !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_H16ZMM))) + return -EINVAL; + + return 0; +} + +static int event_simd_regs_validate(struct perf_event *event) +{ + u64 reserved = ~GENMASK_ULL(PERF_REG_MISC_MAX - 1, 0); + + if (!get_ext_regs_buf(raw_smp_processor_id())) + return -ENOMEM; + /* + * The XMM space in the perf_event_x86_regs is reclaimed + * for eGPRs and other general registers. + */ + if (((event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_regs_intr & reserved)) || + ((event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_regs_user & reserved))) + return -EINVAL; + if (event_needs_egprs(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_APX)) + return -EINVAL; + if (event_needs_ssp(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_CET_USER)) + return -EINVAL; + if (event_needs_xmm(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_SSE)) + return -EINVAL; + if (event_needs_ymm(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_YMM)) + return -EINVAL; + if (event_needs_low16_zmm(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_ZMM_Hi256)) + return -EINVAL; + if (event_needs_high16_zmm(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_Hi16_ZMM)) + return -EINVAL; + if (event_needs_opmask(event) && + !(x86_pmu.ext_regs_mask & XFEATURE_MASK_OPMASK)) + return -EINVAL; + + return 0; +} + int x86_pmu_hw_config(struct perf_event *event) { if (event->attr.precise_ip) { @@ -653,19 +775,38 @@ int x86_pmu_hw_config(struct perf_event *event) return -EINVAL; } - /* sample_regs_user never support XMM registers */ - if (unlikely(event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK)) - return -EINVAL; - /* - * Besides the general purpose registers, XMM registers may - * be collected in PEBS on some platforms, e.g. Icelake - */ - if (unlikely(event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK)) { - if (!(event->pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS)) - return -EINVAL; + if (event->attr.sample_type & (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)) { + int ret; - if (!event->attr.precise_ip) - return -EINVAL; + if (event->attr.sample_simd_regs_enabled) { + if (!(event->pmu->capabilities & PERF_PMU_CAP_SIMD_REGS)) + return -EINVAL; + + if (event->attr.precise_ip) { + ret = pebs_simd_regs_validate(event); + if (ret) + return ret; + } + ret = event_simd_regs_validate(event); + if (ret) + return ret; + } else if (event_has_extended_regs(event)) { + if (!(event->pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS)) + return -EINVAL; + + if (event->attr.precise_ip) { + u64 caps = hybrid(event->pmu, arch_pebs_cap).caps; + + if (x86_pmu.arch_pebs && !(caps & ARCH_PEBS_VECR_XMM)) + return -EINVAL; + if (!x86_pmu.arch_pebs && !x86_pmu.intel_cap.pebs_baseline) + return -EINVAL; + } + if (!get_ext_regs_buf(raw_smp_processor_id())) + return -ENOMEM; + if (!(x86_pmu.ext_regs_mask & XFEATURE_MASK_SSE)) + return -EINVAL; + } } return x86_setup_perfctr(event); @@ -721,9 +862,10 @@ void x86_pmu_disable_all(void) } } -struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data) +struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, + struct x86_guest_pebs *guest_pebs) { - return static_call(x86_pmu_guest_get_msrs)(nr, data); + return static_call(x86_pmu_guest_get_msrs)(nr, guest_pebs); } EXPORT_SYMBOL_FOR_KVM(perf_guest_get_msrs); @@ -1717,6 +1859,302 @@ do_del: static_call_cond(x86_pmu_del)(event); } +void x86_pmu_clear_perf_regs(struct pt_regs *regs) +{ + struct x86_perf_regs *perf_regs = container_of(regs, struct x86_perf_regs, regs); + + perf_regs->abi = PERF_SAMPLE_REGS_ABI_NONE; + perf_regs->xmm_regs = NULL; + perf_regs->ymmh_regs = NULL; + perf_regs->zmmh_regs = NULL; + perf_regs->h16zmm_regs = NULL; + perf_regs->opmask_regs = NULL; + perf_regs->egpr_regs = NULL; + perf_regs->ssp = NULL; +} + +static void update_perf_regs(struct x86_perf_regs *perf_regs, + struct xregs_state *xsave, u64 bitmap) +{ + struct cet_user_state *cet; + u64 mask; + + if (!xsave) + return; + + /* Restrict to features actually saved by XSAVES */ + mask = bitmap & xsave->header.xfeatures; + + if (mask & XFEATURE_MASK_SSE) + perf_regs->xmm_space = xsave->i387.xmm_space; + if (mask & XFEATURE_MASK_YMM) + perf_regs->ymmh = get_xsave_addr(xsave, XFEATURE_YMM); + if (mask & XFEATURE_MASK_ZMM_Hi256) + perf_regs->zmmh = get_xsave_addr(xsave, XFEATURE_ZMM_Hi256); + if (mask & XFEATURE_MASK_Hi16_ZMM) + perf_regs->h16zmm = get_xsave_addr(xsave, XFEATURE_Hi16_ZMM); + if (mask & XFEATURE_MASK_OPMASK) + perf_regs->opmask = get_xsave_addr(xsave, XFEATURE_OPMASK); + if (mask & XFEATURE_MASK_APX) + perf_regs->egpr = get_xsave_addr(xsave, XFEATURE_APX); + if (mask & XFEATURE_MASK_CET_USER) { + cet = get_xsave_addr(xsave, XFEATURE_CET_USER); + perf_regs->ssp = cet ? &cet->user_ssp : NULL; + } +} + +/* + * The x86 specific variant of perf_sample_regs_intr(). + * Update data->regs_intr fields for extended registers (e.g., SIMD). + */ +static void x86_pmu_update_regs_intr(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs, + bool exclude_kernel) +{ + struct x86_perf_regs *perf_regs; + + if (exclude_kernel && !user_mode(regs)) { + data->regs_intr.regs = NULL; + data->regs_intr.abi = PERF_SAMPLE_REGS_ABI_NONE; + } else { + data->regs_intr.regs = regs; + data->regs_intr.abi = perf_reg_abi(current); + } + + data->dyn_size += sizeof(u64); + if (data->regs_intr.regs) { + data->dyn_size += hweight64(event->attr.sample_regs_intr) * + sizeof(u64); + if (event_has_simd_regs(event)) { + data->dyn_size += perf_update_xregs_size(event, true); + data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } + + perf_regs = container_of(data->regs_intr.regs, + struct x86_perf_regs, regs); + perf_regs->abi = data->regs_intr.abi; + } + + /* + * Set PERF_SAMPLE_REGS_INTR to bypass perf_sample_regs_intr() call + * in perf_prepare_sample() function. + */ + data->sample_flags |= PERF_SAMPLE_REGS_INTR; +} + +static DEFINE_PER_CPU(struct x86_perf_regs, x86_user_regs); + +static void x86_pmu_get_regs_user(struct perf_sample_data *data, + struct pt_regs *regs) +{ + struct x86_perf_regs *x86_regs_user = this_cpu_ptr(&x86_user_regs); + struct perf_regs regs_user; + + x86_pmu_clear_perf_regs(&x86_regs_user->regs); + + perf_get_regs_user(®s_user, regs); + data->regs_user.abi = regs_user.abi; + if (regs_user.regs) { + x86_regs_user->regs = *regs_user.regs; + data->regs_user.regs = &x86_regs_user->regs; + } else { + data->regs_user.regs = NULL; + } +} + +/* + * The x86 specific variant of perf_sample_regs_user(). + * Update data->regs_user fields for extended registers (e.g., SIMD). + */ +static void x86_pmu_update_regs_user(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs) +{ + struct x86_perf_regs *x86_regs_user = this_cpu_ptr(&x86_user_regs); + struct perf_event_attr *attr = &event->attr; + struct x86_perf_regs *perf_regs; + + /* + * PERF_SAMPLE_REGS_INTR and PERF_SAMPLE_REGS_USER can both be + * requested for one event. Keep user regs in a separate x86_perf_regs + * instance, so intr-reg collection does not overwrite user-reg data. + */ + if (user_mode(regs)) { + x86_pmu_clear_perf_regs(&x86_regs_user->regs); + perf_regs = container_of(regs, struct x86_perf_regs, regs); + /* Copy all sampled regs data to x86_regs_user. */ + *x86_regs_user = *perf_regs; + data->regs_user.regs = &x86_regs_user->regs; + data->regs_user.abi = perf_reg_abi(current); + } else if (is_user_task(current)) { + x86_pmu_get_regs_user(data, regs); + } else { + data->regs_user.abi = PERF_SAMPLE_REGS_ABI_NONE; + data->regs_user.regs = NULL; + } + + data->dyn_size += sizeof(u64); + if (data->regs_user.regs) { + data->dyn_size += hweight64(attr->sample_regs_user) * sizeof(u64); + if (event_has_simd_regs(event)) { + data->dyn_size += perf_update_xregs_size(event, false); + data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } + + x86_regs_user->abi = data->regs_user.abi; + } + + /* + * Set PERF_SAMPLE_REGS_USER to bypass perf_sample_regs_user() call + * in perf_prepare_sample() function. + */ + data->sample_flags |= PERF_SAMPLE_REGS_USER; +} + +/* + * This function retrieves cached user-space fpu registers (XMM/YMM/ZMM). + * If TIF_NEED_FPU_LOAD is set or PMI hits into guest, it indicates that + * the user-space FPU state is cached. Otherwise, the data should be read + * directly from the hardware registers. + */ +static inline u64 x86_pmu_update_user_xregs(struct perf_sample_data *data, + struct pt_regs *regs, + u64 mask, bool from_pebs) +{ + struct x86_perf_regs *perf_regs; + struct xregs_state *xsave; + struct fpu *fpu; + struct fpstate *fps; + u64 user_mask = mask; + + if (!is_user_task(current)) + return 0; + + if (data->regs_user.abi == PERF_SAMPLE_REGS_ABI_NONE) + return 0; + + /* + * If PEBS hits kernel space, need to re-sample extended + * registers for user space. + */ + if (user_mode(regs)) + user_mask = from_pebs ? 0 : mask; + + fpu = x86_task_fpu(current); + fps = READ_ONCE(fpu->__task_fpstate); + /* + * If fpu->__task_fpstate is set, it points to the cached user + * FPU state (e.g. with KVM guest-state swapping). Otherwise, + * when TIF_NEED_FPU_LOAD is set, fpu->fpstate holds the cached + * user state. If neither is true, the user state is live in hardware. + */ + if (user_mask && (test_thread_flag(TIF_NEED_FPU_LOAD) || fps)) { + perf_regs = container_of(data->regs_user.regs, + struct x86_perf_regs, regs); + if (!fps) + fps = fpu->fpstate; + xsave = &fps->regs.xsave; + + update_perf_regs(perf_regs, xsave, user_mask); + return 0; + } + + return user_mask; +} + +static u64 get_simd_sample_mask(struct perf_event *event, u64 sample_type) +{ + u64 mask = 0; + + if (__event_needs_xmm(event, sample_type)) + mask |= XFEATURE_MASK_SSE; + if (__event_needs_ymm(event, sample_type)) + mask |= XFEATURE_MASK_YMM; + if (__event_needs_low16_zmm(event, sample_type)) + mask |= XFEATURE_MASK_ZMM_Hi256; + if (__event_needs_high16_zmm(event, sample_type)) + mask |= XFEATURE_MASK_Hi16_ZMM; + if (__event_needs_opmask(event, sample_type)) + mask |= XFEATURE_MASK_OPMASK; + if (__event_needs_egprs(event, sample_type)) + mask |= XFEATURE_MASK_APX; + if (__event_needs_ssp(event, sample_type)) + mask |= XFEATURE_MASK_CET_USER; + + return mask; +} + +static void x86_pmu_sample_xregs(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs, + bool from_pebs) +{ + struct xregs_state *xsave = get_ext_regs_buf(smp_processor_id()); + u64 sample_type = event->attr.sample_type; + struct x86_perf_regs *perf_regs; + u64 intr_mask = 0; + u64 user_mask = 0; + + if (WARN_ON_ONCE(!xsave) || !in_nmi()) + return; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && data->regs_intr.regs) { + intr_mask |= get_simd_sample_mask(event, PERF_SAMPLE_REGS_INTR); + intr_mask &= x86_pmu.ext_regs_mask; + intr_mask = from_pebs ? 0 : intr_mask; + } + + if ((sample_type & PERF_SAMPLE_REGS_USER) && data->regs_user.regs) { + user_mask |= get_simd_sample_mask(event, PERF_SAMPLE_REGS_USER); + user_mask &= x86_pmu.ext_regs_mask; + user_mask = x86_pmu_update_user_xregs(data, regs, + user_mask, from_pebs); + } + + if (user_mask | intr_mask) { + xsave->header.xfeatures = 0; + xsaves_nmi(xsave, user_mask | intr_mask); + } + + if (intr_mask) { + perf_regs = container_of(data->regs_intr.regs, + struct x86_perf_regs, regs); + update_perf_regs(perf_regs, xsave, intr_mask); + } + + if (user_mask) { + perf_regs = container_of(data->regs_user.regs, + struct x86_perf_regs, regs); + update_perf_regs(perf_regs, xsave, user_mask); + } +} + +void x86_pmu_update_perf_regs(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs, + bool from_pebs) +{ + u64 sample_type = event->attr.sample_type; + + if (!(sample_type & + (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER))) + return; + + if (!event_needs_xmm(event) && + !event_has_simd_regs(event)) + return; + + if (sample_type & PERF_SAMPLE_REGS_INTR) { + x86_pmu_update_regs_intr(event, data, regs, + event->attr.exclude_kernel); + } + if (sample_type & PERF_SAMPLE_REGS_USER) + x86_pmu_update_regs_user(event, data, regs); + + x86_pmu_sample_xregs(event, data, regs, from_pebs); +} + int x86_pmu_handle_irq(struct pt_regs *regs) { struct perf_sample_data data; @@ -1800,9 +2238,11 @@ void perf_put_guest_lvtpc(void) EXPORT_SYMBOL_FOR_KVM(perf_put_guest_lvtpc); #endif /* CONFIG_PERF_GUEST_MEDIATED_PMU */ +static DEFINE_PER_CPU(struct x86_perf_regs, x86_intr_regs); static int perf_event_nmi_handler(unsigned int cmd, struct pt_regs *regs) { + struct x86_perf_regs *x86_regs = this_cpu_ptr(&x86_intr_regs); u64 start_clock; u64 finish_clock; int ret; @@ -1826,7 +2266,8 @@ perf_event_nmi_handler(unsigned int cmd, struct pt_regs *regs) return NMI_DONE; start_clock = sched_clock(); - ret = static_call(x86_pmu_handle_irq)(regs); + x86_regs->regs = *regs; + ret = static_call(x86_pmu_handle_irq)(&x86_regs->regs); finish_clock = sched_clock(); perf_sample_event_took(finish_clock - start_clock); @@ -2218,8 +2659,20 @@ static int __init init_hw_perf_events(void) pmu.attr_update = x86_pmu.attr_update; - if (!is_hybrid()) + if (!is_hybrid()) { x86_pmu_show_pmu_cap(NULL); + } else { + int i; + + /* + * Init default ops. + * Must be called before registering x86_pmu_starting_cpu(), + * otherwise some key PMU fields, e.g., capabilities + * initialized in x86_pmu_starting_cpu(), would be overwritten. + */ + for (i = 0; i < x86_pmu.num_hybrid_pmus; i++) + x86_pmu.hybrid_pmu[i].pmu = pmu; + } if (!x86_pmu.read) x86_pmu.read = _x86_pmu_read; @@ -2266,7 +2719,6 @@ static int __init init_hw_perf_events(void) for (i = 0; i < x86_pmu.num_hybrid_pmus; i++) { hybrid_pmu = &x86_pmu.hybrid_pmu[i]; - hybrid_pmu->pmu = pmu; hybrid_pmu->pmu.type = -1; hybrid_pmu->pmu.attr_update = x86_pmu.attr_update; hybrid_pmu->pmu.capabilities |= PERF_PMU_CAP_EXTENDED_HW_TYPE; diff --git a/arch/x86/events/intel/core.c b/arch/x86/events/intel/core.c index 3ef80882843e..0a34d674df59 100644 --- a/arch/x86/events/intel/core.c +++ b/arch/x86/events/intel/core.c @@ -14,7 +14,6 @@ #include <linux/slab.h> #include <linux/export.h> #include <linux/nmi.h> -#include <linux/kvm_host.h> #include <asm/cpufeature.h> #include <asm/cpuid/api.h> @@ -2797,7 +2796,7 @@ static void __intel_pmu_enable_all(int added, bool pmi) } wrmsrq(MSR_CORE_PERF_GLOBAL_CTRL, - intel_ctrl & ~cpuc->intel_ctrl_guest_mask); + intel_ctrl & ~cpuc->intel_ctrl_exclude_host_mask); if (test_bit(INTEL_PMC_IDX_FIXED_BTS, cpuc->active_mask)) { struct perf_event *event = @@ -2995,9 +2994,9 @@ static inline void intel_set_masks(struct perf_event *event, int idx) struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); if (event->attr.exclude_host) - __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_guest_mask); + __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask); if (event->attr.exclude_guest) - __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_host_mask); + __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_guest_mask); if (event_is_checkpointed(event)) __set_bit(idx, (unsigned long *)&cpuc->intel_cp_status); } @@ -3006,8 +3005,8 @@ static inline void intel_clear_masks(struct perf_event *event, int idx) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); - __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_guest_mask); - __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_host_mask); + __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask); + __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_guest_mask); __clear_bit(idx, (unsigned long *)&cpuc->intel_cp_status); } @@ -3499,6 +3498,21 @@ static void intel_pmu_enable_event_ext(struct perf_event *event) if (pebs_data_cfg & PEBS_DATACFG_XMMS) ext |= ARCH_PEBS_VECR_XMM & cap.caps; + if (pebs_data_cfg & PEBS_DATACFG_YMMHS) + ext |= ARCH_PEBS_VECR_YMMH & cap.caps; + + if (pebs_data_cfg & PEBS_DATACFG_EGPRS) + ext |= ARCH_PEBS_VECR_EGPRS & cap.caps; + + if (pebs_data_cfg & PEBS_DATACFG_OPMASKS) + ext |= ARCH_PEBS_VECR_OPMASK & cap.caps; + + if (pebs_data_cfg & PEBS_DATACFG_ZMMHS) + ext |= ARCH_PEBS_VECR_ZMMH & cap.caps; + + if (pebs_data_cfg & PEBS_DATACFG_H16ZMMS) + ext |= ARCH_PEBS_VECR_H16ZMM & cap.caps; + if (pebs_data_cfg & PEBS_DATACFG_LBRS) ext |= ARCH_PEBS_LBR & cap.caps; @@ -3770,20 +3784,20 @@ static void intel_pmu_reset(void) * * The contents and other behavior of the guest event do not matter. */ -static void x86_pmu_handle_guest_pebs(struct pt_regs *regs, - struct perf_sample_data *data) +static int x86_pmu_handle_guest_pebs(struct pt_regs *regs, + struct perf_sample_data *data) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); - u64 guest_pebs_idxs = cpuc->pebs_enabled & ~cpuc->intel_ctrl_host_mask; + u64 guest_pebs_idxs = cpuc->pebs_enabled & ~cpuc->intel_ctrl_exclude_guest_mask; struct perf_event *event = NULL; int bit; if (!unlikely(perf_guest_state())) - return; + return 0; if (!x86_pmu.pebs_ept || !x86_pmu.pebs_active || !guest_pebs_idxs) - return; + return 0; for_each_set_bit(bit, (unsigned long *)&guest_pebs_idxs, X86_PMC_IDX_MAX) { event = cpuc->events[bit]; @@ -3793,9 +3807,14 @@ static void x86_pmu_handle_guest_pebs(struct pt_regs *regs, perf_sample_data_init(data, 0, event->hw.last_period); perf_event_overflow(event, data, regs); - /* Inject one fake event is enough. */ - break; + /* + * Inject one fake event is enough. + * Returning 1 to inform PMI is handled. + */ + return 1; } + + return 0; } static int handle_pmi_common(struct pt_regs *regs, u64 status) @@ -3844,9 +3863,11 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status) if (__test_and_clear_bit(GLOBAL_STATUS_BUFFER_OVF_BIT, (unsigned long *)&status)) { u64 pebs_enabled = cpuc->pebs_enabled; - handled++; - x86_pmu_handle_guest_pebs(regs, &data); - static_call(x86_pmu_drain_pebs)(regs, &data); + handled += x86_pmu_handle_guest_pebs(regs, &data); + handled += static_call(x86_pmu_drain_pebs)(regs, &data); + /* Ensure no "suspicious NMI" warning for empty PEBS buffer. */ + if (!handled) + handled++; /* * PMI throttle may be triggered, which stops the PEBS event. @@ -3873,8 +3894,10 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status) */ if (__test_and_clear_bit(GLOBAL_STATUS_ARCH_PEBS_THRESHOLD_BIT, (unsigned long *)&status)) { - handled++; - static_call(x86_pmu_drain_pebs)(regs, &data); + handled += static_call(x86_pmu_drain_pebs)(regs, &data); + /* Ensure no "suspicious NMI" warning for empty PEBS buffer. */ + if (!handled) + handled++; if (cpuc->events[INTEL_PMC_IDX_FIXED_SLOTS] && is_pebs_counter_event_group(cpuc->events[INTEL_PMC_IDX_FIXED_SLOTS])) @@ -3950,6 +3973,9 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status) if (has_branch_stack(event)) intel_pmu_lbr_save_brstack(&data, cpuc, event); + x86_pmu_clear_perf_regs(regs); + x86_pmu_update_perf_regs(event, &data, regs, false); + perf_event_overflow(event, &data, regs); } @@ -4717,14 +4743,20 @@ static void intel_pebs_aliases_skl(struct perf_event *event) static unsigned long intel_pmu_large_pebs_flags(struct perf_event *event) { unsigned long flags = x86_pmu.large_pebs_flags; + u64 gprs_mask = event->attr.sample_simd_regs_enabled ? + PEBS_GP_REGS | PERF_X86_EGPRS_MASK | + BIT_ULL(PERF_REG_X86_SSP) : + PEBS_GP_REGS | PERF_REG_EXTENDED_MASK; if (event->attr.use_clockid) flags &= ~PERF_SAMPLE_TIME; if (!event->attr.exclude_kernel) flags &= ~PERF_SAMPLE_REGS_USER; - if (event->attr.sample_regs_user & ~PEBS_GP_REGS) + if ((event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_regs_user & ~gprs_mask)) flags &= ~PERF_SAMPLE_REGS_USER; - if (event->attr.sample_regs_intr & ~PEBS_GP_REGS) + if ((event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_regs_intr & ~gprs_mask)) flags &= ~PERF_SAMPLE_REGS_INTR; return flags; } @@ -5295,13 +5327,13 @@ static int intel_pmu_hw_config(struct perf_event *event) * when it uses {RD,WR}MSR, which should be handled by the KVM context, * specifically in the intel_pmu_{get,set}_msr(). */ -static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) +static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, + struct x86_guest_pebs *guest_pebs) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct perf_guest_switch_msr *arr = cpuc->guest_switch_msrs; - struct kvm_pmu *kvm_pmu = (struct kvm_pmu *)data; u64 intel_ctrl = hybrid(cpuc->pmu, intel_ctrl); - u64 pebs_mask = cpuc->pebs_enabled & x86_pmu.pebs_capable; + u64 pebs_mask = intel_ctrl & cpuc->pebs_enabled & x86_pmu.pebs_capable; u64 guest_pebs_mask; int global_ctrl; @@ -5316,8 +5348,8 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) global_ctrl = (*nr)++; arr[global_ctrl] = (struct perf_guest_switch_msr){ .msr = MSR_CORE_PERF_GLOBAL_CTRL, - .host = intel_ctrl & ~cpuc->intel_ctrl_guest_mask, - .guest = intel_ctrl & ~cpuc->intel_ctrl_host_mask & ~pebs_mask, + .host = intel_ctrl & ~cpuc->intel_ctrl_exclude_host_mask, + .guest = intel_ctrl & ~cpuc->intel_ctrl_exclude_guest_mask & ~pebs_mask, }; if (!x86_pmu.ds_pebs) @@ -5353,17 +5385,9 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) * the guest wants to use for PEBS, (c) are not excluded from counting * in the guest, and (d) _are_ excluded from counting in the host. */ - guest_pebs_mask = pebs_mask & intel_ctrl & kvm_pmu->pebs_enable & - ~cpuc->intel_ctrl_host_mask & - cpuc->intel_ctrl_guest_mask; - - /* - * Disable counters where the guest PMC is different than the host PMC - * being used on behalf of the guest, as the PEBS record includes - * PERF_GLOBAL_STATUS, i.e. the guest will see overflow status for the - * wrong counter(s). - */ - guest_pebs_mask &= ~kvm_pmu->host_cross_mapped_mask; + guest_pebs_mask = pebs_mask & guest_pebs->enable & + ~cpuc->intel_ctrl_exclude_guest_mask & + cpuc->intel_ctrl_exclude_host_mask; /* * FIXME: Allow guest and host usage of PEBS events to co-exist instead @@ -5371,7 +5395,7 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) * What exactly goes wrong if guest and host are using PEBS is * unknown. */ - if (pebs_mask & ~cpuc->intel_ctrl_guest_mask) + if (pebs_mask & ~cpuc->intel_ctrl_exclude_host_mask) guest_pebs_mask = 0; /* @@ -5382,14 +5406,14 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) arr[(*nr)++] = (struct perf_guest_switch_msr){ .msr = MSR_IA32_DS_AREA, .host = (unsigned long)cpuc->ds, - .guest = guest_pebs_mask ? kvm_pmu->ds_area : (unsigned long)cpuc->ds, + .guest = guest_pebs_mask ? guest_pebs->ds_area : (unsigned long)cpuc->ds, }; if (x86_pmu.intel_cap.pebs_baseline) { arr[(*nr)++] = (struct perf_guest_switch_msr){ .msr = MSR_PEBS_DATA_CFG, .host = cpuc->active_pebs_data_cfg, - .guest = guest_pebs_mask ? kvm_pmu->pebs_data_cfg : + .guest = guest_pebs_mask ? guest_pebs->data_cfg : cpuc->active_pebs_data_cfg, }; } @@ -5405,7 +5429,8 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data) return arr; } -static struct perf_guest_switch_msr *core_guest_get_msrs(int *nr, void *data) +static struct perf_guest_switch_msr *core_guest_get_msrs(int *nr, + struct x86_guest_pebs *guest_pebs) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct perf_guest_switch_msr *arr = cpuc->guest_switch_msrs; @@ -6216,12 +6241,46 @@ static inline bool intel_pmu_broken_perf_cap(void) return false; } -static inline void __intel_update_pmu_caps(struct pmu *pmu) +static inline void __intel_update_pmu_xregs_caps(struct pmu *pmu) { struct pmu *dest_pmu = pmu ? pmu : x86_get_pmu(smp_processor_id()); - if (hybrid(pmu, arch_pebs_cap).caps & ARCH_PEBS_VECR_XMM) - dest_pmu->capabilities |= PERF_PMU_CAP_EXTENDED_REGS; + /* Only support the extension when XSAVES is available. */ + if (!boot_cpu_has(X86_FEATURE_XSAVES)) + return; + + if (!boot_cpu_has(X86_FEATURE_XMM) || + !cpu_has_xfeatures(XFEATURE_MASK_SSE, NULL)) + return; + + /* + * On current hybrid platforms, P-cores and E-cores expose the same + * XSAVE feature set. Therefore, using the global x86_pmu.ext_regs_mask + * is sufficient to represent the hardware-supported XSAVE features. + */ + x86_pmu.ext_regs_mask |= XFEATURE_MASK_SSE; + + if (boot_cpu_has(X86_FEATURE_AVX) && + cpu_has_xfeatures(XFEATURE_MASK_YMM, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_YMM; + if (boot_cpu_has(X86_FEATURE_APX) && + cpu_has_xfeatures(XFEATURE_MASK_APX, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_APX; + if (boot_cpu_has(X86_FEATURE_AVX512F)) { + if (cpu_has_xfeatures(XFEATURE_MASK_OPMASK, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_OPMASK; + if (cpu_has_xfeatures(XFEATURE_MASK_ZMM_Hi256, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_ZMM_Hi256; + if (cpu_has_xfeatures(XFEATURE_MASK_Hi16_ZMM, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_Hi16_ZMM; + } + if (cpu_feature_enabled(X86_FEATURE_USER_SHSTK) && + cpu_has_xfeatures(XFEATURE_MASK_CET_USER, NULL)) + x86_pmu.ext_regs_mask |= XFEATURE_MASK_CET_USER; + + dest_pmu->capabilities |= PERF_PMU_CAP_EXTENDED_REGS; + if (x86_pmu.ext_regs_mask > XFEATURE_MASK_SSE) + dest_pmu->capabilities |= PERF_PMU_CAP_SIMD_REGS; } static inline void __intel_update_large_pebs_flags(struct pmu *pmu) @@ -6292,12 +6351,10 @@ static void update_pmu_cap_from_perfmonext(struct pmu *pmu) hybrid(pmu, arch_pebs_cap).counters = pebs_mask; hybrid(pmu, arch_pebs_cap).pdists = pdists_mask; - if (WARN_ON((pebs_mask | pdists_mask) & ~cntrs_mask)) { + if (WARN_ON((pebs_mask | pdists_mask) & ~cntrs_mask)) x86_pmu.arch_pebs = 0; - } else { - __intel_update_pmu_caps(pmu); + else __intel_update_large_pebs_flags(pmu); - } } else { WARN_ON(x86_pmu.arch_pebs == 1); x86_pmu.arch_pebs = 0; @@ -6321,6 +6378,7 @@ static void intel_update_pmu_caps(struct pmu *pmu) hybrid_pmu(pmu)->pmu_type == hybrid_big) hybrid(pmu, intel_cap).perf_metrics = 1; } + __intel_update_pmu_xregs_caps(pmu); } static void intel_pmu_check_hybrid_pmus(struct x86_hybrid_pmu *pmu) @@ -6474,8 +6532,6 @@ static void intel_pmu_cpu_starting(int cpu) } } - __intel_update_pmu_caps(cpuc->pmu); - if (!cpuc->shared_regs) return; diff --git a/arch/x86/events/intel/ds.c b/arch/x86/events/intel/ds.c index d0329a7eb8a5..8444670cee4a 100644 --- a/arch/x86/events/intel/ds.c +++ b/arch/x86/events/intel/ds.c @@ -1704,6 +1704,7 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event) u64 sample_type = attr->sample_type; u64 pebs_data_cfg = 0; bool gprs, tsx_weight; + u64 xgprs_mask; if (!(sample_type & ~(PERF_SAMPLE_IP|PERF_SAMPLE_TIME)) && attr->precise_ip > 1) @@ -1718,10 +1719,13 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event) * + precise_ip < 2 for the non event IP * + For RTM TSX weight we need GPRs for the abort code. */ + xgprs_mask = event->attr.sample_simd_regs_enabled ? + PEBS_GP_REGS | BIT_ULL(PERF_REG_X86_SSP) : + PEBS_GP_REGS; gprs = ((sample_type & PERF_SAMPLE_REGS_INTR) && - (attr->sample_regs_intr & PEBS_GP_REGS)) || + (attr->sample_regs_intr & xgprs_mask)) || ((sample_type & PERF_SAMPLE_REGS_USER) && - (attr->sample_regs_user & PEBS_GP_REGS)); + (attr->sample_regs_user & xgprs_mask)); tsx_weight = (sample_type & PERF_SAMPLE_WEIGHT_TYPE) && ((attr->config & INTEL_ARCH_EVENT_MASK) == @@ -1730,9 +1734,20 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event) if (gprs || (attr->precise_ip < 2) || tsx_weight) pebs_data_cfg |= PEBS_DATACFG_GP; - if ((sample_type & PERF_SAMPLE_REGS_INTR) && - (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK)) - pebs_data_cfg |= PEBS_DATACFG_XMMS; + if (sample_type & (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)) { + if (event_needs_xmm(event)) + pebs_data_cfg |= PEBS_DATACFG_XMMS; + if (x86_pmu.arch_pebs && event_needs_ymm(event)) + pebs_data_cfg |= PEBS_DATACFG_YMMHS; + if (x86_pmu.arch_pebs && event_needs_low16_zmm(event)) + pebs_data_cfg |= PEBS_DATACFG_ZMMHS; + if (x86_pmu.arch_pebs && event_needs_high16_zmm(event)) + pebs_data_cfg |= PEBS_DATACFG_H16ZMMS; + if (x86_pmu.arch_pebs && event_needs_opmask(event)) + pebs_data_cfg |= PEBS_DATACFG_OPMASKS; + if (x86_pmu.arch_pebs && event_needs_egprs(event)) + pebs_data_cfg |= PEBS_DATACFG_EGPRS; + } if (sample_type & PERF_SAMPLE_BRANCH_STACK) { /* @@ -2523,7 +2538,7 @@ static void setup_pebs_adaptive_sample_data(struct perf_event *event, return; perf_regs = container_of(regs, struct x86_perf_regs, regs); - perf_regs->xmm_regs = NULL; + x86_pmu_clear_perf_regs(regs); format_group = basic->format_group; @@ -2608,6 +2623,8 @@ static void setup_pebs_adaptive_sample_data(struct perf_event *event, next_record += nr * sizeof(u64); } + x86_pmu_update_perf_regs(event, data, regs, true); + WARN_ONCE(next_record != __pebs + basic->format_size, "PEBS record size %u, expected %llu, config %llx\n", basic->format_size, @@ -2640,7 +2657,7 @@ static void setup_arch_pebs_sample_data(struct perf_event *event, return; perf_regs = container_of(regs, struct x86_perf_regs, regs); - perf_regs->xmm_regs = NULL; + x86_pmu_clear_perf_regs(regs); __setup_perf_sample_data(event, iregs, data); @@ -2648,6 +2665,9 @@ static void setup_arch_pebs_sample_data(struct perf_event *event, again: header = at; + if (!header->size) + return; + next_record = at + sizeof(struct arch_pebs_header); if (header->basic) { struct arch_pebs_basic *basic = next_record; @@ -2678,6 +2698,11 @@ again: __setup_pebs_gpr_group(event, regs, (struct pebs_gprs *)gprs, sample_type); + + /* Currently only user space mode enables SSP. */ + if (user_mode(regs) && (sample_type & + (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER))) + perf_regs->ssp = &gprs->ssp; } if (header->aux) { @@ -2690,14 +2715,63 @@ again: meminfo->tsx_tuning, ax); } - if (header->xmm) { + if (header->xmm || header->ymmh || header->egpr || + header->opmask || header->zmmh || header->h16zmm) { + struct arch_pebs_xer_header *xer_header = next_record; struct pebs_xmm *xmm; + struct ymmh_struct *ymmh; + struct avx_512_zmm_uppers_state *zmmh; + struct avx_512_hi16_state *h16zmm; + struct avx_512_opmask_state *opmask; + struct apx_state *egpr; next_record += sizeof(struct arch_pebs_xer_header); - xmm = next_record; - perf_regs->xmm_regs = xmm->xmm; - next_record = xmm + 1; + if (header->xmm) { + xmm = next_record; + /* + * Only output XMM regs to user space when arch-PEBS + * really writes data into xstate area. + */ + if (xer_header->xstate & XFEATURE_MASK_SSE) + perf_regs->xmm_regs = xmm->xmm; + next_record = xmm + 1; + } + + if (header->ymmh) { + ymmh = next_record; + if (xer_header->xstate & XFEATURE_MASK_YMM) + perf_regs->ymmh = ymmh; + next_record = ymmh + 1; + } + + if (header->egpr) { + egpr = next_record; + if (xer_header->xstate & XFEATURE_MASK_APX) + perf_regs->egpr = egpr; + next_record = egpr + 1; + } + + if (header->opmask) { + opmask = next_record; + if (xer_header->xstate & XFEATURE_MASK_OPMASK) + perf_regs->opmask = opmask; + next_record = opmask + 1; + } + + if (header->zmmh) { + zmmh = next_record; + if (xer_header->xstate & XFEATURE_MASK_ZMM_Hi256) + perf_regs->zmmh = zmmh; + next_record = zmmh + 1; + } + + if (header->h16zmm) { + h16zmm = next_record; + if (xer_header->xstate & XFEATURE_MASK_Hi16_ZMM) + perf_regs->h16zmm = h16zmm; + next_record = h16zmm + 1; + } } if (header->lbr) { @@ -2742,6 +2816,8 @@ again: at = at + header->size; goto again; } + + x86_pmu_update_perf_regs(event, data, regs, true); } static inline void * @@ -2865,13 +2941,21 @@ __intel_pmu_pebs_last_event(struct perf_event *event, struct pt_regs *iregs, struct pt_regs *regs, struct perf_sample_data *data, - void *at, - int count, + void *at, int count, bool corrupted, setup_fn setup_sample) { struct hw_perf_event *hwc = &event->hw; - setup_sample(event, iregs, at, data, regs); + /* Skip parsing corrupted PEBS record. */ + if (corrupted) { + /* Clear stale register states in previous records. */ + memset(regs, 0, sizeof(*regs)); + x86_pmu_clear_perf_regs(regs); + perf_sample_data_init(data, 0, event->hw.last_period); + } else { + setup_sample(event, iregs, at, data, regs); + } + if (iregs == &dummy_iregs) { /* * The PEBS records may be drained in the non-overflow context, @@ -2889,12 +2973,16 @@ __intel_pmu_pebs_last_event(struct perf_event *event, } if (hwc->flags & PERF_X86_EVENT_AUTO_RELOAD) { - if ((is_pebs_counter_event_group(event))) { - /* - * The value of each sample has been updated when setup - * the corresponding sample data. - */ - perf_event_update_userpage(event); + if (is_pebs_counter_event_group(event)) { + if (corrupted) { + intel_pmu_save_and_restart_reload(event, 1); + } else { + /* + * The value of each sample has been updated + * when setup the corresponding sample data. + */ + perf_event_update_userpage(event); + } } else { /* * Now, auto-reload is only enabled in fixed period mode. @@ -2918,13 +3006,15 @@ __intel_pmu_pebs_last_event(struct perf_event *event, * counters-snapshotting record, only needs to set the new * period for the counter. */ - if (is_pebs_counter_event_group(event)) + if (is_pebs_counter_event_group(event) && !corrupted) static_call(x86_pmu_set_period)(event); else intel_pmu_save_and_restart(event); } } +static DEFINE_PER_CPU(struct x86_perf_regs, x86_pebs_regs); + static __always_inline void __intel_pmu_pebs_events(struct perf_event *event, struct pt_regs *iregs, @@ -2934,25 +3024,29 @@ __intel_pmu_pebs_events(struct perf_event *event, setup_fn setup_sample) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); - struct x86_perf_regs perf_regs; - struct pt_regs *regs = &perf_regs.regs; + struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs); + struct pt_regs *regs = &perf_regs->regs; void *at = get_next_pebs_record_by_bit(base, top, bit); int cnt = count; + x86_pmu_clear_perf_regs(regs); + if (!iregs) iregs = &dummy_iregs; while (cnt > 1) { - __intel_pmu_pebs_event(event, iregs, regs, data, at, setup_sample); + __intel_pmu_pebs_event(event, iregs, regs, data, + at, setup_sample); at += cpuc->pebs_record_size; at = get_next_pebs_record_by_bit(at, top, bit); cnt--; } - __intel_pmu_pebs_last_event(event, iregs, regs, data, at, count, setup_sample); + __intel_pmu_pebs_last_event(event, iregs, regs, data, at, + count, false, setup_sample); } -static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_data *data) +static int intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_data *data) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct debug_store *ds = cpuc->ds; @@ -2961,7 +3055,7 @@ static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_ int n; if (!x86_pmu.pebs_active) - return; + return 0; at = (struct pebs_record_core *)(unsigned long)ds->pebs_buffer_base; top = (struct pebs_record_core *)(unsigned long)ds->pebs_index; @@ -2972,22 +3066,25 @@ static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_ ds->pebs_index = ds->pebs_buffer_base; if (!test_bit(0, cpuc->active_mask)) - return; + return 0; WARN_ON_ONCE(!event); if (!event->attr.precise_ip) - return; + return 0; n = top - at; if (n <= 0) { if (event->hw.flags & PERF_X86_EVENT_AUTO_RELOAD) intel_pmu_save_and_restart_reload(event, 0); - return; + return 0; } __intel_pmu_pebs_events(event, iregs, data, at, top, 0, n, setup_pebs_fixed_sample_data); + + /* PMC0 only */ + return 1; } static void intel_pmu_pebs_event_update_no_drain(struct cpu_hw_events *cpuc, u64 mask) @@ -3010,7 +3107,7 @@ static void intel_pmu_pebs_event_update_no_drain(struct cpu_hw_events *cpuc, u64 } } -static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_data *data) +static int intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_data *data) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct debug_store *ds = cpuc->ds; @@ -3019,11 +3116,12 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {}; short error[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {}; int max_pebs_events = intel_pmu_max_num_pebs(NULL); + u64 events_bitmap = 0; int bit, i, size; u64 mask; if (!x86_pmu.pebs_active) - return; + return 0; base = (struct pebs_record_nhm *)(unsigned long)ds->pebs_buffer_base; top = (struct pebs_record_nhm *)(unsigned long)ds->pebs_index; @@ -3039,7 +3137,7 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d if (unlikely(base >= top)) { intel_pmu_pebs_event_update_no_drain(cpuc, mask); - return; + return 0; } for (at = base; at < top; at += x86_pmu.pebs_record_size) { @@ -3103,6 +3201,7 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d if ((counts[bit] == 0) && (error[bit] == 0)) continue; + events_bitmap |= BIT_ULL(bit); event = cpuc->events[bit]; if (WARN_ON_ONCE(!event)) continue; @@ -3124,6 +3223,8 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d setup_pebs_fixed_sample_data); } } + + return hweight64(events_bitmap); } static __always_inline void @@ -3158,39 +3259,46 @@ static __always_inline void __intel_pmu_handle_last_pebs_record(struct pt_regs *iregs, struct pt_regs *regs, struct perf_sample_data *data, - u64 mask, short *counts, void **last, + u64 mask, short *counts, + void **last, bool corrupted, setup_fn setup_sample) { struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct perf_event *event; + bool handled = false; int bit; for_each_set_bit(bit, (unsigned long *)&mask, X86_PMC_IDX_MAX) { if (!counts[bit]) continue; + handled = true; event = cpuc->events[bit]; - __intel_pmu_pebs_last_event(event, iregs, regs, data, last[bit], - counts[bit], setup_sample); + counts[bit], corrupted, setup_sample); } + /* All records are corrupted, reset sampling period. */ + if (!handled) + intel_pmu_pebs_event_update_no_drain(cpuc, mask); } -static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_data *data) +static int intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_data *data) { short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {}; void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS]; struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); struct debug_store *ds = cpuc->ds; - struct x86_perf_regs perf_regs; - struct pt_regs *regs = &perf_regs.regs; + struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs); + struct pt_regs *regs = &perf_regs->regs; struct pebs_basic *basic; void *base, *at, *top; + u64 events_bitmap = 0; + bool corrupted = false; u64 mask; if (!x86_pmu.pebs_active) - return; + return 0; base = (struct pebs_basic *)(unsigned long)ds->pebs_buffer_base; top = (struct pebs_basic *)(unsigned long)ds->pebs_index; @@ -3203,7 +3311,7 @@ static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_d if (unlikely(base >= top)) { intel_pmu_pebs_event_update_no_drain(cpuc, mask); - return; + return 0; } if (!iregs) @@ -3214,36 +3322,45 @@ static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_d u64 pebs_status; basic = at; + if (WARN_ON_ONCE(!basic->format_size)) { + corrupted = true; + break; + } if (basic->format_size != cpuc->pebs_record_size) continue; pebs_status = mask & basic->applicable_counters; + events_bitmap |= pebs_status; __intel_pmu_handle_pebs_record(iregs, regs, data, at, pebs_status, counts, last, setup_pebs_adaptive_sample_data); } __intel_pmu_handle_last_pebs_record(iregs, regs, data, mask, counts, last, - setup_pebs_adaptive_sample_data); + corrupted, setup_pebs_adaptive_sample_data); + + return hweight64(events_bitmap); } -static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs, - struct perf_sample_data *data) +static int intel_pmu_drain_arch_pebs(struct pt_regs *iregs, + struct perf_sample_data *data) { short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {}; void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS]; struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); union arch_pebs_index index; - struct x86_perf_regs perf_regs; - struct pt_regs *regs = &perf_regs.regs; + struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs); + struct pt_regs *regs = &perf_regs->regs; void *base, *at, *top; + u64 events_bitmap = 0; + bool corrupted = false; u64 mask; rdmsrq(MSR_IA32_PEBS_INDEX, index.whole); if (unlikely(!index.wr)) { intel_pmu_pebs_event_update_no_drain(cpuc, X86_PMC_IDX_MAX); - return; + return 0; } base = cpuc->pebs_vaddr; @@ -3271,8 +3388,10 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs, header = at; - if (WARN_ON_ONCE(!header->size)) - break; + if (WARN_ON_ONCE(!header->size)) { + corrupted = true; + goto done; + } /* 1st fragment or single record must have basic group */ if (!header->basic) { @@ -3282,6 +3401,7 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs, basic = at + sizeof(struct arch_pebs_header); pebs_status = mask & basic->applicable_counters; + events_bitmap |= pebs_status; __intel_pmu_handle_pebs_record(iregs, regs, data, at, pebs_status, counts, last, setup_arch_pebs_sample_data); @@ -3291,16 +3411,27 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs, if (!header->size) break; at += header->size; + if (WARN_ON_ONCE(at >= top)) { + corrupted = true; + goto done; + } header = at; } /* Skip last fragment or the single record */ at += header->size; + if (WARN_ON_ONCE(at > top)) { + corrupted = true; + goto done; + } } +done: __intel_pmu_handle_last_pebs_record(iregs, regs, data, mask, - counts, last, + counts, last, corrupted, setup_arch_pebs_sample_data); + + return hweight64(events_bitmap); } static void __init intel_arch_pebs_init(void) @@ -3402,7 +3533,6 @@ static void __init intel_ds_pebs_init(void) x86_pmu.flags |= PMU_FL_PEBS_ALL; x86_pmu.pebs_capable = ~0ULL; pebs_qual = "-baseline"; - x86_get_pmu(smp_processor_id())->capabilities |= PERF_PMU_CAP_EXTENDED_REGS; } else { /* Only basic record supported */ x86_pmu.large_pebs_flags &= diff --git a/arch/x86/events/intel/lbr.c b/arch/x86/events/intel/lbr.c index 22e2a06d5786..0b446d8912e4 100644 --- a/arch/x86/events/intel/lbr.c +++ b/arch/x86/events/intel/lbr.c @@ -714,7 +714,7 @@ static inline bool vlbr_exclude_host(void) struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events); return test_bit(INTEL_PMC_IDX_FIXED_VLBR, - (unsigned long *)&cpuc->intel_ctrl_guest_mask); + (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask); } void intel_pmu_lbr_enable_all(bool pmi) diff --git a/arch/x86/events/perf_event.h b/arch/x86/events/perf_event.h index fab9da78a5c7..e274802ef062 100644 --- a/arch/x86/events/perf_event.h +++ b/arch/x86/events/perf_event.h @@ -147,6 +147,199 @@ static inline bool is_acr_self_reload_event(struct perf_event *event) return test_bit(hwc->idx, (unsigned long *)&hwc->config1); } +static inline bool __event_needs_xmm(struct perf_event *event, u64 sample_type) +{ + if (event->attr.sample_simd_regs_enabled) { + if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_XMM_QWORDS) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_simd_vec_reg_user > 0)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_simd_vec_reg_intr > 0)) + return true; + } else { + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK)) + return true; + } + + return false; +} + +static inline bool event_needs_xmm(struct perf_event *event) +{ + return __event_needs_xmm(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_ymm(struct perf_event *event, u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_YMM_QWORDS) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_simd_vec_reg_user > 0)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_simd_vec_reg_intr > 0)) + return true; + + return false; +} + +static inline bool event_needs_ymm(struct perf_event *event) +{ + return __event_needs_ymm(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_low16_zmm(struct perf_event *event, + u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_simd_vec_reg_user > 0)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_simd_vec_reg_intr > 0)) + return true; + + return false; +} + +static inline bool event_needs_low16_zmm(struct perf_event *event) +{ + return __event_needs_low16_zmm(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_high16_zmm(struct perf_event *event, + u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (fls64(event->attr.sample_simd_vec_reg_user) > PERF_X86_H16ZMM_BASE)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (fls64(event->attr.sample_simd_vec_reg_intr) > PERF_X86_H16ZMM_BASE)) + return true; + + return false; +} + +static inline bool event_needs_high16_zmm(struct perf_event *event) +{ + return __event_needs_high16_zmm(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_opmask(struct perf_event *event, + u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + if (event->attr.sample_simd_pred_reg_qwords != PERF_X86_OPMASK_QWORDS) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_simd_pred_reg_user > 0)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_simd_pred_reg_intr > 0)) + return true; + + return false; +} + +static inline bool event_needs_opmask(struct perf_event *event) +{ + return __event_needs_opmask(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_egprs(struct perf_event *event, + u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_regs_user & PERF_X86_EGPRS_MASK)) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_regs_intr & PERF_X86_EGPRS_MASK)) + return true; + + return false; +} + +static inline bool event_needs_egprs(struct perf_event *event) +{ + return __event_needs_egprs(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + +static inline bool __event_needs_ssp(struct perf_event *event, + u64 sample_type) +{ + if (!event->attr.sample_simd_regs_enabled) + return false; + + if ((sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_type & PERF_SAMPLE_REGS_USER) && + (event->attr.sample_regs_user & BIT_ULL(PERF_REG_X86_SSP))) + return true; + + if ((sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) && + (event->attr.sample_regs_intr & BIT_ULL(PERF_REG_X86_SSP))) + return true; + + return false; +} + +static inline bool event_needs_ssp(struct perf_event *event) +{ + return __event_needs_ssp(event, + PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER); +} + struct amd_nb { int nb_id; /* NorthBridge id */ int refcnt; /* reference count */ @@ -347,8 +540,8 @@ struct cpu_hw_events { /* * Intel host/guest exclude bits */ - u64 intel_ctrl_guest_mask; - u64 intel_ctrl_host_mask; + u64 intel_ctrl_exclude_host_mask; + u64 intel_ctrl_exclude_guest_mask; struct perf_guest_switch_msr guest_switch_msrs[X86_PMC_IDX_MAX]; /* @@ -948,7 +1141,7 @@ struct x86_pmu { int pebs_record_size; int pebs_buffer_size; u64 pebs_events_mask; - void (*drain_pebs)(struct pt_regs *regs, struct perf_sample_data *data); + int (*drain_pebs)(struct pt_regs *regs, struct perf_sample_data *data); struct event_constraint *pebs_constraints; void (*pebs_aliases)(struct perf_event *event); u64 (*pebs_latency_data)(struct perf_event *event, u64 status); @@ -1025,9 +1218,16 @@ struct x86_pmu { unsigned int flags; /* + * Extended regs, e.g., vector registers + * Utilize the same format as the XFEATURE_MASK_* + */ + u64 ext_regs_mask; + + /* * Intel host/guest support (KVM) */ - struct perf_guest_switch_msr *(*guest_get_msrs)(int *nr, void *data); + struct perf_guest_switch_msr *(*guest_get_msrs)(int *nr, + struct x86_guest_pebs *guest_pebs); /* * Check period value for PERF_EVENT_IOC_PERIOD ioctl. @@ -1046,7 +1246,7 @@ struct x86_pmu { * unique capabilities. */ int num_hybrid_pmus; - struct x86_hybrid_pmu *hybrid_pmu; + struct x86_hybrid_pmu *hybrid_pmu __counted_by_ptr(num_hybrid_pmus); enum intel_cpu_type (*get_hybrid_cpu_type) (void); }; @@ -1311,6 +1511,13 @@ void x86_pmu_enable_event(struct perf_event *event); int x86_pmu_handle_irq(struct pt_regs *regs); +void x86_pmu_clear_perf_regs(struct pt_regs *regs); + +void x86_pmu_update_perf_regs(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs, + bool from_pebs); + void x86_pmu_show_pmu_cap(struct pmu *pmu); static inline int x86_pmu_num_counters(struct pmu *pmu) diff --git a/arch/x86/include/asm/cpufeatures.h b/arch/x86/include/asm/cpufeatures.h index f70ee74b5f92..bce3cc6bbadd 100644 --- a/arch/x86/include/asm/cpufeatures.h +++ b/arch/x86/include/asm/cpufeatures.h @@ -76,7 +76,7 @@ #define X86_FEATURE_K8 ( 3*32+ 4) /* Opteron, Athlon64 */ #define X86_FEATURE_ZEN5 ( 3*32+ 5) /* CPU based on Zen5 microarchitecture */ #define X86_FEATURE_ZEN6 ( 3*32+ 6) /* CPU based on Zen6 microarchitecture */ -/* Free ( 3*32+ 7) */ +#define X86_FEATURE_RMPOPT ( 3*32+ 7) /* Support for AMD RMPOPT instruction */ #define X86_FEATURE_CONSTANT_TSC ( 3*32+ 8) /* "constant_tsc" TSC ticks at a constant rate */ /* free: was #define X86_FEATURE_UP ( 3*32+ 9) * "up" SMP kernel running on UP */ #define X86_FEATURE_ART ( 3*32+10) /* "art" Always running timer (ART) */ @@ -430,6 +430,7 @@ #define X86_FEATURE_SUCCOR (17*32+ 1) /* "succor" Uncorrectable error containment and recovery */ #define X86_FEATURE_CPPC_PERF_PRIO (17*32+ 2) /* CPPC Floor Perf support */ #define X86_FEATURE_SMCA (17*32+ 3) /* "smca" Scalable MCA */ +#define X86_FEATURE_BTB_CTX_ISOLATION (17*32+ 4) /* AMD: Branch predictions contexts isolated */ /* Intel-defined CPU features, CPUID level 0x00000007:0 (EDX), word 18 */ #define X86_FEATURE_AVX512_4VNNIW (18*32+ 2) /* "avx512_4vnniw" AVX-512 Neural Network Instructions */ @@ -483,6 +484,8 @@ #define X86_FEATURE_AUTOIBRS (20*32+ 8) /* Automatic IBRS */ #define X86_FEATURE_NO_SMM_CTL_MSR (20*32+ 9) /* SMM_CTL MSR is not present */ +#define X86_FEATURE_L2_TLB_SIZE_X32 (20*32+14) /* L2 TLB sizes are encoded as multiples of 32 */ + #define X86_FEATURE_GP_ON_USER_CPUID (20*32+17) /* User CPUID faulting */ #define X86_FEATURE_PREFETCHI (20*32+20) /* Prefetch Data/Instruction to Cache Level */ diff --git a/arch/x86/include/asm/cpuid/api.h b/arch/x86/include/asm/cpuid/api.h index 82eddfa2347b..2d9f3d4d63de 100644 --- a/arch/x86/include/asm/cpuid/api.h +++ b/arch/x86/include/asm/cpuid/api.h @@ -204,7 +204,7 @@ static inline u32 cpuid_base_hypervisor(const char *sig, u32 leaves) * from PVH early boot code before instrumentation is set up * and memcmp() itself may be instrumented. */ - if (!__builtin_memcmp(sig, signature, 12) && + if (!__inline_memcmp(sig, signature, 12) && (leaves == 0 || ((eax - base) >= leaves))) return base; } diff --git a/arch/x86/include/asm/fpu/regset.h b/arch/x86/include/asm/fpu/regset.h index 697b77e96025..433720990f5d 100644 --- a/arch/x86/include/asm/fpu/regset.h +++ b/arch/x86/include/asm/fpu/regset.h @@ -9,10 +9,8 @@ extern user_regset_active_fn regset_fpregs_active, regset_xregset_fpregs_active, ssp_active; -extern user_regset_get2_fn fpregs_get, xfpregs_get, fpregs_soft_get, - xstateregs_get, ssp_get; -extern user_regset_set_fn fpregs_set, xfpregs_set, fpregs_soft_set, - xstateregs_set, ssp_set; +extern user_regset_get2_fn fpregs_get, xfpregs_get, xstateregs_get, ssp_get; +extern user_regset_set_fn fpregs_set, xfpregs_set, xstateregs_set, ssp_set; /* * xstateregs_active == regset_fpregs_active. Please refer to the comment diff --git a/arch/x86/include/asm/fpu/sched.h b/arch/x86/include/asm/fpu/sched.h index 89004f4ca208..67b0dfa3530f 100644 --- a/arch/x86/include/asm/fpu/sched.h +++ b/arch/x86/include/asm/fpu/sched.h @@ -10,6 +10,8 @@ #include <asm/trace/fpu.h> extern void save_fpregs_to_fpstate(struct fpu *fpu); +extern void update_fpu_state_and_flag(struct fpu *fpu, + struct task_struct *task); extern void fpu__drop(struct task_struct *tsk); extern int fpu_clone(struct task_struct *dst, u64 clone_flags, bool minimal, unsigned long shstk_addr); @@ -32,12 +34,10 @@ extern void fpu_flush_thread(void); static inline void switch_fpu(struct task_struct *old, int cpu) { if (!test_tsk_thread_flag(old, TIF_NEED_FPU_LOAD) && - cpu_feature_enabled(X86_FEATURE_FPU) && !(old->flags & (PF_KTHREAD | PF_USER_WORKER))) { struct fpu *old_fpu = x86_task_fpu(old); - set_tsk_thread_flag(old, TIF_NEED_FPU_LOAD); - save_fpregs_to_fpstate(old_fpu); + update_fpu_state_and_flag(old_fpu, old); /* * The save operation preserved register state, so the * fpu_fpregs_owner_ctx is still @old_fpu. Store the diff --git a/arch/x86/include/asm/fpu/types.h b/arch/x86/include/asm/fpu/types.h index 93e99d2583d6..ea02832f7043 100644 --- a/arch/x86/include/asm/fpu/types.h +++ b/arch/x86/include/asm/fpu/types.h @@ -75,30 +75,6 @@ struct fxregs_state { #define MXCSR_AND_FLAGS_SIZE sizeof(u64) /* - * Software based FPU emulation state. This is arbitrary really, - * it matches the x87 format to make it easier to understand: - */ -struct swregs_state { - u32 cwd; - u32 swd; - u32 twd; - u32 fip; - u32 fcs; - u32 foo; - u32 fos; - /* 8*10 bytes for each FP-reg = 80 bytes: */ - u32 st_space[20]; - u8 ftop; - u8 changed; - u8 lookahead; - u8 no_update; - u8 rm; - u8 alimit; - struct math_emu_info *info; - u32 entry_eip; -}; - -/* * List of XSAVE features Linux knows about: */ enum xfeature { @@ -369,7 +345,6 @@ struct xregs_state { union fpregs_state { struct fregs_state fsave; struct fxregs_state fxsave; - struct swregs_state soft; struct xregs_state xsave; u8 __padding[PAGE_SIZE]; }; diff --git a/arch/x86/include/asm/fpu/xstate.h b/arch/x86/include/asm/fpu/xstate.h index 7a7dc9d56027..19dec5f0b1c7 100644 --- a/arch/x86/include/asm/fpu/xstate.h +++ b/arch/x86/include/asm/fpu/xstate.h @@ -110,6 +110,9 @@ int xfeature_size(int xfeature_nr); void xsaves(struct xregs_state *xsave, u64 mask); void xrstors(struct xregs_state *xsave, u64 mask); +void xsaves_nmi(struct xregs_state *xsave, u64 mask); + +unsigned int xstate_calculate_size(u64 xfeatures, bool compacted); int xfd_enable_feature(u64 xfd_err); diff --git a/arch/x86/include/asm/kvm-x86-ops.h b/arch/x86/include/asm/kvm-x86-ops.h index e213c9ae3e30..5c358c40eae8 100644 --- a/arch/x86/include/asm/kvm-x86-ops.h +++ b/arch/x86/include/asm/kvm-x86-ops.h @@ -99,6 +99,7 @@ KVM_X86_OP_OPTIONAL_RET0(tdp_has_smep) KVM_X86_OP(load_mmu_pgd) KVM_X86_OP_OPTIONAL_RET0(set_external_spte) KVM_X86_OP_OPTIONAL(free_external_spt) +KVM_X86_OP_OPTIONAL_RET0(topup_external_cache) KVM_X86_OP(has_wbinvd_exit) KVM_X86_OP(get_l2_tsc_offset) KVM_X86_OP(get_l2_tsc_multiplier) diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_host.h index 683bb8bf43a9..20b9db3f5203 100644 --- a/arch/x86/include/asm/kvm_host.h +++ b/arch/x86/include/asm/kvm_host.h @@ -609,15 +609,6 @@ struct kvm_pmu { u64 pebs_data_cfg_rsvd; /* - * If a guest counter is cross-mapped to host counter with different - * index, its PEBS capability will be temporarily disabled. - * - * The user should make sure that this mask is updated - * after disabling interrupts and before perf_guest_get_msrs(); - */ - u64 host_cross_mapped_mask; - - /* * The gate to release perf_events not marked in * pmc_in_use only once in a vcpu time slice. */ @@ -1644,6 +1635,7 @@ struct kvm_x86_ops { /* Update external page tables for page table about to be freed. */ void (*free_external_spt)(struct kvm *kvm, struct kvm_mmu_page *sp); + int (*topup_external_cache)(struct kvm_vcpu *vcpu, int min_nr_spts); bool (*has_wbinvd_exit)(void); diff --git a/arch/x86/include/asm/local.h b/arch/x86/include/asm/local.h index 4957018fef3e..68af7b74450a 100644 --- a/arch/x86/include/asm/local.h +++ b/arch/x86/include/asm/local.h @@ -32,14 +32,14 @@ static inline void local_add(long i, local_t *l) { asm volatile(_ASM_ADD "%1,%0" : "+m" (l->a.counter) - : "ir" (i)); + : "er" (i)); } static inline void local_sub(long i, local_t *l) { asm volatile(_ASM_SUB "%1,%0" : "+m" (l->a.counter) - : "ir" (i)); + : "er" (i)); } /** diff --git a/arch/x86/include/asm/math_emu.h b/arch/x86/include/asm/math_emu.h deleted file mode 100644 index 3c42743083ed..000000000000 --- a/arch/x86/include/asm/math_emu.h +++ /dev/null @@ -1,15 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef _ASM_X86_MATH_EMU_H -#define _ASM_X86_MATH_EMU_H - -#include <asm/ptrace.h> - -/* This structure matches the layout of the data saved to the stack - following a device-not-present interrupt, part of it saved - automatically by the 80386/80486. - */ -struct math_emu_info { - long ___orig_eip; - struct pt_regs *regs; -}; -#endif /* _ASM_X86_MATH_EMU_H */ diff --git a/arch/x86/include/asm/msr-index.h b/arch/x86/include/asm/msr-index.h index 3a8e51a0c9e8..82ca6356dc62 100644 --- a/arch/x86/include/asm/msr-index.h +++ b/arch/x86/include/asm/msr-index.h @@ -350,6 +350,13 @@ #define ARCH_PEBS_LBR_SHIFT 40 #define ARCH_PEBS_LBR (0x3ull << ARCH_PEBS_LBR_SHIFT) #define ARCH_PEBS_VECR_XMM BIT_ULL(49) +#define ARCH_PEBS_VECR_YMMH BIT_ULL(50) +#define ARCH_PEBS_VECR_EGPRS BIT_ULL(51) +#define ARCH_PEBS_VECR_OPMASK BIT_ULL(53) +#define ARCH_PEBS_VECR_ZMMH BIT_ULL(54) +#define ARCH_PEBS_VECR_H16ZMM BIT_ULL(55) +#define ARCH_PEBS_VECR_EXT_SHIFT 49 +#define ARCH_PEBS_VECR_EXT (0x7full << ARCH_PEBS_VECR_EXT_SHIFT) #define ARCH_PEBS_GPR BIT_ULL(61) #define ARCH_PEBS_AUX BIT_ULL(62) #define ARCH_PEBS_EN BIT_ULL(63) @@ -761,6 +768,9 @@ #define MSR_AMD64_SEG_RMP_ENABLED_BIT 0 #define MSR_AMD64_SEG_RMP_ENABLED BIT_ULL(MSR_AMD64_SEG_RMP_ENABLED_BIT) #define MSR_AMD64_RMP_SEGMENT_SHIFT(x) (((x) & GENMASK_ULL(13, 8)) >> 8) +#define MSR_AMD64_RMPOPT_BASE 0xc0010139 +#define MSR_AMD64_RMPOPT_ENABLE_BIT 0 +#define MSR_AMD64_RMPOPT_ENABLE BIT_ULL(MSR_AMD64_RMPOPT_ENABLE_BIT) #define MSR_SVSM_CAA 0xc001f000 diff --git a/arch/x86/include/asm/perf_event.h b/arch/x86/include/asm/perf_event.h index 1eb13673e889..5c92d43bbef8 100644 --- a/arch/x86/include/asm/perf_event.h +++ b/arch/x86/include/asm/perf_event.h @@ -150,6 +150,11 @@ #define PEBS_DATACFG_LBRS BIT_ULL(3) #define PEBS_DATACFG_CNTR BIT_ULL(4) #define PEBS_DATACFG_METRICS BIT_ULL(5) +#define PEBS_DATACFG_YMMHS BIT_ULL(6) +#define PEBS_DATACFG_OPMASKS BIT_ULL(7) +#define PEBS_DATACFG_ZMMHS BIT_ULL(8) +#define PEBS_DATACFG_H16ZMMS BIT_ULL(9) +#define PEBS_DATACFG_EGPRS BIT_ULL(10) #define PEBS_DATACFG_LBR_SHIFT 24 #define PEBS_DATACFG_CNTR_SHIFT 32 #define PEBS_DATACFG_CNTR_MASK GENMASK_ULL(15, 0) @@ -547,7 +552,8 @@ struct arch_pebs_header { rsvd3:7, xmm:1, ymmh:1, - rsvd4:2, + egpr:1, + rsvd4:1, opmask:1, zmmh:1, h16zmm:1, @@ -728,7 +734,32 @@ extern void perf_events_lapic_init(void); struct pt_regs; struct x86_perf_regs { struct pt_regs regs; - u64 *xmm_regs; + u64 abi; + union { + u64 *xmm_regs; + u32 *xmm_space; /* for xsaves */ + }; + union { + u64 *ymmh_regs; + struct ymmh_struct *ymmh; + }; + union { + u64 *zmmh_regs; + struct avx_512_zmm_uppers_state *zmmh; + }; + union { + u64 *h16zmm_regs; + struct avx_512_hi16_state *h16zmm; + }; + union { + u64 *opmask_regs; + struct avx_512_opmask_state *opmask; + }; + union { + u64 *egpr_regs; + struct apx_state *egpr; + }; + u64 *ssp; }; extern unsigned long perf_arch_instruction_pointer(struct pt_regs *regs); @@ -788,11 +819,18 @@ extern void perf_load_guest_lvtpc(u32 guest_lvtpc); extern void perf_put_guest_lvtpc(void); #endif +struct x86_guest_pebs { + u64 enable; + u64 ds_area; + u64 data_cfg; +}; #if defined(CONFIG_PERF_EVENTS) && defined(CONFIG_CPU_SUP_INTEL) -extern struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data); +extern struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, + struct x86_guest_pebs *guest_pebs); extern void x86_perf_get_lbr(struct x86_pmu_lbr *lbr); #else -struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data); +struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, + struct x86_guest_pebs *guest_pebs); static inline void x86_perf_get_lbr(struct x86_pmu_lbr *lbr) { memset(lbr, 0, sizeof(*lbr)); diff --git a/arch/x86/include/asm/preempt.h b/arch/x86/include/asm/preempt.h index fafb6f8cdac3..d16d9c2f7b08 100644 --- a/arch/x86/include/asm/preempt.h +++ b/arch/x86/include/asm/preempt.h @@ -137,43 +137,15 @@ static __always_inline bool should_resched(int preempt_offset) extern asmlinkage void preempt_schedule(void); extern asmlinkage void preempt_schedule_thunk(void); -#define preempt_schedule_dynamic_enabled preempt_schedule_thunk -#define preempt_schedule_dynamic_disabled NULL - extern asmlinkage void preempt_schedule_notrace(void); extern asmlinkage void preempt_schedule_notrace_thunk(void); -#define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace_thunk -#define preempt_schedule_notrace_dynamic_disabled NULL - -#ifdef CONFIG_PREEMPT_DYNAMIC - -DECLARE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled); - -#define __preempt_schedule() \ -do { \ - __STATIC_CALL_MOD_ADDRESSABLE(preempt_schedule); \ - asm volatile ("call " STATIC_CALL_TRAMP_STR(preempt_schedule) : ASM_CALL_CONSTRAINT); \ -} while (0) - -DECLARE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled); - -#define __preempt_schedule_notrace() \ -do { \ - __STATIC_CALL_MOD_ADDRESSABLE(preempt_schedule_notrace); \ - asm volatile ("call " STATIC_CALL_TRAMP_STR(preempt_schedule_notrace) : ASM_CALL_CONSTRAINT); \ -} while (0) - -#else /* PREEMPT_DYNAMIC */ - #define __preempt_schedule() \ asm volatile ("call preempt_schedule_thunk" : ASM_CALL_CONSTRAINT); #define __preempt_schedule_notrace() \ asm volatile ("call preempt_schedule_notrace_thunk" : ASM_CALL_CONSTRAINT); -#endif /* PREEMPT_DYNAMIC */ - #endif /* PREEMPTION */ #undef __pc_op diff --git a/arch/x86/include/asm/processor.h b/arch/x86/include/asm/processor.h index ec9db0dfa0df..77488ff5eb8c 100644 --- a/arch/x86/include/asm/processor.h +++ b/arch/x86/include/asm/processor.h @@ -10,7 +10,7 @@ struct mm_struct; struct io_bitmap; struct vm86; -#include <asm/math_emu.h> +#include <asm/ptrace.h> #include <asm/segment.h> #include <asm/types.h> #include <uapi/asm/sigcontext.h> diff --git a/arch/x86/include/asm/sev.h b/arch/x86/include/asm/sev.h index 9e7a077c445d..c7ea9845351e 100644 --- a/arch/x86/include/asm/sev.h +++ b/arch/x86/include/asm/sev.h @@ -517,6 +517,7 @@ void snp_accept_memory(phys_addr_t start, phys_addr_t end); u64 snp_get_unsupported_features(u64 status); u64 sev_get_status(void); void sev_show_status(void); +bool early_is_sevsnp_guest(void); int prepare_pte_enc(struct pte_enc_desc *d); void set_pte_enc_mask(pte_t *kpte, unsigned long pfn, pgprot_t new_prot); void snp_kexec_finish(void); @@ -626,6 +627,7 @@ static inline void snp_accept_memory(phys_addr_t start, phys_addr_t end) { } static inline u64 snp_get_unsupported_features(u64 status) { return 0; } static inline u64 sev_get_status(void) { return 0; } static inline void sev_show_status(void) { } +static inline bool early_is_sevsnp_guest(void) { return false; } static inline int prepare_pte_enc(struct pte_enc_desc *d) { return 0; } static inline void set_pte_enc_mask(pte_t *kpte, unsigned long pfn, pgprot_t new_prot) { } static inline void snp_kexec_finish(void) { } @@ -662,6 +664,7 @@ static inline void snp_leak_pages(u64 pfn, unsigned int pages) __snp_leak_pages(pfn, pages, true); } int snp_prepare(void); +void snp_enable_rmpopt(void); void snp_shutdown(void); #else static inline bool snp_probe_rmptable_info(void) { return false; } @@ -680,6 +683,7 @@ static inline void snp_leak_pages(u64 pfn, unsigned int npages) {} static inline void kdump_sev_callback(void) { } static inline void snp_fixup_e820_tables(void) {} static inline int snp_prepare(void) { return -ENODEV; } +static inline void snp_enable_rmpopt(void) {} static inline void snp_shutdown(void) {} #endif diff --git a/arch/x86/include/asm/shared/string.h b/arch/x86/include/asm/shared/string.h new file mode 100644 index 000000000000..6bef90d62a21 --- /dev/null +++ b/arch/x86/include/asm/shared/string.h @@ -0,0 +1,52 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef _ASM_X86_SHARED_STRING_H +#define _ASM_X86_SHARED_STRING_H + +static __always_inline void *__inline_memcpy(void *to, const void *from, size_t len) +{ + void *ret = to; + + asm volatile("rep movsb" + : "+D" (to), "+S" (from), "+c" (len) + : : "memory"); + return ret; +} + +static __always_inline void *__inline_memset(void *s, int v, size_t n) +{ + void *ret = s; + + asm volatile("rep stosb" + : "+D" (s), "+c" (n) + : "a" ((uint8_t)v) + : "memory"); + return ret; +} + +/* + * Returns: 0 (equal) + * 1 (not equal) + * + * In contrast, the regular memcmp() follows glibc return value semantics. + */ +static __always_inline int __inline_memcmp(const void *s1, const void *s2, size_t len) +{ + bool diff; + + /* + * Make sure ZF is properly set in the len==0 case because in it, + * RCX==0 and the REPE; CMPSB won't get executed. + * + * The "cc" clobber has no meaning anymore, just source compatibility. + * On x86 the flag status bits are automatically added to the clobber + * set when there are no =@ccXY constraints. Keep it as documentation. + */ + asm volatile("test %3, %3\n\t" + "repe cmpsb" + : "=@ccnz" (diff), "+D" (s1), "+S" (s2), "+c" (len) + : : /* "cc", */ "memory"); + + return diff; +} + +#endif /* _ASM_X86_SHARED_STRING_H */ diff --git a/arch/x86/include/asm/shared/tdx.h b/arch/x86/include/asm/shared/tdx.h index f20e91d7ac35..bf000f1fb42e 100644 --- a/arch/x86/include/asm/shared/tdx.h +++ b/arch/x86/include/asm/shared/tdx.h @@ -143,6 +143,11 @@ struct tdx_module_args { u64 rbx; u64 rdi; u64 rsi; + /* + * Leaf ABI version. Note that it gets encoded into RAX along with the + * leaf number. + */ + u8 version; }; /* Used to communicate with the TDX module */ @@ -171,6 +176,7 @@ static inline u64 _tdx_hypercall(u64 fn, u64 r12, u64 r13, u64 r14, u64 r15) return __tdx_hypercall(&args); } +void __noreturn tdx_panic(const char *msg); /* Called from __tdx_hypercall() for unrecoverable failure */ void __noreturn __tdx_hypercall_failed(void); diff --git a/arch/x86/include/asm/string.h b/arch/x86/include/asm/string.h index 9cb5aae7fba9..dbf59f0d4cca 100644 --- a/arch/x86/include/asm/string.h +++ b/arch/x86/include/asm/string.h @@ -8,25 +8,6 @@ # include <asm/string_64.h> #endif -static __always_inline void *__inline_memcpy(void *to, const void *from, size_t len) -{ - void *ret = to; - - asm volatile("rep movsb" - : "+D" (to), "+S" (from), "+c" (len) - : : "memory"); - return ret; -} - -static __always_inline void *__inline_memset(void *s, int v, size_t n) -{ - void *ret = s; - - asm volatile("rep stosb" - : "+D" (s), "+c" (n) - : "a" ((uint8_t)v) - : "memory"); - return ret; -} +#include <asm/shared/string.h> #endif /* _ASM_X86_STRING_H */ diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 831d3dda3b38..2dae70c2b014 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -3,7 +3,6 @@ #define _ASM_X86_STRING_64_H #ifdef __KERNEL__ -#include <linux/jump_label.h> /* Written 2002 by Andi Kleen */ diff --git a/arch/x86/include/asm/tdx.h b/arch/x86/include/asm/tdx.h index 89e97d5761d8..e186dfe5bf88 100644 --- a/arch/x86/include/asm/tdx.h +++ b/arch/x86/include/asm/tdx.h @@ -36,6 +36,7 @@ /* Bit definitions of TDX_FEATURES0 metadata field */ #define TDX_FEATURES0_TD_PRESERVING BIT_ULL(1) #define TDX_FEATURES0_NO_RBP_MOD BIT_ULL(18) +#define TDX_FEATURES0_DYNAMIC_PAMT BIT_ULL(36) #ifndef __ASSEMBLER__ @@ -118,12 +119,34 @@ static inline bool tdx_supports_runtime_update(const struct tdx_sys_info *sysinf return sysinfo->features.tdx_features0 & TDX_FEATURES0_TD_PRESERVING; } +bool tdx_supports_dynamic_pamt(const struct tdx_sys_info *sysinfo); + +/* Simple structure for pre-allocating DPAMT pages outside of spinlocks. */ +struct tdx_pamt_cache { + struct list_head page_list; + int cnt; +}; + +static inline void tdx_init_pamt_cache(struct tdx_pamt_cache *cache) +{ + INIT_LIST_HEAD(&cache->page_list); + cache->cnt = 0; +} + +void tdx_free_pamt_cache(struct tdx_pamt_cache *cache); +int tdx_topup_pamt_cache(struct tdx_pamt_cache *cache, unsigned long npages); +int tdx_pamt_get(kvm_pfn_t pfn, struct tdx_pamt_cache *cache); +void tdx_pamt_put(kvm_pfn_t pfn); + int tdx_guest_keyid_alloc(void); u32 tdx_get_nr_guest_keyids(void); void tdx_guest_keyid_free(unsigned int keyid); void tdx_quirk_reset_paddr(unsigned long base, unsigned long size); +struct page *tdx_alloc_control_page(void); +void tdx_free_control_page(struct page *page); + struct tdx_td { /* TD root structure: */ struct page *tdr_page; diff --git a/arch/x86/include/asm/tdx_global_metadata.h b/arch/x86/include/asm/tdx_global_metadata.h index 41150d546589..8a3cc1a2a41e 100644 --- a/arch/x86/include/asm/tdx_global_metadata.h +++ b/arch/x86/include/asm/tdx_global_metadata.h @@ -1,7 +1,7 @@ /* SPDX-License-Identifier: GPL-2.0 */ -/* Automatically generated TDX global metadata structures. */ -#ifndef _X86_VIRT_TDX_AUTO_GENERATED_TDX_GLOBAL_METADATA_H -#define _X86_VIRT_TDX_AUTO_GENERATED_TDX_GLOBAL_METADATA_H +/* TDX global metadata structures. */ +#ifndef _X86_VIRT_TDX_TDX_GLOBAL_METADATA_H +#define _X86_VIRT_TDX_TDX_GLOBAL_METADATA_H #include <linux/types.h> @@ -21,6 +21,9 @@ struct tdx_sys_info_tdmr { u16 pamt_4k_entry_size; u16 pamt_2m_entry_size; u16 pamt_1g_entry_size; + + /* Optional metadata, if DPAMT is supported */ + u8 pamt_page_bitmap_entry_bits; }; struct tdx_sys_info_td_ctrl { diff --git a/arch/x86/include/asm/traps.h b/arch/x86/include/asm/traps.h index 3f24cc472ce9..e13f1025c62e 100644 --- a/arch/x86/include/asm/traps.h +++ b/arch/x86/include/asm/traps.h @@ -37,8 +37,6 @@ static inline int get_si_code(unsigned long condition) return TRAP_BRKPT; } -void math_emulate(struct math_emu_info *); - bool fault_in_kernel_space(unsigned long address); #ifdef CONFIG_VMAP_STACK diff --git a/arch/x86/include/uapi/asm/perf_regs.h b/arch/x86/include/uapi/asm/perf_regs.h index 7c9d2bb3833b..faaa82df688d 100644 --- a/arch/x86/include/uapi/asm/perf_regs.h +++ b/arch/x86/include/uapi/asm/perf_regs.h @@ -2,6 +2,8 @@ #ifndef _ASM_X86_PERF_REGS_H #define _ASM_X86_PERF_REGS_H +#include <linux/bits.h> + enum perf_event_x86_regs { PERF_REG_X86_AX, PERF_REG_X86_BX, @@ -27,9 +29,35 @@ enum perf_event_x86_regs { PERF_REG_X86_R13, PERF_REG_X86_R14, PERF_REG_X86_R15, + /* + * The eGPRs/SSP and XMM have overlaps. Only one can be used + * at a time. The ABI PERF_SAMPLE_REGS_ABI_SIMD is used to + * distinguish which one is used. If PERF_SAMPLE_REGS_ABI_SIMD + * is set, then eGPRs/SSP is used, otherwise, XMM is used. + * + * Extended GPRs (eGPRs) + */ + PERF_REG_X86_R16, + PERF_REG_X86_R17, + PERF_REG_X86_R18, + PERF_REG_X86_R19, + PERF_REG_X86_R20, + PERF_REG_X86_R21, + PERF_REG_X86_R22, + PERF_REG_X86_R23, + PERF_REG_X86_R24, + PERF_REG_X86_R25, + PERF_REG_X86_R26, + PERF_REG_X86_R27, + PERF_REG_X86_R28, + PERF_REG_X86_R29, + PERF_REG_X86_R30, + PERF_REG_X86_R31, + PERF_REG_X86_SSP, /* These are the limits for the GPRs. */ PERF_REG_X86_32_MAX = PERF_REG_X86_GS + 1, PERF_REG_X86_64_MAX = PERF_REG_X86_R15 + 1, + PERF_REG_MISC_MAX = PERF_REG_X86_SSP + 1, /* These all need two bits set because they are 128bit */ PERF_REG_X86_XMM0 = 32, @@ -54,5 +82,30 @@ enum perf_event_x86_regs { }; #define PERF_REG_EXTENDED_MASK (~((1ULL << PERF_REG_X86_XMM0) - 1)) +#define PERF_X86_EGPRS_MASK __GENMASK_ULL(PERF_REG_X86_R31, PERF_REG_X86_R16) + +enum { + PERF_X86_SIMD_XMM_REGS = 16, + PERF_X86_SIMD_YMM_REGS = 16, + PERF_X86_SIMD_ZMM_REGS = 32, + PERF_X86_SIMD_VEC_REGS_MAX = PERF_X86_SIMD_ZMM_REGS, + + PERF_X86_SIMD_OPMASK_REGS = 8, + PERF_X86_SIMD_PRED_REGS_MAX = PERF_X86_SIMD_OPMASK_REGS, +}; + +#define PERF_X86_SIMD_PRED_MASK __GENMASK(PERF_X86_SIMD_PRED_REGS_MAX - 1, 0) +#define PERF_X86_SIMD_VEC_MASK __GENMASK_ULL(PERF_X86_SIMD_VEC_REGS_MAX - 1, 0) + +#define PERF_X86_H16ZMM_BASE 16 + +enum { + /* 1 qword = 8 bytes */ + PERF_X86_OPMASK_QWORDS = 1, + PERF_X86_XMM_QWORDS = 2, + PERF_X86_YMM_QWORDS = 4, + PERF_X86_ZMM_QWORDS = 8, + PERF_X86_SIMD_QWORDS_MAX = PERF_X86_ZMM_QWORDS, +}; #endif /* _ASM_X86_PERF_REGS_H */ diff --git a/arch/x86/include/uapi/asm/sigcontext.h b/arch/x86/include/uapi/asm/sigcontext.h index d0d9b331d3a1..cff01406c0f4 100644 --- a/arch/x86/include/uapi/asm/sigcontext.h +++ b/arch/x86/include/uapi/asm/sigcontext.h @@ -34,6 +34,21 @@ * fpstate+extended_size-FP_XSTATE_MAGIC2_SIZE address) is set to * FP_XSTATE_MAGIC2 so that you can sanity check your size calculations.) * + * The xstate_size field indicates the actual size of the xstate context + * (including the 512-byte FXSAVE area and the 64-byte XSAVE header struct + * _header). This size is used in conjunction with the pointer to the xstate + * context to locate FP_XSTATE_MAGIC2. + * + * In 64-bit signal frames, the fpstate pointer points directly to the xstate + * context. In 32-bit signal frames (including 32-bit compat tasks on 64-bit + * kernels), the fpstate pointer points to struct _fpstate_32, which contains + * the 112-byte legacy FPU state followed by the 512-byte FXSR state (and any + * extended xstate), so the xstate context starts at fpstate + 112. + * + * This makes the signal frame self-describing and portable across machines + * with different xstate features. See Documentation/arch/x86/xstate.rst + * for details on signal frame portability and its architectural constraints. + * * This extended area typically grows with newer CPUs that have larger and * larger XSAVE areas. */ diff --git a/arch/x86/kernel/asm-offsets.c b/arch/x86/kernel/asm-offsets.c index 081816888f7a..b3c00ff4d819 100644 --- a/arch/x86/kernel/asm-offsets.c +++ b/arch/x86/kernel/asm-offsets.c @@ -95,6 +95,7 @@ static void __used common(void) OFFSET(TDX_MODULE_rbx, tdx_module_args, rbx); OFFSET(TDX_MODULE_rdi, tdx_module_args, rdi); OFFSET(TDX_MODULE_rsi, tdx_module_args, rsi); + OFFSET(TDX_MODULE_version, tdx_module_args, version); BLANK(); OFFSET(BP_scratch, boot_params, scratch); diff --git a/arch/x86/kernel/cpu/amd.c b/arch/x86/kernel/cpu/amd.c index 54e14ed276b5..e5279bc648d3 100644 --- a/arch/x86/kernel/cpu/amd.c +++ b/arch/x86/kernel/cpu/amd.c @@ -1192,7 +1192,7 @@ static unsigned int amd_size_cache(struct cpuinfo_x86 *c, unsigned int size) static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) { - u32 ebx, eax, ecx, edx; + u32 ebx, eax, ecx, edx, shift, tmp; u16 mask = 0xfff; if (c->x86 < 0xf) @@ -1201,10 +1201,12 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) if (c->extended_cpuid_level < 0x80000006) return; + shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5; + cpuid(0x80000006, &eax, &ebx, &ecx, &edx); - tlb_lld_4k = (ebx >> 16) & mask; - tlb_lli_4k = ebx & mask; + tlb_lld_4k = ((ebx >> 16) & mask) << shift; + tlb_lli_4k = (ebx & mask) << shift; /* * K8 doesn't have 2M/4M entries in the L2 TLB so read out the L1 TLB @@ -1216,16 +1218,18 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) } /* Handle DTLB 2M and 4M sizes, fall back to L1 if L2 is disabled */ - if (!((eax >> 16) & mask)) + tmp = ((eax >> 16) & mask) << shift; + if (!tmp) tlb_lld_2m = (cpuid_eax(0x80000005) >> 16) & 0xff; else - tlb_lld_2m = (eax >> 16) & mask; + tlb_lld_2m = tmp; /* a 4M entry uses two 2M entries */ tlb_lld_4m = tlb_lld_2m >> 1; /* Handle ITLB 2M and 4M sizes, fall back to L1 if L2 is disabled */ - if (!(eax & mask)) { + tmp = (eax & mask) << shift; + if (!tmp) { /* Erratum 658 */ if (c->x86 == 0x15 && c->x86_model <= 0x1f) { tlb_lli_2m = 1024; @@ -1233,8 +1237,9 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) cpuid(0x80000005, &eax, &ebx, &ecx, &edx); tlb_lli_2m = eax & 0xff; } - } else - tlb_lli_2m = eax & mask; + } else { + tlb_lli_2m = tmp; + } tlb_lli_4m = tlb_lli_2m >> 1; diff --git a/arch/x86/kernel/cpu/bugs.c b/arch/x86/kernel/cpu/bugs.c index 56eac5611c31..1b2381da4d83 100644 --- a/arch/x86/kernel/cpu/bugs.c +++ b/arch/x86/kernel/cpu/bugs.c @@ -1175,6 +1175,7 @@ enum srso_mitigation { SRSO_MITIGATION_IBPB, SRSO_MITIGATION_IBPB_ON_VMEXIT, SRSO_MITIGATION_BP_SPEC_REDUCE, + SRSO_MITIGATION_USER_IBPB, }; static enum srso_mitigation srso_mitigation __ro_after_init = SRSO_MITIGATION_AUTO; @@ -2908,7 +2909,8 @@ static const char * const srso_strings[] = { [SRSO_MITIGATION_SAFE_RET] = "Mitigation: Safe RET", [SRSO_MITIGATION_IBPB] = "Mitigation: IBPB", [SRSO_MITIGATION_IBPB_ON_VMEXIT] = "Mitigation: IBPB on VMEXIT only", - [SRSO_MITIGATION_BP_SPEC_REDUCE] = "Mitigation: Reduced Speculation" + [SRSO_MITIGATION_BP_SPEC_REDUCE] = "Mitigation: Reduced Speculation", + [SRSO_MITIGATION_USER_IBPB] = "Mitigation: IBPB on context switch", }; static int __init srso_parse_cmdline(char *str) @@ -2948,7 +2950,9 @@ static void __init srso_select_mitigation(void) * required. Otherwise the 'microcode' mitigation is sufficient * to protect the user->user and guest->guest vectors. */ - if (cpu_attack_vector_mitigated(CPU_MITIGATE_GUEST_HOST) || + if ((cpu_attack_vector_mitigated(CPU_MITIGATE_GUEST_HOST) && + !boot_cpu_has(X86_FEATURE_BTB_CTX_ISOLATION)) + || (cpu_attack_vector_mitigated(CPU_MITIGATE_USER_KERNEL) && !boot_cpu_has(X86_FEATURE_SRSO_USER_KERNEL_NO))) { srso_mitigation = SRSO_MITIGATION_SAFE_RET; @@ -3024,6 +3028,16 @@ static void __init srso_update_mitigation(void) boot_cpu_has(X86_FEATURE_IBPB_BRTYPE)) srso_mitigation = SRSO_MITIGATION_IBPB; + /* + * See if IBPB on context switch is the only thing needed to address + * GUEST/GUEST and USER/USER vectors. + */ + if (srso_mitigation == SRSO_MITIGATION_MICROCODE && + boot_cpu_has(X86_FEATURE_SRSO_USER_KERNEL_NO) && + boot_cpu_has(X86_FEATURE_BTB_CTX_ISOLATION) && + spectre_v2_user_ibpb != SPECTRE_V2_USER_NONE) + srso_mitigation = SRSO_MITIGATION_USER_IBPB; + pr_info("%s\n", srso_strings[srso_mitigation]); } diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c index c7352827f491..7d0b9bdc64cc 100644 --- a/arch/x86/kernel/cpu/common.c +++ b/arch/x86/kernel/cpu/common.c @@ -857,7 +857,7 @@ static void get_model_name(struct cpuinfo_x86 *c) void cpu_detect_cache_sizes(struct cpuinfo_x86 *c) { - unsigned int n, dummy, ebx, ecx, edx, l2size; + unsigned int n, dummy, ebx, ecx, edx, l2size, shift __maybe_unused; n = c->extended_cpuid_level; @@ -877,7 +877,9 @@ void cpu_detect_cache_sizes(struct cpuinfo_x86 *c) l2size = ecx >> 16; #ifdef CONFIG_X86_64 + shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5; c->x86_tlbsize += ((ebx >> 16) & 0xfff) + (ebx & 0xfff); + c->x86_tlbsize <<= shift; #else /* do processor-specific cache resizing */ if (this_cpu->legacy_cache_size) @@ -1427,7 +1429,7 @@ static bool __init vulnerable_to_its(u64 x86_arch_cap_msr) return false; } -static struct x86_cpu_id cpu_latest_microcode[] = { +static const struct x86_cpu_id cpu_latest_microcode[] __initconst = { #include "microcode/intel-ucode-defs.h" {} }; @@ -1782,25 +1784,50 @@ static void __init cpu_parse_early_param(void) } } +static void init_cpu_info(struct cpuinfo_x86 *c) +{ + c->x86_cache_size = 0; + c->x86_vendor = X86_VENDOR_UNKNOWN; + c->x86_model = c->x86_stepping = 0; /* So far unknown... */ + c->x86_vendor_id[0] = '\0'; /* Unset */ + c->x86_model_id[0] = '\0'; /* Unset */ +#ifdef CONFIG_X86_64 + c->x86_clflush_size = 64; + c->x86_phys_bits = 36; + c->x86_virt_bits = 48; +#else + c->cpuid_level = -1; /* CPUID not detected */ + c->x86_clflush_size = 32; + c->x86_phys_bits = 32; + c->x86_virt_bits = 32; +#endif + c->x86_cache_alignment = c->x86_clflush_size; + memset(&c->x86_capability, 0, sizeof(c->x86_capability)); + memset(&c->cpuid, 0, sizeof(c->cpuid)); +#ifdef CONFIG_X86_VMX_FEATURE_NAMES + memset(&c->vmx_capability, 0, sizeof(c->vmx_capability)); +#endif + c->extended_cpuid_level = 0; +} + /* * Do minimum CPU detection early. * Fields really needed: vendor, cpuid_level, family, model, mask, * cache alignment. - * The others are not touched to avoid unwanted side effects. + * The others are reset to their defaults here and only filled in later, + * by identify_cpu(). * * WARNING: this function is only called on the boot CPU. Don't add code * here that is supposed to run on all CPUs. */ static void __init early_identify_cpu(struct cpuinfo_x86 *c) { - memset(&c->x86_capability, 0, sizeof(c->x86_capability)); - memset(&c->cpuid, 0, sizeof(c->cpuid)); - c->extended_cpuid_level = 0; + init_cpu_info(c); if (!cpuid_feature()) identify_cpu_without_cpuid(c); - /* cyrix could have cpuid enabled via c_identify()*/ + /* Cyrix could have CPUID enabled via c_identify(). */ if (cpuid_feature()) { cpuid_scan_cpu(c); cpu_detect(c); @@ -1964,16 +1991,21 @@ void check_null_seg_clears_base(struct cpuinfo_x86 *c) set_cpu_bug(c, X86_BUG_NULL_SEG); } -static void generic_identify(struct cpuinfo_x86 *c) +/* + * This does the hard work of actually picking apart the CPU stuff... + */ +static void identify_cpu(struct cpuinfo_x86 *c) { - c->extended_cpuid_level = 0; + int i; + + c->loops_per_jiffy = loops_per_jiffy; if (!cpuid_feature()) identify_cpu_without_cpuid(c); - /* cyrix could have cpuid enabled via c_identify()*/ + /* Cyrix could have CPUID enabled via c_identify(). */ if (!cpuid_feature()) - return; + goto no_cpuid; cpuid_scan_cpu(c); cpu_detect(c); @@ -2001,40 +2033,8 @@ static void generic_identify(struct cpuinfo_x86 *c) #ifdef CONFIG_X86_32 set_cpu_bug(c, X86_BUG_ESPFIX); #endif -} - -/* - * This does the hard work of actually picking apart the CPU stuff... - */ -static void identify_cpu(struct cpuinfo_x86 *c) -{ - int i; - - c->loops_per_jiffy = loops_per_jiffy; - c->x86_cache_size = 0; - c->x86_vendor = X86_VENDOR_UNKNOWN; - c->x86_model = c->x86_stepping = 0; /* So far unknown... */ - c->x86_vendor_id[0] = '\0'; /* Unset */ - c->x86_model_id[0] = '\0'; /* Unset */ -#ifdef CONFIG_X86_64 - c->x86_clflush_size = 64; - c->x86_phys_bits = 36; - c->x86_virt_bits = 48; -#else - c->cpuid_level = -1; /* CPUID not detected */ - c->x86_clflush_size = 32; - c->x86_phys_bits = 32; - c->x86_virt_bits = 32; -#endif - c->x86_cache_alignment = c->x86_clflush_size; - memset(&c->x86_capability, 0, sizeof(c->x86_capability)); - memset(&c->cpuid, 0, sizeof(c->cpuid)); -#ifdef CONFIG_X86_VMX_FEATURE_NAMES - memset(&c->vmx_capability, 0, sizeof(c->vmx_capability)); -#endif - - generic_identify(c); +no_cpuid: cpu_parse_topology(c); if (this_cpu->c_identify) @@ -2126,6 +2126,9 @@ static void identify_cpu(struct cpuinfo_x86 *c) mcheck_cpu_init(c); numa_add_cpu(smp_processor_id()); + + if (IS_ENABLED(CONFIG_X86_32)) + enable_sep_cpu(); } /* @@ -2163,9 +2166,6 @@ static __init void identify_boot_cpu(void) identify_cpu(&boot_cpu_data); if (HAS_KERNEL_IBT && cpu_feature_enabled(X86_FEATURE_IBT)) pr_info("CET detected: Indirect Branch Tracking enabled\n"); -#ifdef CONFIG_X86_32 - enable_sep_cpu(); -#endif cpu_detect_tlb(&boot_cpu_data); setup_cr_pinning(); @@ -2184,10 +2184,8 @@ void identify_secondary_cpu(unsigned int cpu) *c = boot_cpu_data; c->cpu_index = cpu; + init_cpu_info(c); identify_cpu(c); -#ifdef CONFIG_X86_32 - enable_sep_cpu(); -#endif x86_spec_ctrl_setup_ap(); update_srbds_msr(); if (boot_cpu_has_bug(X86_BUG_GDS)) diff --git a/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h b/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h index af8b1d8b26f6..7f463c85c5f8 100644 --- a/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h +++ b/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h @@ -136,35 +136,35 @@ { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x5f, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x3e }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x66, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x2a }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0xc0002f0 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0xd000410 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6c, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x10002e0 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0xd000421 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6c, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x10002f1 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7a, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x42 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7a, .steppings = 0x0100, .platform_mask = 0x01, .driver_data = 0x26 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7e, .steppings = 0x0020, .platform_mask = 0x80, .driver_data = 0xca }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7e, .steppings = 0x0020, .platform_mask = 0x80, .driver_data = 0xcc }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8a, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x33 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0xbc }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0004, .platform_mask = 0xc2, .driver_data = 0x3c }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8d, .steppings = 0x0002, .platform_mask = 0xc2, .driver_data = 0x56 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0xbe }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0004, .platform_mask = 0xc2, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8d, .steppings = 0x0002, .platform_mask = 0xc2, .driver_data = 0x58 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0200, .platform_mask = 0x10, .driver_data = 0xf6 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0200, .platform_mask = 0xc0, .driver_data = 0xf6 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0400, .platform_mask = 0xc0, .driver_data = 0xf6 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0800, .platform_mask = 0xd0, .driver_data = 0xf6 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x1000, .platform_mask = 0x94, .driver_data = 0x100 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x10, .driver_data = 0x2c000410 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x87, .driver_data = 0x2b000650 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x10, .driver_data = 0x2c000410 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0x2b000650 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x10, .driver_data = 0x2c000410 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0x2b000650 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0080, .platform_mask = 0x87, .driver_data = 0x2b000650 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x10, .driver_data = 0x2c000410 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x87, .driver_data = 0x2b000650 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x10, .driver_data = 0x2c000421 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x87, .driver_data = 0x2b000670 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x10, .driver_data = 0x2c000421 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0x2b000670 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x10, .driver_data = 0x2c000421 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0x2b000670 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0080, .platform_mask = 0x87, .driver_data = 0x2b000670 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x10, .driver_data = 0x2c000421 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x87, .driver_data = 0x2b000670 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x96, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x1a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x43a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x40, .driver_data = 0xb }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x80, .driver_data = 0x43a }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x43b }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x40, .driver_data = 0xc }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x80, .driver_data = 0x43b }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9c, .steppings = 0x0001, .platform_mask = 0x01, .driver_data = 0x24000026 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9e, .steppings = 0x0200, .platform_mask = 0x2a, .driver_data = 0xf8 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9e, .steppings = 0x0400, .platform_mask = 0x22, .driver_data = 0xfa }, @@ -176,30 +176,33 @@ { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa5, .steppings = 0x0020, .platform_mask = 0x22, .driver_data = 0x100 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa6, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0x102 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa6, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x100 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa7, .steppings = 0x0002, .platform_mask = 0x02, .driver_data = 0x64 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaa, .steppings = 0x0010, .platform_mask = 0xe6, .driver_data = 0x25 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x20, .driver_data = 0xa000124 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x95, .driver_data = 0x10003f0 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xae, .steppings = 0x0002, .platform_mask = 0x97, .driver_data = 0x1000273 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaf, .steppings = 0x0008, .platform_mask = 0x01, .driver_data = 0x3000382 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb5, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0xa }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0002, .platform_mask = 0x32, .driver_data = 0x132 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0010, .platform_mask = 0x32, .driver_data = 0x132 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0004, .platform_mask = 0xe0, .driver_data = 0x6133 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0008, .platform_mask = 0xe0, .driver_data = 0x6133 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0100, .platform_mask = 0xe0, .driver_data = 0x6133 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbd, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x125 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbe, .steppings = 0x0001, .platform_mask = 0x19, .driver_data = 0x1e }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0040, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0080, .platform_mask = 0x07, .driver_data = 0x3d }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc5, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0010, .platform_mask = 0x82, .driver_data = 0x11a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xca, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0002, .platform_mask = 0x87, .driver_data = 0x210002c0 }, -{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0004, .platform_mask = 0x87, .driver_data = 0x210002c0 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa7, .steppings = 0x0002, .platform_mask = 0x02, .driver_data = 0x65 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaa, .steppings = 0x0010, .platform_mask = 0xe6, .driver_data = 0x28 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x20, .driver_data = 0xa000142 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x95, .driver_data = 0x1000423 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xae, .steppings = 0x0002, .platform_mask = 0x97, .driver_data = 0x1000307 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaf, .steppings = 0x0008, .platform_mask = 0x01, .driver_data = 0x30003a3 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb5, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0xd }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0002, .platform_mask = 0x32, .driver_data = 0x133 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0010, .platform_mask = 0x32, .driver_data = 0x133 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0004, .platform_mask = 0xe0, .driver_data = 0x6134 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0008, .platform_mask = 0xe0, .driver_data = 0x6134 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0100, .platform_mask = 0xe0, .driver_data = 0x6134 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbd, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x126 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbe, .steppings = 0x0001, .platform_mask = 0x19, .driver_data = 0x21 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0040, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0080, .platform_mask = 0x07, .driver_data = 0x3e }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc5, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0010, .platform_mask = 0x82, .driver_data = 0x121 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xca, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0002, .platform_mask = 0x90, .driver_data = 0x11b }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0004, .platform_mask = 0x90, .driver_data = 0x11b }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0008, .platform_mask = 0x90, .driver_data = 0x11b }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0002, .platform_mask = 0x87, .driver_data = 0x210002e0 }, +{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0004, .platform_mask = 0x87, .driver_data = 0x210002e0 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0080, .platform_mask = 0x01, .driver_data = 0x12 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0080, .platform_mask = 0x02, .driver_data = 0x8 }, { .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0400, .platform_mask = 0x01, .driver_data = 0x13 }, diff --git a/arch/x86/kernel/cpu/mtrr/amd.c b/arch/x86/kernel/cpu/mtrr/amd.c index a73715d6f05c..9e440e30c179 100644 --- a/arch/x86/kernel/cpu/mtrr/amd.c +++ b/arch/x86/kernel/cpu/mtrr/amd.c @@ -51,11 +51,10 @@ amd_get_mtrr(unsigned int reg, unsigned long *base, /** * amd_set_mtrr - Set variable MTRR register on the local CPU. - * - * @reg The register to set. - * @base The base address of the region. - * @size The size of the region. If this is 0 the region is disabled. - * @type The type of the region. + * @reg: The register to set. + * @base: The base address of the region. + * @size: The size of the region. If this is 0 the region is disabled. + * @type: The type of the region. * * Returns nothing. */ diff --git a/arch/x86/kernel/cpu/resctrl/ctrlmondata.c b/arch/x86/kernel/cpu/resctrl/ctrlmondata.c index e74f1ed54b86..62044489b052 100644 --- a/arch/x86/kernel/cpu/resctrl/ctrlmondata.c +++ b/arch/x86/kernel/cpu/resctrl/ctrlmondata.c @@ -16,9 +16,15 @@ #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt #include <linux/cpu.h> +#include <linux/math.h> #include "internal.h" +u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val) +{ + return roundup(val, (unsigned long)r->membw.bw_gran); +} + int resctrl_arch_update_one(struct rdt_resource *r, struct rdt_ctrl_domain *d, u32 closid, enum resctrl_conf_type t, u32 cfg_val) { diff --git a/arch/x86/kernel/cpu/scattered.c b/arch/x86/kernel/cpu/scattered.c index 8665a6474806..580b252e07cc 100644 --- a/arch/x86/kernel/cpu/scattered.c +++ b/arch/x86/kernel/cpu/scattered.c @@ -64,9 +64,11 @@ static const struct cpuid_bit cpuid_bits[] = { { X86_FEATURE_AMD_WORKLOAD_CLASS, CPUID_EAX, 22, 0x80000021, 0 }, { X86_FEATURE_TSA_SQ_NO, CPUID_ECX, 1, 0x80000021, 0 }, { X86_FEATURE_TSA_L1_NO, CPUID_ECX, 2, 0x80000021, 0 }, + { X86_FEATURE_BTB_CTX_ISOLATION, CPUID_ECX, 8, 0x80000021, 0 }, { X86_FEATURE_PERFMON_V2, CPUID_EAX, 0, 0x80000022, 0 }, { X86_FEATURE_AMD_LBR_V2, CPUID_EAX, 1, 0x80000022, 0 }, { X86_FEATURE_AMD_LBR_PMC_FREEZE, CPUID_EAX, 2, 0x80000022, 0 }, + { X86_FEATURE_RMPOPT, CPUID_EDX, 0, 0x80000025, 0 }, { X86_FEATURE_AMD_HTR_CORES, CPUID_EAX, 30, 0x80000026, 0 }, { 0, 0, 0, 0, 0 } }; diff --git a/arch/x86/kernel/cpu/sgx/main.c b/arch/x86/kernel/cpu/sgx/main.c index 4505f808af5e..a5f2aabb2da1 100644 --- a/arch/x86/kernel/cpu/sgx/main.c +++ b/arch/x86/kernel/cpu/sgx/main.c @@ -106,7 +106,13 @@ static unsigned long __sgx_sanitize_pages(struct list_head *dirty_page_list) left_dirty++; } - cond_resched(); + /* + * cond_resched() only schedules when TIF_NEED_RESCHED is set. + * During this boot-time loop that condition may not happen for a + * long time, so report an RCU-Tasks quiescent state explicitly. + * Therefore, change cond_resched() to cond_resched_tasks_rcu_qs(). + */ + cond_resched_tasks_rcu_qs(); } list_splice(&dirty, dirty_page_list); diff --git a/arch/x86/kernel/cpu/vmware.c b/arch/x86/kernel/cpu/vmware.c index 34b73573b108..f7ab9e7902cf 100644 --- a/arch/x86/kernel/cpu/vmware.c +++ b/arch/x86/kernel/cpu/vmware.c @@ -328,9 +328,9 @@ static int vmware_cpu_down_prepare(unsigned int cpu) static __init int activate_jump_labels(void) { if (has_steal_clock) { - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); } return 0; diff --git a/arch/x86/kernel/crash.c b/arch/x86/kernel/crash.c index e681ec9cf1dc..e6f23933a6df 100644 --- a/arch/x86/kernel/crash.c +++ b/arch/x86/kernel/crash.c @@ -369,9 +369,9 @@ int crash_load_segments(struct kimage *image) * maximum CPUs and maximum memory ranges. */ if (IS_ENABLED(CONFIG_MEMORY_HOTPLUG)) - pnum = 2 + CONFIG_NR_CPUS_DEFAULT + CONFIG_CRASH_MAX_MEMORY_RANGES; + pnum = 2 + CONFIG_NR_CPUS + CONFIG_CRASH_MAX_MEMORY_RANGES; else - pnum += 2 + CONFIG_NR_CPUS_DEFAULT; + pnum += 2 + CONFIG_NR_CPUS; if (pnum < (unsigned long)PN_XNUM) { kbuf.memsz = pnum * sizeof(Elf64_Phdr); @@ -430,7 +430,7 @@ unsigned int arch_crash_get_elfcorehdr_size(void) unsigned int sz; /* kernel_map, VMCOREINFO and maximum CPUs */ - sz = 2 + CONFIG_NR_CPUS_DEFAULT; + sz = 2 + CONFIG_NR_CPUS; if (IS_ENABLED(CONFIG_MEMORY_HOTPLUG)) sz += CONFIG_CRASH_MAX_MEMORY_RANGES; sz *= sizeof(Elf64_Phdr); diff --git a/arch/x86/kernel/fpu/bugs.c b/arch/x86/kernel/fpu/bugs.c index edbafc5940e3..4b84cd0a9d24 100644 --- a/arch/x86/kernel/fpu/bugs.c +++ b/arch/x86/kernel/fpu/bugs.c @@ -29,10 +29,6 @@ void __init fpu__init_check_bugs(void) { s32 fdiv_bug; - /* kernel_fpu_begin/end() relies on patched alternative instructions. */ - if (!boot_cpu_has(X86_FEATURE_FPU)) - return; - kernel_fpu_begin(); /* diff --git a/arch/x86/kernel/fpu/core.c b/arch/x86/kernel/fpu/core.c index d1aeecd57f5e..80a7f41cf3ad 100644 --- a/arch/x86/kernel/fpu/core.c +++ b/arch/x86/kernel/fpu/core.c @@ -213,6 +213,19 @@ void restore_fpregs_from_fpstate(struct fpstate *fpstate, u64 mask) } } +/* + * Save the FPU register state in fpu->fpstate->regs and set + * TIF_NEED_FPU_LOAD subsequently. + * + * Must be called with fpregs_lock() held, ensuring flag + * TIF_NEED_FPU_LOAD is set last. + */ +void update_fpu_state_and_flag(struct fpu *fpu, struct task_struct *task) +{ + save_fpregs_to_fpstate(fpu); + set_tsk_thread_flag(task, TIF_NEED_FPU_LOAD); +} + void fpu_reset_from_exception_fixup(void) { restore_fpregs_from_fpstate(&init_fpstate, XFEATURE_MASK_FPSTATE); @@ -383,13 +396,13 @@ int fpu_swap_kvm_fpstate(struct fpu_guest *guest_fpu, bool enter_guest) /* Swap fpstate */ if (enter_guest) { - fpu->__task_fpstate = cur_fps; + WRITE_ONCE(fpu->__task_fpstate, cur_fps); + barrier(); fpu->fpstate = guest_fps; guest_fps->in_use = true; } else { guest_fps->in_use = false; fpu->fpstate = fpu->__task_fpstate; - fpu->__task_fpstate = NULL; } cur_fps = fpu->fpstate; @@ -406,6 +419,16 @@ int fpu_swap_kvm_fpstate(struct fpu_guest *guest_fpu, bool enter_guest) xfd_update_state(cur_fps); } + /* + * Clear fpu->__task_fpstate after switching back to host state. + * A non-NULL __task_fpstate means guest state is still resident in + * hardware; reset it only once host state has been restored. + */ + if (!enter_guest) { + barrier(); + WRITE_ONCE(fpu->__task_fpstate, NULL); + } + fpregs_mark_activate(); fpregs_unlock(); return 0; @@ -481,17 +504,15 @@ void kernel_fpu_begin_mask(unsigned int kfpu_mask) this_cpu_write(kernel_fpu_allowed, false); if (!(current->flags & (PF_KTHREAD | PF_USER_WORKER)) && - !test_thread_flag(TIF_NEED_FPU_LOAD)) { - set_thread_flag(TIF_NEED_FPU_LOAD); - save_fpregs_to_fpstate(x86_task_fpu(current)); - } + !test_thread_flag(TIF_NEED_FPU_LOAD)) + update_fpu_state_and_flag(x86_task_fpu(current), current); __cpu_invalidate_fpregs_state(); /* Put sane initial values into the control registers. */ if (likely(kfpu_mask & KFPU_MXCSR) && boot_cpu_has(X86_FEATURE_XMM)) ldmxcsr(MXCSR_DEFAULT); - if (unlikely(kfpu_mask & KFPU_387) && boot_cpu_has(X86_FEATURE_FPU)) + if (unlikely(kfpu_mask & KFPU_387)) asm volatile ("fninit"); } EXPORT_SYMBOL_GPL(kernel_fpu_begin_mask); @@ -672,9 +693,6 @@ int fpu_clone(struct task_struct *dst, u64 clone_flags, bool minimal, fpstate_reset(dst_fpu); - if (!cpu_feature_enabled(X86_FEATURE_FPU)) - return 0; - /* * Enforce reload for user space tasks and prevent kernel threads * from trying to save the FPU registers on context switch. @@ -833,11 +851,6 @@ void fpu__clear_user_states(struct fpu *fpu) WARN_ON_FPU(fpu != x86_task_fpu(current)); fpregs_lock(); - if (!cpu_feature_enabled(X86_FEATURE_FPU)) { - fpu_reset_fpstate_regs(); - fpregs_unlock(); - return; - } /* * Ensure that current's supervisor states are loaded into their @@ -874,9 +887,6 @@ void fpu_flush_thread(void) */ void switch_fpu_return(void) { - if (!cpu_feature_enabled(X86_FEATURE_FPU)) - return; - fpregs_restore_userregs(); } EXPORT_SYMBOL_FOR_KVM(switch_fpu_return); diff --git a/arch/x86/kernel/fpu/init.c b/arch/x86/kernel/fpu/init.c index 0d33c217b71c..1a92ec433f9d 100644 --- a/arch/x86/kernel/fpu/init.c +++ b/arch/x86/kernel/fpu/init.c @@ -31,8 +31,6 @@ static void fpu__init_cpu_generic(void) cr0 = read_cr0(); cr0 &= ~(X86_CR0_TS|X86_CR0_EM); /* clear TS and EM */ - if (!boot_cpu_has(X86_FEATURE_FPU)) - cr0 |= X86_CR0_EM; write_cr0(cr0); /* Flush out any pending x87 state: */ @@ -184,9 +182,7 @@ static void __init fpu__init_system_xstate_size_legacy(void) * Note that the size configuration might be overwritten later * during fpu__init_system_xstate(). */ - if (!cpu_feature_enabled(X86_FEATURE_FPU)) { - size = sizeof(struct swregs_state); - } else if (cpu_feature_enabled(X86_FEATURE_FXSR)) { + if (cpu_feature_enabled(X86_FEATURE_FXSR)) { size = sizeof(struct fxregs_state); fpu_user_cfg.legacy_features = XFEATURE_MASK_FPSSE; } else { diff --git a/arch/x86/kernel/fpu/regset.c b/arch/x86/kernel/fpu/regset.c index 0986c2200adc..f96affd834a1 100644 --- a/arch/x86/kernel/fpu/regset.c +++ b/arch/x86/kernel/fpu/regset.c @@ -407,9 +407,6 @@ int fpregs_get(struct task_struct *target, const struct user_regset *regset, sync_fpstate(fpu); - if (!cpu_feature_enabled(X86_FEATURE_FPU)) - return fpregs_soft_get(target, regset, to); - if (!cpu_feature_enabled(X86_FEATURE_FXSR)) { return membuf_write(&to, &fpu->fpstate->regs.fsave, sizeof(struct fregs_state)); @@ -441,9 +438,6 @@ int fpregs_set(struct task_struct *target, const struct user_regset *regset, if (pos != 0 || count != sizeof(struct user_i387_ia32_struct)) return -EINVAL; - if (!cpu_feature_enabled(X86_FEATURE_FPU)) - return fpregs_soft_set(target, regset, pos, count, kbuf, ubuf); - ret = user_regset_copyin(&pos, &count, &kbuf, &ubuf, &env, 0, -1); if (ret) return ret; diff --git a/arch/x86/kernel/fpu/signal.c b/arch/x86/kernel/fpu/signal.c index 33e1284bf3e4..1f721ac84283 100644 --- a/arch/x86/kernel/fpu/signal.c +++ b/arch/x86/kernel/fpu/signal.c @@ -24,23 +24,24 @@ * Check for the presence of extended state information in the * user fpstate pointer in the sigcontext. */ -static inline bool check_xstate_in_sigframe(struct fxregs_state __user *fxbuf, +static inline bool check_xstate_in_sigframe(struct fxregs_state __user *buf_fx, struct _fpx_sw_bytes *fx_sw) { + struct fpstate *fpstate = x86_task_fpu(current)->fpstate; int min_xstate_size = sizeof(struct fxregs_state) + sizeof(struct xstate_header); - void __user *fpstate = fxbuf; + void __user *buf = buf_fx; unsigned int magic2; - if (__copy_from_user(fx_sw, &fxbuf->sw_reserved[0], sizeof(*fx_sw))) + if (__copy_from_user(fx_sw, &buf_fx->sw_reserved[0], sizeof(*fx_sw))) return false; /* Check for the first magic field and other error scenarios. */ if (fx_sw->magic1 != FP_XSTATE_MAGIC1 || fx_sw->xstate_size < min_xstate_size || - fx_sw->xstate_size > x86_task_fpu(current)->fpstate->user_size || - fx_sw->xstate_size > fx_sw->extended_size) - goto setfx; + fx_sw->xstate_size > fpstate->user_size || + fx_sw->extended_size < fx_sw->xstate_size + FP_XSTATE_MAGIC2_SIZE) + goto err_setfx; /* * Check for the presence of second magic word at the end of memory @@ -48,12 +49,36 @@ static inline bool check_xstate_in_sigframe(struct fxregs_state __user *fxbuf, * fpstate layout with out copying the extended state information * in the memory layout. */ - if (__get_user(magic2, (__u32 __user *)(fpstate + fx_sw->xstate_size))) + if (__get_user(magic2, (__u32 __user *)(buf + fx_sw->xstate_size))) return false; + if (unlikely(magic2 != FP_XSTATE_MAGIC2)) + goto err_setfx; - if (likely(magic2 == FP_XSTATE_MAGIC2)) - return true; -setfx: + if (fx_sw->xstate_size != fpstate->user_size || + fx_sw->xfeatures != fpstate->user_xfeatures) { + unsigned int xsize; + u64 xfeatures; + + /* Calculate size of enabled features only. */ + xfeatures = fx_sw->xfeatures & fpstate->user_xfeatures; + + xsize = xstate_calculate_size(xfeatures, false); + if (fx_sw->xstate_size < xsize) + return false; + + fx_sw->xstate_size = xsize; + } + + return true; +err_setfx: + /* + * The fallback to FX-only state is used to preserve backward + * compatibility with user-space processes that are not aware of xsave + * states. + * + * In all other cases, returning false (to trigger SIGSEGV) is + * preferred to avoid silent user-space state corruption. + */ trace_x86_fpu_xstate_check_failed(x86_task_fpu(current)); /* Set the parameters for fx only state */ @@ -187,14 +212,6 @@ bool copy_fpstate_to_sigframe(void __user *buf, void __user *buf_fx, int size, u ia32_fxstate &= (IS_ENABLED(CONFIG_X86_32) || IS_ENABLED(CONFIG_IA32_EMULATION)); - if (!cpu_feature_enabled(X86_FEATURE_FPU)) { - struct user_i387_ia32_struct fp; - - fpregs_soft_get(current, NULL, (struct membuf){.p = &fp, - .left = sizeof(fp)}); - return !copy_to_user(buf, &fp, sizeof(fp)); - } - if (!access_ok(buf, size)) return false; @@ -240,15 +257,18 @@ retry: return true; } -static int __restore_fpregs_from_user(void __user *buf, u64 ufeatures, - u64 xrestore, bool fx_only) +static int __restore_fpregs_from_user(void __user *buf, u64 task_xfeatures, + u64 xrestore_mask, bool fx_only) { if (use_xsave()) { - u64 init_bv = ufeatures & ~xrestore; + u64 init_bv; int ret; + /* Restore enabled features only. */ + xrestore_mask &= task_xfeatures; + init_bv = task_xfeatures & ~xrestore_mask; if (likely(!fx_only)) - ret = xrstor_from_user_sigframe(buf, xrestore); + ret = xrstor_from_user_sigframe(buf, xrestore_mask); else ret = fxrstor_from_user_sigframe(buf); @@ -266,20 +286,19 @@ static int __restore_fpregs_from_user(void __user *buf, u64 ufeatures, * Attempt to restore the FPU registers directly from user memory. * Pagefaults are handled and any errors returned are fatal. */ -static bool restore_fpregs_from_user(void __user *buf, u64 xrestore, bool fx_only) +static bool restore_fpregs_from_user(void __user *buf, u64 xrestore_mask, + bool fx_only, size_t xstate_size) { struct fpu *fpu = x86_task_fpu(current); int ret; - /* Restore enabled features only. */ - xrestore &= fpu->fpstate->user_xfeatures; retry: fpregs_lock(); /* Ensure that XFD is up to date */ xfd_update_state(fpu->fpstate); pagefault_disable(); ret = __restore_fpregs_from_user(buf, fpu->fpstate->user_xfeatures, - xrestore, fx_only); + xrestore_mask, fx_only); pagefault_enable(); if (unlikely(ret)) { @@ -302,7 +321,7 @@ retry: if (ret != X86_TRAP_PF) return false; - if (!fault_in_readable(buf, fpu->fpstate->user_size)) + if (!fault_in_readable(buf, xstate_size)) goto retry; return false; } @@ -324,39 +343,33 @@ retry: return true; } -static bool __fpu_restore_sig(void __user *buf, void __user *buf_fx, - bool ia32_fxstate) +/* + * Restore FPU state from a signal frame when a legacy 32-bit FP frame + * (buf_f) is present. + * + * The legacy FP frame duplicates the FP state portion of the FX/XSAVE + * frame (buf_fx). For backward compatibility, the legacy FP frame is + * treated as the source of truth, and its state is folded into the + * FX/XSAVE state before restoring the registers. + */ +static bool restore_from_ia32_fxstate(void __user *buf_f, void __user *buf_fx, + u64 xrestore_mask, bool fx_only) { struct task_struct *tsk = current; struct fpu *fpu = x86_task_fpu(tsk); struct user_i387_ia32_struct env; - bool success, fx_only = false; union fpregs_state *fpregs; - u64 user_xfeatures = 0; - - if (use_xsave()) { - struct _fpx_sw_bytes fx_sw_user; - - if (!check_xstate_in_sigframe(buf_fx, &fx_sw_user)) - return false; - - fx_only = !fx_sw_user.magic1; - user_xfeatures = fx_sw_user.xfeatures; - } else { - user_xfeatures = XFEATURE_MASK_FPSSE; - } + bool success; - if (likely(!ia32_fxstate)) { - /* Restore the FPU registers directly from user memory. */ - return restore_fpregs_from_user(buf_fx, user_xfeatures, fx_only); - } + if (!IS_ENABLED(CONFIG_X86_32) && !IS_ENABLED(CONFIG_IA32_EMULATION)) + return false; /* * Copy the legacy state because the FP portion of the FX frame has * to be ignored for histerical raisins. The legacy state is folded * in once the larger state has been copied. */ - if (__copy_from_user(&env, buf, sizeof(env))) + if (__copy_from_user(&env, buf_f, sizeof(env))) return false; /* @@ -420,7 +433,7 @@ static bool __fpu_restore_sig(void __user *buf, void __user *buf_fx, * * Preserve supervisor states! */ - u64 mask = user_xfeatures | xfeatures_mask_supervisor(); + u64 mask = xrestore_mask | xfeatures_mask_supervisor(); fpregs->xsave.header.xfeatures &= mask; success = !os_xrstor_safe(fpu->fpstate, @@ -449,10 +462,11 @@ static inline unsigned int xstate_sigframe_size(struct fpstate *fpstate) bool fpu__restore_sig(void __user *buf, int ia32_frame) { struct fpu *fpu = x86_task_fpu(current); - void __user *buf_fx = buf; + bool success = false, fx_only = false; bool ia32_fxstate = false; - bool success = false; + void __user *buf_fx = buf; unsigned int size; + u64 xrestore_mask; if (unlikely(!buf)) { fpu__clear_user_states(fpu); @@ -477,14 +491,24 @@ bool fpu__restore_sig(void __user *buf, int ia32_frame) if (!access_ok(buf, size)) goto out; - if (!IS_ENABLED(CONFIG_X86_64) && !cpu_feature_enabled(X86_FEATURE_FPU)) { - success = !fpregs_soft_set(current, NULL, 0, - sizeof(struct user_i387_ia32_struct), - NULL, buf); + if (use_xsave()) { + struct _fpx_sw_bytes fx_sw_user; + + if (!check_xstate_in_sigframe(buf_fx, &fx_sw_user)) + goto out; + + fx_only = !fx_sw_user.magic1; + xrestore_mask = fx_sw_user.xfeatures; + size = fx_sw_user.xstate_size; } else { - success = __fpu_restore_sig(buf, buf_fx, ia32_fxstate); + xrestore_mask = XFEATURE_MASK_FPSSE; + size = fpu->fpstate->user_size; } + if (ia32_fxstate) + success = restore_from_ia32_fxstate(buf, buf_fx, xrestore_mask, fx_only); + else + success = restore_fpregs_from_user(buf_fx, xrestore_mask, fx_only, size); out: if (unlikely(!success)) fpu__clear_user_states(fpu); diff --git a/arch/x86/kernel/fpu/xstate.c b/arch/x86/kernel/fpu/xstate.c index a7b6524a9dea..da303714379c 100644 --- a/arch/x86/kernel/fpu/xstate.c +++ b/arch/x86/kernel/fpu/xstate.c @@ -587,14 +587,15 @@ static bool __init check_xstate_against_struct(int nr) return true; } -static unsigned int xstate_calculate_size(u64 xfeatures, bool compacted) +unsigned int xstate_calculate_size(u64 xfeatures, bool compacted) { - unsigned int topmost = fls64(xfeatures) - 1; - unsigned int offset, i; + unsigned int topmost, offset, i; - if (topmost <= XFEATURE_SSE) + if (!(xfeatures & ~XFEATURE_MASK_FPSSE)) return sizeof(struct xregs_state); + topmost = fls64(xfeatures) - 1; + if (compacted) { offset = xfeature_get_offset(xfeatures, topmost); } else { @@ -806,18 +807,15 @@ static u64 __init guest_default_mask(void) void __init fpu__init_system_xstate(unsigned int legacy_size) { unsigned int eax, ebx, ecx, edx; - u64 xfeatures; + u64 xfeatures, mask; int err; int i; - if (!boot_cpu_has(X86_FEATURE_FPU)) { - pr_info("x86/fpu: No FPU detected\n"); - return; - } - if (!boot_cpu_has(X86_FEATURE_XSAVE)) { pr_info("x86/fpu: x87 FPU will use %s\n", boot_cpu_has(X86_FEATURE_FXSR) ? "FXSAVE" : "FSAVE"); + /* Disable all dependent flags too */ + setup_clear_cpu_cap(X86_FEATURE_XSAVE); return; } @@ -833,7 +831,8 @@ void __init fpu__init_system_xstate(unsigned int legacy_size) cpuid_count(CPUID_LEAF_XSTATE, 1, &eax, &ebx, &ecx, &edx); fpu_kernel_cfg.max_features |= ecx + ((u64)edx << 32); - if ((fpu_kernel_cfg.max_features & XFEATURE_MASK_FPSSE) != XFEATURE_MASK_FPSSE) { + mask = XFEATURE_MASK_FPSSE; + if ((fpu_kernel_cfg.max_features & mask) != mask) { /* * This indicates that something really unexpected happened * with the enumeration. Disable XSAVE and try to continue @@ -844,6 +843,24 @@ void __init fpu__init_system_xstate(unsigned int legacy_size) goto out_disable; } + mask |= XFEATURE_MASK_YMM; + if (boot_cpu_has(X86_FEATURE_AVX)) { + if ((fpu_kernel_cfg.max_features & mask) != mask) { + pr_err(FW_BUG + "x86/fpu: Disabling AVX support due to missing xstate features\n"); + setup_clear_cpu_cap(X86_FEATURE_AVX); + } + } + + mask |= XFEATURE_MASK_AVX512; + if (boot_cpu_has(X86_FEATURE_AVX512F)) { + if ((fpu_kernel_cfg.max_features & mask) != mask) { + pr_err(FW_BUG + "x86/fpu: Disabling AVX-512 support due to missing xstate features\n"); + setup_clear_cpu_cap(X86_FEATURE_AVX512F); + } + } + if (fpu_kernel_cfg.max_features & XFEATURE_MASK_APX && fpu_kernel_cfg.max_features & (XFEATURE_MASK_BNDREGS | XFEATURE_MASK_BNDCSR)) { /* @@ -1474,6 +1491,29 @@ void xrstors(struct xregs_state *xstate, u64 mask) WARN_ON_ONCE(err); } +/** + * xsaves_nmi - Save selected components to a kernel xstate buffer in NMI + * @xstate: Pointer to the buffer + * @mask: Feature mask to select the components to save + * + * This function is similar to xsaves(), but should only be called within + * the NMI handler. This function returns the actual register contents at + * the moment the NMI occurs. + * + * Currently, the perf subsystem is the sole user of this helper. It uses + * the function to snapshot SIMD (XMM/YMM/ZMM) and APX eGPRs registers. + */ +void xsaves_nmi(struct xregs_state *xstate, u64 mask) +{ + int err; + + if (!in_nmi()) + return; + + XSTATE_OP(XSAVES, xstate, (u32)mask, (u32)(mask >> 32), err); + WARN_ON_ONCE(err); +} + #if IS_ENABLED(CONFIG_KVM) void fpstate_clear_xstate_component(struct fpstate *fpstate, unsigned int xfeature) { diff --git a/arch/x86/kernel/kvm.c b/arch/x86/kernel/kvm.c index 6b0a5861ccb8..2ce01fdb8b4b 100644 --- a/arch/x86/kernel/kvm.c +++ b/arch/x86/kernel/kvm.c @@ -1054,9 +1054,9 @@ const __initconst struct hypervisor_x86 x86_hyper_kvm = { static __init int activate_jump_labels(void) { if (has_steal_clock) { - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (steal_acc) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); } return 0; diff --git a/arch/x86/kernel/perf_regs.c b/arch/x86/kernel/perf_regs.c index 624703af80a1..f8952f7c36cc 100644 --- a/arch/x86/kernel/perf_regs.c +++ b/arch/x86/kernel/perf_regs.c @@ -61,11 +61,29 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) { struct x86_perf_regs *perf_regs; - if (idx >= PERF_REG_X86_XMM0 && idx < PERF_REG_X86_XMM_MAX) { + if (idx > PERF_REG_X86_R15) { perf_regs = container_of(regs, struct x86_perf_regs, regs); - if (!perf_regs->xmm_regs) + if (perf_regs->abi == PERF_SAMPLE_REGS_ABI_NONE) return 0; - return perf_regs->xmm_regs[idx - PERF_REG_X86_XMM0]; + + if (perf_regs->abi & PERF_SAMPLE_REGS_ABI_SIMD) { + if (idx <= PERF_REG_X86_R31) { + if (!perf_regs->egpr_regs) + return 0; + return perf_regs->egpr_regs[idx - PERF_REG_X86_R16]; + } + if (idx == PERF_REG_X86_SSP) { + if (!perf_regs->ssp) + return 0; + return *perf_regs->ssp; + } + } else { + if (idx >= PERF_REG_X86_XMM0 && idx < PERF_REG_X86_XMM_MAX) { + if (!perf_regs->xmm_regs) + return 0; + return perf_regs->xmm_regs[idx - PERF_REG_X86_XMM0]; + } + } } if (WARN_ON_ONCE(idx >= ARRAY_SIZE(pt_regs_offset))) @@ -74,22 +92,130 @@ u64 perf_reg_value(struct pt_regs *regs, int idx) return regs_get_register(regs, pt_regs_offset[idx]); } -#define PERF_REG_X86_RESERVED (((1ULL << PERF_REG_X86_XMM0) - 1) & \ - ~((1ULL << PERF_REG_X86_MAX) - 1)) +#define PERF_X86_YMMH_QWORDS (PERF_X86_YMM_QWORDS / 2) +#define PERF_X86_ZMMH_QWORDS (PERF_X86_ZMM_QWORDS / 2) + +u64 perf_simd_reg_value(struct pt_regs *regs, int idx, + u16 qwords_idx, bool pred) +{ + struct x86_perf_regs *perf_regs = + container_of(regs, struct x86_perf_regs, regs); + + if (!(perf_regs->abi & PERF_SAMPLE_REGS_ABI_SIMD)) + return 0; + + if (pred) { + if (WARN_ON_ONCE(idx >= PERF_X86_SIMD_PRED_REGS_MAX || + qwords_idx >= PERF_X86_OPMASK_QWORDS)) + return 0; + if (!perf_regs->opmask_regs) + return 0; + return perf_regs->opmask_regs[idx]; + } + + if (WARN_ON_ONCE(idx >= PERF_X86_SIMD_VEC_REGS_MAX || + qwords_idx >= PERF_X86_SIMD_QWORDS_MAX)) + return 0; + + if (idx >= PERF_X86_H16ZMM_BASE) { + if (!perf_regs->h16zmm_regs) + return 0; + return perf_regs->h16zmm_regs[(idx - PERF_X86_H16ZMM_BASE) * + PERF_X86_ZMM_QWORDS + qwords_idx]; + } + + if (qwords_idx < PERF_X86_XMM_QWORDS) { + if (!perf_regs->xmm_regs) + return 0; + return perf_regs->xmm_regs[idx * PERF_X86_XMM_QWORDS + + qwords_idx]; + } else if (qwords_idx < PERF_X86_YMM_QWORDS) { + if (!perf_regs->ymmh_regs) + return 0; + return perf_regs->ymmh_regs[idx * PERF_X86_YMMH_QWORDS + + qwords_idx - PERF_X86_XMM_QWORDS]; + } else if (qwords_idx < PERF_X86_ZMM_QWORDS) { + if (!perf_regs->zmmh_regs) + return 0; + return perf_regs->zmmh_regs[idx * PERF_X86_ZMMH_QWORDS + + qwords_idx - PERF_X86_YMM_QWORDS]; + } + + return 0; +} + +int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask, + u16 pred_qwords, u32 pred_mask) +{ + unsigned long mask; + u64 size; + + if (!vec_qwords && !pred_qwords) { + if (vec_mask || pred_mask) + return -EINVAL; + } + + if (vec_qwords) { + if (vec_qwords != PERF_X86_XMM_QWORDS && + vec_qwords != PERF_X86_YMM_QWORDS && + vec_qwords != PERF_X86_ZMM_QWORDS) + return -EINVAL; + if (vec_mask & ~PERF_X86_SIMD_VEC_MASK) + return -EINVAL; + /* Only full-register sampling is allowed. */ + mask = vec_mask; + if (vec_qwords == PERF_X86_XMM_QWORDS && mask && + !bitmap_full(&mask, PERF_X86_SIMD_XMM_REGS)) + return -EINVAL; + if (vec_qwords == PERF_X86_YMM_QWORDS && mask && + !bitmap_full(&mask, PERF_X86_SIMD_YMM_REGS)) + return -EINVAL; + if (vec_qwords == PERF_X86_ZMM_QWORDS && mask && + !bitmap_full(&mask, PERF_X86_SIMD_ZMM_REGS)) + return -EINVAL; + } + + if (pred_qwords) { + if (pred_qwords != PERF_X86_OPMASK_QWORDS) + return -EINVAL; + if (pred_mask & ~PERF_X86_SIMD_PRED_MASK) + return -EINVAL; + /* Only full-register sampling is allowed. */ + mask = pred_mask; + if (pred_qwords == PERF_X86_OPMASK_QWORDS && mask && + !bitmap_full(&mask, PERF_X86_SIMD_OPMASK_REGS)) + return -EINVAL; + } + + size = sizeof(u64) * 4; + size += (hweight64(vec_mask) * vec_qwords + + hweight32(pred_mask) * pred_qwords) * sizeof(u64); + /* + * INTR_REGS and USR_REGS could be sampled simultaneously, + * so roughly restrict the size to half of U16_MAX. + */ + if (size >= U16_MAX / 2) + return -EINVAL; + + return 0; +} + +#define PERF_REG_X86_RESERVED (GENMASK_ULL(PERF_REG_X86_XMM0 - 1, PERF_REG_X86_AX) & \ + ~GENMASK_ULL(PERF_REG_X86_R15, PERF_REG_X86_AX)) +#define PERF_REG_X86_EXT_RESERVED (~GENMASK_ULL(PERF_REG_MISC_MAX - 1, PERF_REG_X86_AX)) #ifdef CONFIG_X86_32 -#define REG_NOSUPPORT ((1ULL << PERF_REG_X86_R8) | \ - (1ULL << PERF_REG_X86_R9) | \ - (1ULL << PERF_REG_X86_R10) | \ - (1ULL << PERF_REG_X86_R11) | \ - (1ULL << PERF_REG_X86_R12) | \ - (1ULL << PERF_REG_X86_R13) | \ - (1ULL << PERF_REG_X86_R14) | \ - (1ULL << PERF_REG_X86_R15)) - -int perf_reg_validate(u64 mask) +#define REG_NOSUPPORT GENMASK_ULL(PERF_REG_X86_R15, PERF_REG_X86_R8) + +int perf_reg_validate(u64 mask, bool simd_enabled) { - if (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED))) + if (!simd_enabled && + (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED)))) + return -EINVAL; + + /* The mask could be 0 if only the SIMD registers are interested */ + if (simd_enabled && + (mask & ~GENMASK_ULL(PERF_REG_X86_GS, PERF_REG_X86_AX))) return -EINVAL; return 0; @@ -100,21 +226,21 @@ u64 perf_reg_abi(struct task_struct *task) return PERF_SAMPLE_REGS_ABI_32; } -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} #else /* CONFIG_X86_64 */ #define REG_NOSUPPORT ((1ULL << PERF_REG_X86_DS) | \ (1ULL << PERF_REG_X86_ES) | \ (1ULL << PERF_REG_X86_FS) | \ (1ULL << PERF_REG_X86_GS)) -int perf_reg_validate(u64 mask) +int perf_reg_validate(u64 mask, bool simd_enabled) { - if (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED))) + if (!simd_enabled && + (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED)))) + return -EINVAL; + + /* The mask could be 0 if only the SIMD registers are interested */ + if (simd_enabled && + (mask & (REG_NOSUPPORT | PERF_REG_X86_EXT_RESERVED))) return -EINVAL; return 0; diff --git a/arch/x86/kernel/shstk.c b/arch/x86/kernel/shstk.c index 0ca64900192f..eb690ba90180 100644 --- a/arch/x86/kernel/shstk.c +++ b/arch/x86/kernel/shstk.c @@ -490,7 +490,7 @@ static int wrss_control(bool enable) * when disabling. */ if (!features_enabled(ARCH_SHSTK_SHSTK)) - return -EPERM; + return -EINVAL; /* Already enabled/disabled? */ if (features_enabled(ARCH_SHSTK_WRSS) == enable) diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index a0d608e3fceb..46c033d87d39 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -617,6 +617,10 @@ static int mmu_topup_memory_caches(struct kvm_vcpu *vcpu, bool maybe_indirect) PT64_ROOT_MAX_LEVEL); if (r) return r; + + r = kvm_x86_call(topup_external_cache)(vcpu, PT64_ROOT_MAX_LEVEL); + if (r) + return r; } r = kvm_mmu_topup_memory_cache(&vcpu->arch.mmu_shadow_page_cache, PT64_ROOT_MAX_LEVEL); diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index 0c1ebb16cec6..1fe26778db9e 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -3048,6 +3048,8 @@ void sev_vm_destroy(struct kvm *kvm) */ if (snp_decommission_context(kvm)) return; + + snp_enable_rmpopt(); } else { sev_unbind_asid(kvm, sev->handle); } diff --git a/arch/x86/kvm/vmx/pmu_intel.c b/arch/x86/kvm/vmx/pmu_intel.c index 70a8c4816135..fb7ad53855f1 100644 --- a/arch/x86/kvm/vmx/pmu_intel.c +++ b/arch/x86/kvm/vmx/pmu_intel.c @@ -750,24 +750,36 @@ static void intel_pmu_cleanup(struct kvm_vcpu *vcpu) intel_pmu_release_guest_lbr_event(vcpu); } -void intel_pmu_cross_mapped_check(struct kvm_pmu *pmu) +u64 __intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu) { - struct kvm_pmc *pmc = NULL; + u64 guest_pebs_enable = pmu->pebs_enable & pmu->global_ctrl; + u64 pebs_enable = 0; + struct kvm_pmc *pmc; int bit, hw_idx; - kvm_for_each_pmc(pmu, pmc, bit, (unsigned long *)&pmu->global_ctrl) { - if (!pmc_is_locally_enabled(pmc) || - !pmc_is_globally_enabled(pmc) || !pmc->perf_event) + /* + * Omit counters that are locally disabled, don't have a perf event, or + * ended up with a perf event that is using a different counter than + * the guest, i.e. where the guest PMC is different than the host PMC + * being used on behalf of the guest. PEBS records include + * PERF_GLOBAL_STATUS, and so using a counter with a different index + * means the guest will see overflow status for the wrong counter(s). + */ + kvm_for_each_pmc(pmu, pmc, bit, (unsigned long *)&guest_pebs_enable) { + if (!pmc_is_locally_enabled(pmc) || !pmc->perf_event) continue; /* - * A negative index indicates the event isn't mapped to a + * Note, a negative index indicates the event isn't mapped to a * physical counter in the host, e.g. due to contention. */ hw_idx = pmc->perf_event->hw.idx; - if (hw_idx != pmc->idx && hw_idx > -1) - pmu->host_cross_mapped_mask |= BIT_ULL(hw_idx); + if (hw_idx != pmc->idx) + continue; + + pebs_enable |= BIT_ULL(pmc->idx); } + return pebs_enable; } static bool intel_pmu_is_mediated_pmu_supported(struct x86_pmu_capability *host_pmu) diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c index 00850a3a77d6..aac79fe08005 100644 --- a/arch/x86/kvm/vmx/tdx.c +++ b/arch/x86/kvm/vmx/tdx.c @@ -362,7 +362,7 @@ static void tdx_reclaim_control_page(struct page *ctrl_page) if (tdx_reclaim_page(ctrl_page)) return; - __free_page(ctrl_page); + tdx_free_control_page(ctrl_page); } struct tdx_flush_vp_arg { @@ -589,7 +589,7 @@ static void tdx_reclaim_td_control_pages(struct kvm *kvm) tdx_quirk_reset_paddr(page_to_phys(kvm_tdx->td.tdr_page), PAGE_SIZE); - __free_page(kvm_tdx->td.tdr_page); + tdx_free_control_page(kvm_tdx->td.tdr_page); kvm_tdx->td.tdr_page = NULL; } @@ -681,6 +681,8 @@ int tdx_vcpu_create(struct kvm_vcpu *vcpu) if (!irqchip_split(vcpu->kvm)) return -EINVAL; + tdx_init_pamt_cache(&tdx->pamt_cache); + fpstate_set_confidential(&vcpu->arch.guest_fpu); vcpu->arch.apic->guest_apic_protected = true; INIT_LIST_HEAD(&tdx->vt.pi_wakeup_list); @@ -866,6 +868,8 @@ void tdx_vcpu_free(struct kvm_vcpu *vcpu) struct vcpu_tdx *tdx = to_tdx(vcpu); int i; + tdx_free_pamt_cache(&tdx->pamt_cache); + if (vcpu->cpu != -1) { KVM_BUG_ON(tdx->state == VCPU_TD_STATE_INITIALIZED, vcpu->kvm); tdx_flush_vp_on_cpu(vcpu); @@ -1618,6 +1622,17 @@ void tdx_load_mmu_pgd(struct kvm_vcpu *vcpu, hpa_t root_hpa, int pgd_level) td_vmcs_write64(to_tdx(vcpu), SHARED_EPT_POINTER, root_hpa); } +static int tdx_topup_external_pamt_cache(struct kvm_vcpu *vcpu, int min_nr_spts) +{ + /* + * Minus one page to exclude the root SPT, but plus one page for a + * possible 4KB private mapping. + */ + min_nr_spts += -1 + 1; + + return tdx_topup_pamt_cache(&to_tdx(vcpu)->pamt_cache, min_nr_spts); +} + static int tdx_mem_page_add(struct kvm *kvm, gfn_t gfn, enum pg_level level, kvm_pfn_t pfn) { @@ -1676,16 +1691,28 @@ static struct page *tdx_spte_to_sept_pt(struct kvm *kvm, gfn_t gfn, static int tdx_sept_map_nonleaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level level, u64 new_spte) { + struct kvm_vcpu *vcpu = kvm_get_running_vcpu(); gpa_t gpa = gfn_to_gpa(gfn); u64 err, entry, level_state; struct page *sept_pt; + int ret; + + if (KVM_BUG_ON(!vcpu, kvm)) + return -EIO; sept_pt = tdx_spte_to_sept_pt(kvm, gfn, new_spte, level); if (!sept_pt) return -EIO; + ret = tdx_pamt_get(page_to_pfn(sept_pt), &to_tdx(vcpu)->pamt_cache); + if (KVM_BUG_ON(ret, kvm)) + return ret; + err = tdh_mem_sept_add(&to_kvm_tdx(kvm)->td, gpa, level, sept_pt, &entry, &level_state); + if (err) + tdx_pamt_put(page_to_pfn(sept_pt)); + if (unlikely(tdx_operand_busy(err))) return -EBUSY; @@ -1698,8 +1725,13 @@ static int tdx_sept_map_nonleaf_spte(struct kvm *kvm, gfn_t gfn, static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level level, u64 new_spte) { + struct kvm_vcpu *vcpu = kvm_get_running_vcpu(); struct kvm_tdx *kvm_tdx = to_kvm_tdx(kvm); kvm_pfn_t pfn = spte_to_pfn(new_spte); + int ret; + + if (KVM_BUG_ON(!vcpu, kvm)) + return -EIO; /* TODO: handle large pages. */ if (KVM_BUG_ON(level != PG_LEVEL_4K, kvm)) @@ -1707,6 +1739,10 @@ static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level leve WARN_ON_ONCE((new_spte & VMX_EPT_RWX_MASK) != VMX_EPT_RWX_MASK); + ret = tdx_pamt_get(pfn, &to_tdx(vcpu)->pamt_cache); + if (KVM_BUG_ON(ret, kvm)) + return ret; + /* * Ensure pre_fault_allowed is read by kvm_arch_vcpu_pre_fault_memory() * before kvm_tdx->state. Userspace must not be allowed to pre-fault @@ -1719,10 +1755,15 @@ static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level leve * If the TD isn't finalized/runnable, then userspace is initializing * the VM image via KVM_TDX_INIT_MEM_REGION; ADD the page to the TD. */ - if (unlikely(kvm_tdx->state != TD_STATE_RUNNABLE)) - return tdx_mem_page_add(kvm, gfn, level, pfn); + if (likely(kvm_tdx->state == TD_STATE_RUNNABLE)) + ret = tdx_mem_page_aug(kvm, gfn, level, pfn); + else + ret = tdx_mem_page_add(kvm, gfn, level, pfn); + + if (ret) + tdx_pamt_put(pfn); - return tdx_mem_page_aug(kvm, gfn, level, pfn); + return ret; } /* @@ -1819,6 +1860,7 @@ static int tdx_sept_remove_leaf_spte(struct kvm *kvm, gfn_t gfn, return -EIO; tdx_quirk_reset_paddr(PFN_PHYS(pfn), PAGE_SIZE); + tdx_pamt_put(pfn); return 0; } @@ -1862,6 +1904,8 @@ static int tdx_sept_set_private_spte(struct kvm *kvm, gfn_t gfn, u64 old_spte, */ static void tdx_sept_free_private_spt(struct kvm *kvm, struct kvm_mmu_page *sp) { + struct page *sept_pt = virt_to_page(sp->external_spt); + /* * KVM doesn't (yet) zap page table pages in mirror page table while * TD is active, though guest pages mapped in mirror page table could be @@ -1875,15 +1919,15 @@ static void tdx_sept_free_private_spt(struct kvm *kvm, struct kvm_mmu_page *sp) * the page to prevent the kernel from accessing the encrypted page. */ if (KVM_BUG_ON(is_hkid_assigned(to_kvm_tdx(kvm)), kvm) || - tdx_reclaim_page(virt_to_page(sp->external_spt))) + tdx_reclaim_page(sept_pt)) goto out; /* - * Immediately free the S-EPT page because RCU-time free is unnecessary - * after TDH.PHYMEM.PAGE.RECLAIM ensures there are no outstanding - * readers. + * Immediately free the S-EPT page as the TDX subsystem doesn't support + * freeing pages from RCU callbacks, and more importantly because + * TDH.PHYMEM.PAGE.RECLAIM ensures there are no outstanding readers. */ - free_page((unsigned long)sp->external_spt); + tdx_free_control_page(sept_pt); out: sp->external_spt = NULL; } @@ -2456,7 +2500,7 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params, ret = -ENOMEM; - tdr_page = alloc_page(GFP_KERNEL_ACCOUNT); + tdr_page = tdx_alloc_control_page(); if (!tdr_page) goto free_hkid; @@ -2469,7 +2513,7 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params, goto free_tdr; for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++) { - tdcs_pages[i] = alloc_page(GFP_KERNEL_ACCOUNT); + tdcs_pages[i] = tdx_alloc_control_page(); if (!tdcs_pages[i]) goto free_tdcs; } @@ -2587,10 +2631,8 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params, teardown: /* Only free pages not yet added, so start at 'i' */ for (; i < kvm_tdx->td.tdcs_nr_pages; i++) { - if (tdcs_pages[i]) { - __free_page(tdcs_pages[i]); - tdcs_pages[i] = NULL; - } + tdx_free_control_page(tdcs_pages[i]); + tdcs_pages[i] = NULL; } if (!kvm_tdx->td.tdcs_pages) kfree(tdcs_pages); @@ -2605,16 +2647,13 @@ free_packages: free_cpumask_var(packages); free_tdcs: - for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++) { - if (tdcs_pages[i]) - __free_page(tdcs_pages[i]); - } + for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++) + tdx_free_control_page(tdcs_pages[i]); kfree(tdcs_pages); kvm_tdx->td.tdcs_pages = NULL; free_tdr: - if (tdr_page) - __free_page(tdr_page); + tdx_free_control_page(tdr_page); kvm_tdx->td.tdr_page = NULL; free_hkid: @@ -2943,7 +2982,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx) int ret, i; u64 err; - page = alloc_page(GFP_KERNEL_ACCOUNT); + page = tdx_alloc_control_page(); if (!page) return -ENOMEM; tdx->vp.tdvpr_page = page; @@ -2963,7 +3002,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx) } for (i = 0; i < kvm_tdx->td.tdcx_nr_pages; i++) { - page = alloc_page(GFP_KERNEL_ACCOUNT); + page = tdx_alloc_control_page(); if (!page) { ret = -ENOMEM; goto free_tdcx; @@ -2985,7 +3024,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx) * method, but the rest are freed here. */ for (; i < kvm_tdx->td.tdcx_nr_pages; i++) { - __free_page(tdx->vp.tdcx_pages[i]); + tdx_free_control_page(tdx->vp.tdcx_pages[i]); tdx->vp.tdcx_pages[i] = NULL; } return -EIO; @@ -3013,16 +3052,14 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx) free_tdcx: for (i = 0; i < kvm_tdx->td.tdcx_nr_pages; i++) { - if (tdx->vp.tdcx_pages[i]) - __free_page(tdx->vp.tdcx_pages[i]); + tdx_free_control_page(tdx->vp.tdcx_pages[i]); tdx->vp.tdcx_pages[i] = NULL; } kfree(tdx->vp.tdcx_pages); tdx->vp.tdcx_pages = NULL; free_tdvpr: - if (tdx->vp.tdvpr_page) - __free_page(tdx->vp.tdvpr_page); + tdx_free_control_page(tdx->vp.tdvpr_page); tdx->vp.tdvpr_page = NULL; tdx->vp.tdvpr_pa = 0; @@ -3482,6 +3519,10 @@ int __init tdx_hardware_setup(void) vt_x86_ops.set_external_spte = tdx_sept_set_private_spte; vt_x86_ops.free_external_spt = tdx_sept_free_private_spt; + + if (tdx_supports_dynamic_pamt(tdx_sysinfo)) + vt_x86_ops.topup_external_cache = tdx_topup_external_pamt_cache; + vt_x86_ops.protected_apic_has_interrupt = tdx_protected_apic_has_interrupt; return 0; diff --git a/arch/x86/kvm/vmx/tdx.h b/arch/x86/kvm/vmx/tdx.h index ac8323a68b16..fd368e3ee060 100644 --- a/arch/x86/kvm/vmx/tdx.h +++ b/arch/x86/kvm/vmx/tdx.h @@ -72,6 +72,8 @@ struct vcpu_tdx { u64 map_gpa_next; u64 map_gpa_end; + + struct tdx_pamt_cache pamt_cache; }; void tdh_vp_rd_failed(struct vcpu_tdx *tdx, char *uclass, u32 field, u64 err); diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c index 612ab07d4100..2d0443562a16 100644 --- a/arch/x86/kvm/vmx/vmx.c +++ b/arch/x86/kvm/vmx/vmx.c @@ -7363,12 +7363,14 @@ static void atomic_switch_perf_msrs(struct vcpu_vmx *vmx) if (kvm_vcpu_has_mediated_pmu(&vmx->vcpu)) return; - pmu->host_cross_mapped_mask = 0; - if (pmu->pebs_enable & pmu->global_ctrl) - intel_pmu_cross_mapped_check(pmu); + struct x86_guest_pebs guest_pebs = { + .enable = intel_pmu_compute_pebs_enable(pmu), + .ds_area = pmu->ds_area, + .data_cfg = pmu->pebs_data_cfg, + }; /* Note, nr_msrs may be garbage if perf_guest_get_msrs() returns NULL. */ - msrs = perf_guest_get_msrs(&nr_msrs, (void *)pmu); + msrs = perf_guest_get_msrs(&nr_msrs, &guest_pebs); if (!msrs) return; diff --git a/arch/x86/kvm/vmx/vmx.h b/arch/x86/kvm/vmx/vmx.h index dc8517f15bc4..1db461060c7e 100644 --- a/arch/x86/kvm/vmx/vmx.h +++ b/arch/x86/kvm/vmx/vmx.h @@ -664,7 +664,20 @@ static __always_inline struct vcpu_vmx *to_vmx(struct kvm_vcpu *vcpu) return container_of(vcpu, struct vcpu_vmx, vcpu); } -void intel_pmu_cross_mapped_check(struct kvm_pmu *pmu); +u64 __intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu); + +static inline u64 intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu) +{ + /* + * Avoid the function call overhead in the common case that the guest + * isn't using PEBS. + */ + if (!(pmu->pebs_enable & pmu->global_ctrl)) + return 0; + + return __intel_pmu_compute_pebs_enable(pmu); +} + int intel_pmu_create_guest_lbr_event(struct kvm_vcpu *vcpu); void vmx_passthrough_lbr_msrs(struct kvm_vcpu *vcpu); diff --git a/arch/x86/platform/Makefile b/arch/x86/platform/Makefile index 3ed03a2552d0..727b92d0ca25 100644 --- a/arch/x86/platform/Makefile +++ b/arch/x86/platform/Makefile @@ -10,5 +10,4 @@ obj-y += intel-mid/ obj-y += intel-quark/ obj-y += olpc/ obj-y += scx200/ -obj-y += ts5500/ obj-y += uv/ diff --git a/arch/x86/platform/olpc/olpc-xo15-sci.c b/arch/x86/platform/olpc/olpc-xo15-sci.c index 75ed6ceb9df3..a13dd6b8bc6e 100644 --- a/arch/x86/platform/olpc/olpc-xo15-sci.c +++ b/arch/x86/platform/olpc/olpc-xo15-sci.c @@ -210,8 +210,8 @@ static int xo15_sci_resume(struct device *dev) static SIMPLE_DEV_PM_OPS(xo15_sci_pm, NULL, xo15_sci_resume); static const struct acpi_device_id xo15_sci_device_ids[] = { - {"XO15EC", 0}, - {"", 0}, + { .id = "XO15EC" }, + { } }; static struct platform_driver xo15_sci_drv = { diff --git a/arch/x86/platform/pvh/enlighten.c b/arch/x86/platform/pvh/enlighten.c index f2053cbe9b0c..cb442cbd9d82 100644 --- a/arch/x86/platform/pvh/enlighten.c +++ b/arch/x86/platform/pvh/enlighten.c @@ -8,6 +8,7 @@ #include <asm/hypervisor.h> #include <asm/e820/api.h> #include <asm/x86_init.h> +#include <asm/string.h> #include <asm/xen/interface.h> @@ -129,7 +130,7 @@ void __init xen_prepare_pvh(void) * This must not compile to "call memset" because memset() may be * instrumented. */ - __builtin_memset(&pvh_bootparams, 0, sizeof(pvh_bootparams)); + __inline_memset(&pvh_bootparams, 0, sizeof(pvh_bootparams)); hypervisor_specific_init(xen_guest); diff --git a/arch/x86/platform/ts5500/Makefile b/arch/x86/platform/ts5500/Makefile deleted file mode 100644 index 910fe9e3ffb4..000000000000 --- a/arch/x86/platform/ts5500/Makefile +++ /dev/null @@ -1,2 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -obj-$(CONFIG_TS5500) += ts5500.o diff --git a/arch/x86/platform/ts5500/ts5500.c b/arch/x86/platform/ts5500/ts5500.c deleted file mode 100644 index 0b67da056fd9..000000000000 --- a/arch/x86/platform/ts5500/ts5500.c +++ /dev/null @@ -1,341 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-or-later -/* - * Technologic Systems TS-5500 Single Board Computer support - * - * Copyright (C) 2013-2014 Savoir-faire Linux Inc. - * Vivien Didelot <vivien.didelot@savoirfairelinux.com> - * - * This driver registers the Technologic Systems TS-5500 Single Board Computer - * (SBC) and its devices, and exposes information to userspace such as jumpers' - * state or available options. For further information about sysfs entries, see - * Documentation/ABI/testing/sysfs-platform-ts5500. - * - * This code may be extended to support similar x86-based platforms. - * Actually, the TS-5500 and TS-5400 are supported. - */ - -#include <linux/delay.h> -#include <linux/io.h> -#include <linux/kernel.h> -#include <linux/leds.h> -#include <linux/init.h> -#include <linux/platform_data/max197.h> -#include <linux/platform_device.h> -#include <linux/slab.h> - -/* Product code register */ -#define TS5500_PRODUCT_CODE_ADDR 0x74 -#define TS5500_PRODUCT_CODE 0x60 /* TS-5500 product code */ -#define TS5400_PRODUCT_CODE 0x40 /* TS-5400 product code */ - -/* SRAM/RS-485/ADC options, and RS-485 RTS/Automatic RS-485 flags register */ -#define TS5500_SRAM_RS485_ADC_ADDR 0x75 -#define TS5500_SRAM BIT(0) /* SRAM option */ -#define TS5500_RS485 BIT(1) /* RS-485 option */ -#define TS5500_ADC BIT(2) /* A/D converter option */ -#define TS5500_RS485_RTS BIT(6) /* RTS for RS-485 */ -#define TS5500_RS485_AUTO BIT(7) /* Automatic RS-485 */ - -/* External Reset/Industrial Temperature Range options register */ -#define TS5500_ERESET_ITR_ADDR 0x76 -#define TS5500_ERESET BIT(0) /* External Reset option */ -#define TS5500_ITR BIT(1) /* Indust. Temp. Range option */ - -/* LED/Jumpers register */ -#define TS5500_LED_JP_ADDR 0x77 -#define TS5500_LED BIT(0) /* LED flag */ -#define TS5500_JP1 BIT(1) /* Automatic CMOS */ -#define TS5500_JP2 BIT(2) /* Enable Serial Console */ -#define TS5500_JP3 BIT(3) /* Write Enable Drive A */ -#define TS5500_JP4 BIT(4) /* Fast Console (115K baud) */ -#define TS5500_JP5 BIT(5) /* User Jumper */ -#define TS5500_JP6 BIT(6) /* Console on COM1 (req. JP2) */ -#define TS5500_JP7 BIT(7) /* Undocumented (Unused) */ - -/* A/D Converter registers */ -#define TS5500_ADC_CONV_BUSY_ADDR 0x195 /* Conversion state register */ -#define TS5500_ADC_CONV_BUSY BIT(0) -#define TS5500_ADC_CONV_INIT_LSB_ADDR 0x196 /* Start conv. / LSB register */ -#define TS5500_ADC_CONV_MSB_ADDR 0x197 /* MSB register */ -#define TS5500_ADC_CONV_DELAY 12 /* usec */ - -/** - * struct ts5500_sbc - TS-5500 board description - * @name: Board model name. - * @id: Board product ID. - * @sram: Flag for SRAM option. - * @rs485: Flag for RS-485 option. - * @adc: Flag for Analog/Digital converter option. - * @ereset: Flag for External Reset option. - * @itr: Flag for Industrial Temperature Range option. - * @jumpers: Bitfield for jumpers' state. - */ -struct ts5500_sbc { - const char *name; - int id; - bool sram; - bool rs485; - bool adc; - bool ereset; - bool itr; - u8 jumpers; -}; - -/* Board signatures in BIOS shadow RAM */ -static const struct { - const char * const string; - const ssize_t offset; -} ts5500_signatures[] __initconst = { - { "TS-5x00 AMD Elan", 0xb14 }, -}; - -static int __init ts5500_check_signature(void) -{ - void __iomem *bios; - int i, ret = -ENODEV; - - bios = ioremap(0xf0000, 0x10000); - if (!bios) - return -ENOMEM; - - for (i = 0; i < ARRAY_SIZE(ts5500_signatures); i++) { - if (check_signature(bios + ts5500_signatures[i].offset, - ts5500_signatures[i].string, - strlen(ts5500_signatures[i].string))) { - ret = 0; - break; - } - } - - iounmap(bios); - return ret; -} - -static int __init ts5500_detect_config(struct ts5500_sbc *sbc) -{ - u8 tmp; - int ret = 0; - - if (!request_region(TS5500_PRODUCT_CODE_ADDR, 4, "ts5500")) - return -EBUSY; - - sbc->id = inb(TS5500_PRODUCT_CODE_ADDR); - if (sbc->id == TS5500_PRODUCT_CODE) { - sbc->name = "TS-5500"; - } else if (sbc->id == TS5400_PRODUCT_CODE) { - sbc->name = "TS-5400"; - } else { - pr_err("ts5500: unknown product code 0x%x\n", sbc->id); - ret = -ENODEV; - goto cleanup; - } - - tmp = inb(TS5500_SRAM_RS485_ADC_ADDR); - sbc->sram = tmp & TS5500_SRAM; - sbc->rs485 = tmp & TS5500_RS485; - sbc->adc = tmp & TS5500_ADC; - - tmp = inb(TS5500_ERESET_ITR_ADDR); - sbc->ereset = tmp & TS5500_ERESET; - sbc->itr = tmp & TS5500_ITR; - - tmp = inb(TS5500_LED_JP_ADDR); - sbc->jumpers = tmp & ~TS5500_LED; - -cleanup: - release_region(TS5500_PRODUCT_CODE_ADDR, 4); - return ret; -} - -static ssize_t name_show(struct device *dev, struct device_attribute *attr, - char *buf) -{ - struct ts5500_sbc *sbc = dev_get_drvdata(dev); - - return sprintf(buf, "%s\n", sbc->name); -} -static DEVICE_ATTR_RO(name); - -static ssize_t id_show(struct device *dev, struct device_attribute *attr, - char *buf) -{ - struct ts5500_sbc *sbc = dev_get_drvdata(dev); - - return sprintf(buf, "0x%.2x\n", sbc->id); -} -static DEVICE_ATTR_RO(id); - -static ssize_t jumpers_show(struct device *dev, struct device_attribute *attr, - char *buf) -{ - struct ts5500_sbc *sbc = dev_get_drvdata(dev); - - return sprintf(buf, "0x%.2x\n", sbc->jumpers >> 1); -} -static DEVICE_ATTR_RO(jumpers); - -#define TS5500_ATTR_BOOL(_field) \ - static ssize_t _field##_show(struct device *dev, \ - struct device_attribute *attr, char *buf) \ - { \ - struct ts5500_sbc *sbc = dev_get_drvdata(dev); \ - \ - return sprintf(buf, "%d\n", sbc->_field); \ - } \ - static DEVICE_ATTR_RO(_field) - -TS5500_ATTR_BOOL(sram); -TS5500_ATTR_BOOL(rs485); -TS5500_ATTR_BOOL(adc); -TS5500_ATTR_BOOL(ereset); -TS5500_ATTR_BOOL(itr); - -static struct attribute *ts5500_attributes[] = { - &dev_attr_id.attr, - &dev_attr_name.attr, - &dev_attr_jumpers.attr, - &dev_attr_sram.attr, - &dev_attr_rs485.attr, - &dev_attr_adc.attr, - &dev_attr_ereset.attr, - &dev_attr_itr.attr, - NULL -}; - -static const struct attribute_group ts5500_attr_group = { - .attrs = ts5500_attributes, -}; - -static struct resource ts5500_dio1_resource[] = { - DEFINE_RES_IRQ_NAMED(7, "DIO1 interrupt"), -}; - -static struct platform_device ts5500_dio1_pdev = { - .name = "ts5500-dio1", - .id = -1, - .resource = ts5500_dio1_resource, - .num_resources = 1, -}; - -static struct resource ts5500_dio2_resource[] = { - DEFINE_RES_IRQ_NAMED(6, "DIO2 interrupt"), -}; - -static struct platform_device ts5500_dio2_pdev = { - .name = "ts5500-dio2", - .id = -1, - .resource = ts5500_dio2_resource, - .num_resources = 1, -}; - -static void ts5500_led_set(struct led_classdev *led_cdev, - enum led_brightness brightness) -{ - outb(!!brightness, TS5500_LED_JP_ADDR); -} - -static enum led_brightness ts5500_led_get(struct led_classdev *led_cdev) -{ - return (inb(TS5500_LED_JP_ADDR) & TS5500_LED) ? LED_FULL : LED_OFF; -} - -static struct led_classdev ts5500_led_cdev = { - .name = "ts5500:green:", - .brightness_set = ts5500_led_set, - .brightness_get = ts5500_led_get, -}; - -static int ts5500_adc_convert(u8 ctrl) -{ - u8 lsb, msb; - - /* Start conversion (ensure the 3 MSB are set to 0) */ - outb(ctrl & 0x1f, TS5500_ADC_CONV_INIT_LSB_ADDR); - - /* - * The platform has CPLD logic driving the A/D converter. - * The conversion must complete within 11 microseconds, - * otherwise we have to re-initiate a conversion. - */ - udelay(TS5500_ADC_CONV_DELAY); - if (inb(TS5500_ADC_CONV_BUSY_ADDR) & TS5500_ADC_CONV_BUSY) - return -EBUSY; - - /* Read the raw data */ - lsb = inb(TS5500_ADC_CONV_INIT_LSB_ADDR); - msb = inb(TS5500_ADC_CONV_MSB_ADDR); - - return (msb << 8) | lsb; -} - -static struct max197_platform_data ts5500_adc_pdata = { - .convert = ts5500_adc_convert, -}; - -static struct platform_device ts5500_adc_pdev = { - .name = "max197", - .id = -1, - .dev = { - .platform_data = &ts5500_adc_pdata, - }, -}; - -static int __init ts5500_init(void) -{ - struct platform_device *pdev; - struct ts5500_sbc *sbc; - int err; - - /* - * There is no DMI available or PCI bridge subvendor info, - * only the BIOS provides a 16-bit identification call. - * It is safer to find a signature in the BIOS shadow RAM. - */ - err = ts5500_check_signature(); - if (err) - return err; - - pdev = platform_device_register_simple("ts5500", -1, NULL, 0); - if (IS_ERR(pdev)) - return PTR_ERR(pdev); - - sbc = devm_kzalloc(&pdev->dev, sizeof(struct ts5500_sbc), GFP_KERNEL); - if (!sbc) { - err = -ENOMEM; - goto error; - } - - err = ts5500_detect_config(sbc); - if (err) - goto error; - - platform_set_drvdata(pdev, sbc); - - err = sysfs_create_group(&pdev->dev.kobj, &ts5500_attr_group); - if (err) - goto error; - - if (sbc->id == TS5500_PRODUCT_CODE) { - ts5500_dio1_pdev.dev.parent = &pdev->dev; - if (platform_device_register(&ts5500_dio1_pdev)) - dev_warn(&pdev->dev, "DIO1 block registration failed\n"); - ts5500_dio2_pdev.dev.parent = &pdev->dev; - if (platform_device_register(&ts5500_dio2_pdev)) - dev_warn(&pdev->dev, "DIO2 block registration failed\n"); - } - - if (led_classdev_register(&pdev->dev, &ts5500_led_cdev)) - dev_warn(&pdev->dev, "LED registration failed\n"); - - if (sbc->adc) { - ts5500_adc_pdev.dev.parent = &pdev->dev; - if (platform_device_register(&ts5500_adc_pdev)) - dev_warn(&pdev->dev, "ADC registration failed\n"); - } - - return 0; -error: - platform_device_unregister(pdev); - return err; -} -device_initcall(ts5500_init); diff --git a/arch/x86/virt/svm/sev.c b/arch/x86/virt/svm/sev.c index cff285d8ad8e..bd70c4d0b774 100644 --- a/arch/x86/virt/svm/sev.c +++ b/arch/x86/virt/svm/sev.c @@ -19,6 +19,7 @@ #include <linux/iommu.h> #include <linux/amd-iommu.h> #include <linux/nospec.h> +#include <linux/workqueue.h> #include <asm/sev.h> #include <asm/processor.h> @@ -124,6 +125,28 @@ static void *rmp_bookkeeping __ro_after_init; static u64 probed_rmp_base, probed_rmp_size; +static u64 rmpopt_pa_start, rmpopt_pa_end; + +enum rmpopt_op_type { + RMPOPT_OP_VERIFY_AND_REPORT_STATUS, + RMPOPT_OP_REPORT_STATUS +}; + +static struct workqueue_struct *rmpopt_wq; +static struct delayed_work rmpopt_delayed_work; + +/* Software RMPOPT facilities initialized */ +static bool rmpopt_soft_init; + +/* + * Delay, in milliseconds, before the RMP re-optimization pass runs after an + * SNP guest is torn down. This coalesces a burst of teardowns into a single + * scan and gives each guest's pages time to be converted back to the shared, + * hypervisor-owned state. The 10 second value is a heuristic trading + * re-optimization latency against scanning too eagerly. + */ +#define RMPOPT_WORK_TIMEOUT (10 * MSEC_PER_SEC) + static LIST_HEAD(snp_leaked_pages_list); static DEFINE_SPINLOCK(snp_leaked_pages_list_lock); @@ -513,7 +536,6 @@ static void clear_hsave_pa(void *arg) int snp_prepare(void) { - int ret; u64 val; /* @@ -526,14 +548,18 @@ int snp_prepare(void) clear_rmp(); - cpus_read_lock(); + /* + * No CPU may come online without SnpEn while SNP is active; disable + * hotplug here and re-enable it in snp_shutdown(). + */ + cpu_hotplug_disable(); if (!cpumask_equal(cpu_online_mask, cpu_present_mask)) { - ret = -EOPNOTSUPP; + cpu_hotplug_enable(); pr_warn("SNP init failed: not all CPUs online. (%*pbl online <-> %*pbl present masks).\n", cpumask_pr_args(cpu_online_mask), cpumask_pr_args(cpu_present_mask)); - goto unlock; + return -EOPNOTSUPP; } wbinvd_on_all_cpus(); @@ -548,12 +574,7 @@ int snp_prepare(void) /* SNP_INIT requires MSR_VM_HSAVE_PA to be cleared on all CPUs. */ on_each_cpu(clear_hsave_pa, NULL, 1); - ret = 0; - -unlock: - cpus_read_unlock(); - - return ret; + return 0; } EXPORT_SYMBOL_FOR_MODULES(snp_prepare, "ccp"); @@ -565,18 +586,122 @@ void snp_shutdown(void) if (syscfg & MSR_AMD64_SYSCFG_SNP_EN) return; + if (rmpopt_soft_init) + cancel_delayed_work_sync(&rmpopt_delayed_work); + clear_rmp(); on_each_cpu(mfd_reconfigure, NULL, 1); + + /* + * The firmware has disabled SNP (SnpEn is clear), so re-enable CPU + * hotplug. A legacy SNP shutdown returns above with SnpEn still set and + * leaves hotplug disabled. + */ + cpu_hotplug_enable(); } EXPORT_SYMBOL_FOR_MODULES(snp_shutdown, "ccp"); /* + * RMPOPT optimizations skip RMP checks at 1GB granularity if this range of + * memory does not contain any SNP guest memory. + * + * @pa is a system physical address; RMPOPT operates on the containing 1GB. + */ +static void rmpopt(u64 pa) +{ + enum rmpopt_op_type op = RMPOPT_OP_VERIFY_AND_REPORT_STATUS; + u64 pa_start = ALIGN_DOWN(pa, SZ_1G); + + /* Supported by binutils 2.48+ */ + asm volatile(".byte 0xf2, 0x0f, 0x01, 0xfc" + :: "a" (pa_start), "c" (op) + : "memory", "cc"); +} + +static void rmpopt_scan_range(void *arg) +{ + u64 pa; + + for (pa = rmpopt_pa_start; pa < rmpopt_pa_end; pa += SZ_1G) + rmpopt(pa); +} + +static void do_rmpopt_work(struct work_struct *work) +{ + /* + * Warm up the RMPOPT cache on this pinned per-CPU worker with interrupts + * enabled, so the IRQ-disabled fan-out below only issues cache-hit RMPOPTs. + */ + rmpopt_scan_range(NULL); + + on_each_cpu_mask(cpu_primary_thread_mask, rmpopt_scan_range, NULL, true); +} + +static int __init rmpopt_init(void) +{ + if (!cpu_feature_enabled(X86_FEATURE_RMPOPT)) + return -ENODEV; + + rmpopt_wq = alloc_workqueue("rmpopt_wq", WQ_PERCPU, 1); + if (!rmpopt_wq) { + pr_err("Failed to allocate RMPOPT workqueue\n"); + return -ENOMEM; + } + + INIT_DELAYED_WORK(&rmpopt_delayed_work, do_rmpopt_work); + + /* The optimization range is fixed at boot; compute it once. */ + rmpopt_pa_start = ALIGN_DOWN(PFN_PHYS(min_low_pfn), SZ_1G); + rmpopt_pa_end = ALIGN(PFN_PHYS(max_pfn), SZ_1G); + if ((rmpopt_pa_end - rmpopt_pa_start) > SZ_2T) + rmpopt_pa_end = rmpopt_pa_start + SZ_2T; + + pr_info("RMPOPT optimizations enabled\n"); + + rmpopt_soft_init = true; + + return 0; +} +device_initcall(rmpopt_init); + +void snp_enable_rmpopt(void) +{ + u64 base; + int cpu; + + if (!cpu_feature_enabled(X86_FEATURE_RMPOPT)) + return; + + if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) + return; + + if (!rmpopt_soft_init) + return; + + /* + * Per-CPU RMPOPT tables cover at most 2 TB. Program each core's + * RMPOPT_BASE with the start of RAM to optimize up to 2 TB. + */ + rdmsrq(MSR_AMD64_RMPOPT_BASE, base); + if (!(base & MSR_AMD64_RMPOPT_ENABLE)) + for_each_cpu(cpu, cpu_primary_thread_mask) + wrmsrq_on_cpu(cpu, MSR_AMD64_RMPOPT_BASE, + rmpopt_pa_start | MSR_AMD64_RMPOPT_ENABLE); + + mod_delayed_work(rmpopt_wq, &rmpopt_delayed_work, + msecs_to_jiffies(RMPOPT_WORK_TIMEOUT)); +} +EXPORT_SYMBOL_FOR_MODULES(snp_enable_rmpopt, "ccp,kvm-amd"); + +/* * Do the necessary preparations which are verified by the firmware as * described in the SNP_INIT_EX firmware command description in the SNP * firmware ABI spec. */ int __init snp_rmptable_init(void) { + u64 val; + if (WARN_ON_ONCE(!cc_platform_has(CC_ATTR_HOST_SEV_SNP))) return -ENOSYS; @@ -587,6 +712,15 @@ int __init snp_rmptable_init(void) return -ENOSYS; /* + * On a kexec boot SNP may already be enabled (legacy firmware leaves + * SnpEn set across shutdown), in which case snp_prepare() bails without + * disabling CPU hotplug, so disable it here. + */ + rdmsrq(MSR_AMD64_SYSCFG, val); + if (val & MSR_AMD64_SYSCFG_SNP_EN) + cpu_hotplug_disable(); + + /* * Setting crash_kexec_post_notifiers to 'true' to ensure that SNP panic * notifier is invoked to do SNP IOMMU shutdown before kdump. */ @@ -683,13 +817,21 @@ static bool probe_segmented_rmptable_info(void) bool snp_probe_rmptable_info(void) { - if (cpu_feature_enabled(X86_FEATURE_SEGMENTED_RMP)) + if (cpu_feature_enabled(X86_FEATURE_SEGMENTED_RMP)) { rdmsrq(MSR_AMD64_RMP_CFG, rmp_cfg); - if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) - return probe_segmented_rmptable_info(); - else - return probe_contiguous_rmptable_info(); + if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) { + if (probe_segmented_rmptable_info()) + return true; + + setup_clear_cpu_cap(X86_FEATURE_RMPOPT); + return false; + } + } else { + setup_clear_cpu_cap(X86_FEATURE_RMPOPT); + } + + return probe_contiguous_rmptable_info(); } /* diff --git a/arch/x86/virt/vmx/tdx/seamcall_internal.h b/arch/x86/virt/vmx/tdx/seamcall_internal.h index be5f446467df..051ad2d45cab 100644 --- a/arch/x86/virt/vmx/tdx/seamcall_internal.h +++ b/arch/x86/virt/vmx/tdx/seamcall_internal.h @@ -11,6 +11,7 @@ #ifndef _X86_VIRT_SEAMCALL_INTERNAL_H #define _X86_VIRT_SEAMCALL_INTERNAL_H +#include <linux/bitfield.h> #include <linux/printk.h> #include <linux/types.h> #include <asm/archrandom.h> @@ -23,9 +24,27 @@ u64 __seamcall_saved_ret(u64 fn, struct tdx_module_args *args); typedef u64 (*sc_func_t)(u64 fn, struct tdx_module_args *args); +/* + * SEAMCALL leaf: + * + * Bit 15:0 Leaf number + * Bit 23:16 Leaf ABI version number + * Bit 24 Pending interrupts detection mode + * Bit 63 1 for P-SEAMLDR leaf, 0 for TDX module leaf + */ +#define SEAMCALL_LEAF_MASK GENMASK_U64(15, 0) +#define SEAMCALL_SEAMLDR_MASK BIT_U64(63) + static __always_inline u64 __seamcall_dirty_cache(sc_func_t func, u64 fn, struct tdx_module_args *args) { + /* + * fn contains leaf number for TDX module calls and P-SEAMLDR calls. + * Other fields in SEAMCALL leaf like leaf ABI version number are in + * struct tdx_module_args. + */ + BUILD_BUG_ON(fn & ~(SEAMCALL_LEAF_MASK | SEAMCALL_SEAMLDR_MASK)); + lockdep_assert_preemption_disabled(); /* diff --git a/arch/x86/virt/vmx/tdx/tdx.c b/arch/x86/virt/vmx/tdx/tdx.c index 1b9ff749dd8e..96ced0494b68 100644 --- a/arch/x86/virt/vmx/tdx/tdx.c +++ b/arch/x86/virt/vmx/tdx/tdx.c @@ -30,6 +30,7 @@ #include <linux/suspend.h> #include <linux/syscore_ops.h> #include <linux/idr.h> +#include <linux/vmalloc.h> #include <asm/page.h> #include <asm/special_insns.h> #include <asm/msr-index.h> @@ -46,6 +47,9 @@ #include "seamcall_internal.h" #include "tdx.h" +/* Number of DPAMT pages to be provided to TDX module per 2MB region of PA */ +#define TDX_DPAMT_ENTRY_PAGE_CNT 2 + struct tdx_module_state { bool initialized; bool sysinit_done; @@ -63,6 +67,14 @@ static DEFINE_PER_CPU(bool, tdx_lp_initialized); static struct tdmr_info_list tdx_tdmr_list; +/* + * On a machine with DPAMT, the kernel maintains a reference counter + * for every 2MB range. The counter indicates how many users there are for + * the DPAMT at the 2MB range. The kernel allocates DPAMT refcounts at + * initialization. + */ +static atomic_t *dpamt_refcounts; + /* All TDX-usable memory regions. Protected by mem_hotplug_lock. */ static LIST_HEAD(tdx_memlist); @@ -253,6 +265,42 @@ static struct syscore tdx_syscore = { }; /* + * Allocate DPAMT reference counters for all physical memory. + * + * It consumes 2MB for every 1TB of physical memory. + */ +static __init int init_dpamt_refcounts(void) +{ + size_t size = DIV_ROUND_UP(max_pfn, PTRS_PER_PTE) * sizeof(*dpamt_refcounts); + + if (!tdx_supports_dynamic_pamt(&tdx_sysinfo)) + return 0; + + dpamt_refcounts = vzalloc(size); + if (!dpamt_refcounts) + return -ENOMEM; + + return 0; +} + +static __init void free_dpamt_refcounts(void) +{ + if (!tdx_supports_dynamic_pamt(&tdx_sysinfo)) + return; + + vfree(dpamt_refcounts); + dpamt_refcounts = NULL; +} + +static atomic_t *tdx_find_dpamt_refcount(unsigned long pfn) +{ + /* Find which PMD a PFN is in. */ + unsigned long index = pfn >> (PMD_SHIFT - PAGE_SHIFT); + + return &dpamt_refcounts[index]; +} + +/* * Add a memory region as a TDX memory block. The caller must make sure * all memory regions are added in address ascending order and don't * overlap. @@ -510,35 +558,37 @@ static __init int fill_out_tdmrs(struct list_head *tmb_list, return 0; } +static __init unsigned long tdmr_get_pamt_bitmap_sz(struct tdmr_info *tdmr) +{ + unsigned long pamt_sz, nr_pamt_entries; + int bits_per_entry; + + bits_per_entry = tdx_sysinfo.tdmr.pamt_page_bitmap_entry_bits; + nr_pamt_entries = tdmr->size >> PAGE_SHIFT; + pamt_sz = DIV_ROUND_UP(nr_pamt_entries * bits_per_entry, BITS_PER_BYTE); + + return PAGE_ALIGN(pamt_sz); +} + /* * Calculate PAMT size given a TDMR and a page size. The returned * PAMT size is always aligned up to 4K page boundary. */ -static __init unsigned long tdmr_get_pamt_sz(struct tdmr_info *tdmr, int pgsz, - u16 pamt_entry_size) +static __init unsigned long tdmr_get_pamt_sz(struct tdmr_info *tdmr, int pgsz) { unsigned long pamt_sz, nr_pamt_entries; + const int tdx_pg_size_shift[TDX_PS_NR] = { PAGE_SHIFT, PMD_SHIFT, PUD_SHIFT }; + const u16 pamt_entry_size[TDX_PS_NR] = { + tdx_sysinfo.tdmr.pamt_4k_entry_size, + tdx_sysinfo.tdmr.pamt_2m_entry_size, + tdx_sysinfo.tdmr.pamt_1g_entry_size, + }; - switch (pgsz) { - case TDX_PS_4K: - nr_pamt_entries = tdmr->size >> PAGE_SHIFT; - break; - case TDX_PS_2M: - nr_pamt_entries = tdmr->size >> PMD_SHIFT; - break; - case TDX_PS_1G: - nr_pamt_entries = tdmr->size >> PUD_SHIFT; - break; - default: - WARN_ON_ONCE(1); - return 0; - } + nr_pamt_entries = tdmr->size >> tdx_pg_size_shift[pgsz]; + pamt_sz = nr_pamt_entries * pamt_entry_size[pgsz]; - pamt_sz = nr_pamt_entries * pamt_entry_size; /* TDX requires PAMT size must be 4K aligned */ - pamt_sz = ALIGN(pamt_sz, PAGE_SIZE); - - return pamt_sz; + return PAGE_ALIGN(pamt_sz); } /* @@ -576,15 +626,11 @@ static __init int tdmr_get_nid(struct tdmr_info *tdmr, struct list_head *tmb_lis * within @tdmr, and set up PAMTs for @tdmr. */ static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr, - struct list_head *tmb_list, - u16 pamt_entry_size[]) + struct list_head *tmb_list) { - unsigned long pamt_base[TDX_PS_NR]; - unsigned long pamt_size[TDX_PS_NR]; - unsigned long tdmr_pamt_base; unsigned long tdmr_pamt_size; struct page *pamt; - int pgsz, nid; + int nid; nid = tdmr_get_nid(tdmr, tmb_list); @@ -592,13 +638,18 @@ static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr, * Calculate the PAMT size for each TDX supported page size * and the total PAMT size. */ - tdmr_pamt_size = 0; - for (pgsz = TDX_PS_4K; pgsz < TDX_PS_NR; pgsz++) { - pamt_size[pgsz] = tdmr_get_pamt_sz(tdmr, pgsz, - pamt_entry_size[pgsz]); - tdmr_pamt_size += pamt_size[pgsz]; + tdmr->pamt_1g_size = tdmr_get_pamt_sz(tdmr, TDX_PS_1G); + tdmr->pamt_2m_size = tdmr_get_pamt_sz(tdmr, TDX_PS_2M); + + if (tdx_supports_dynamic_pamt(&tdx_sysinfo)) { + /* With DPAMT, PAMT_4K is replaced with a bitmap */ + tdmr->pamt_4k_size = tdmr_get_pamt_bitmap_sz(tdmr); + } else { + tdmr->pamt_4k_size = tdmr_get_pamt_sz(tdmr, TDX_PS_4K); } + tdmr_pamt_size = tdmr->pamt_4k_size + tdmr->pamt_2m_size + tdmr->pamt_1g_size; + /* * Allocate one chunk of physically contiguous memory for all * PAMTs. This helps minimize the PAMT's use of reserved areas @@ -606,25 +657,17 @@ static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr, */ pamt = alloc_contig_pages(tdmr_pamt_size >> PAGE_SHIFT, GFP_KERNEL, nid, &node_online_map); - if (!pamt) - return -ENOMEM; /* - * Break the contiguous allocation back up into the - * individual PAMTs for each page size. + * tdmr->pamt_4k_base is still zero so the error + * path of the caller will skip freeing the PAMT. */ - tdmr_pamt_base = page_to_pfn(pamt) << PAGE_SHIFT; - for (pgsz = TDX_PS_4K; pgsz < TDX_PS_NR; pgsz++) { - pamt_base[pgsz] = tdmr_pamt_base; - tdmr_pamt_base += pamt_size[pgsz]; - } + if (!pamt) + return -ENOMEM; - tdmr->pamt_4k_base = pamt_base[TDX_PS_4K]; - tdmr->pamt_4k_size = pamt_size[TDX_PS_4K]; - tdmr->pamt_2m_base = pamt_base[TDX_PS_2M]; - tdmr->pamt_2m_size = pamt_size[TDX_PS_2M]; - tdmr->pamt_1g_base = pamt_base[TDX_PS_1G]; - tdmr->pamt_1g_size = pamt_size[TDX_PS_1G]; + tdmr->pamt_4k_base = page_to_phys(pamt); + tdmr->pamt_2m_base = tdmr->pamt_4k_base + tdmr->pamt_4k_size; + tdmr->pamt_1g_base = tdmr->pamt_2m_base + tdmr->pamt_2m_size; return 0; } @@ -655,10 +698,7 @@ static __init void tdmr_do_pamt_func(struct tdmr_info *tdmr, tdmr_get_pamt(tdmr, &pamt_base, &pamt_size); /* Do nothing if PAMT hasn't been allocated for this TDMR */ - if (!pamt_size) - return; - - if (WARN_ON_ONCE(!pamt_base)) + if (!pamt_base) return; pamt_func(pamt_base, pamt_size); @@ -684,14 +724,12 @@ static __init void tdmrs_free_pamt_all(struct tdmr_info_list *tdmr_list) /* Allocate and set up PAMTs for all TDMRs */ static __init int tdmrs_set_up_pamt_all(struct tdmr_info_list *tdmr_list, - struct list_head *tmb_list, - u16 pamt_entry_size[]) + struct list_head *tmb_list) { int i, ret = 0; for (i = 0; i < tdmr_list->nr_consumed_tdmrs; i++) { - ret = tdmr_set_up_pamt(tdmr_entry(tdmr_list, i), tmb_list, - pamt_entry_size); + ret = tdmr_set_up_pamt(tdmr_entry(tdmr_list, i), tmb_list); if (ret) goto err; } @@ -968,18 +1006,13 @@ static __init int construct_tdmrs(struct list_head *tmb_list, struct tdmr_info_list *tdmr_list, struct tdx_sys_info_tdmr *sysinfo_tdmr) { - u16 pamt_entry_size[TDX_PS_NR] = { - sysinfo_tdmr->pamt_4k_entry_size, - sysinfo_tdmr->pamt_2m_entry_size, - sysinfo_tdmr->pamt_1g_entry_size, - }; int ret; ret = fill_out_tdmrs(tmb_list, tdmr_list); if (ret) return ret; - ret = tdmrs_set_up_pamt_all(tdmr_list, tmb_list, pamt_entry_size); + ret = tdmrs_set_up_pamt_all(tdmr_list, tmb_list); if (ret) return ret; @@ -998,6 +1031,8 @@ static __init int construct_tdmrs(struct list_head *tmb_list, return ret; } +#define TDX_SYS_CONFIG_DYNAMIC_PAMT BIT(16) + static __init int config_tdx_module(struct tdmr_info_list *tdmr_list, u64 global_keyid) { @@ -1026,6 +1061,12 @@ static __init int config_tdx_module(struct tdmr_info_list *tdmr_list, args.rcx = __pa(tdmr_pa_array); args.rdx = tdmr_list->nr_consumed_tdmrs; args.r8 = global_keyid; + + if (tdx_supports_dynamic_pamt(&tdx_sysinfo)) { + pr_info("Enable Dynamic PAMT\n"); + args.r8 |= TDX_SYS_CONFIG_DYNAMIC_PAMT; + } + ret = seamcall_prerr(TDH_SYS_CONFIG, &args); /* Free the array as it is not required anymore. */ @@ -1167,10 +1208,14 @@ static __init int init_tdx_module(void) */ get_online_mems(); - ret = build_tdx_memlist(&tdx_memlist); + ret = init_dpamt_refcounts(); if (ret) goto out_put_tdxmem; + ret = build_tdx_memlist(&tdx_memlist); + if (ret) + goto err_free_dpamt_refcounts; + /* Allocate enough space for constructing TDMRs */ ret = alloc_tdmr_list(&tdx_tdmr_list, &tdx_sysinfo.tdmr); if (ret) @@ -1220,6 +1265,8 @@ err_free_tdmrs: free_tdmr_list(&tdx_tdmr_list); err_free_tdxmem: free_tdx_memlist(&tdx_memlist); +err_free_dpamt_refcounts: + free_dpamt_refcounts(); goto out_put_tdxmem; } @@ -1912,10 +1959,11 @@ u64 tdh_vp_init(struct tdx_vp *vp, u64 initial_rcx, u32 x2apicid) .rcx = vp->tdvpr_pa, .rdx = initial_rcx, .r8 = x2apicid, + /* apicid requires version == 1. */ + .version = 1, }; - /* apicid requires version == 1. */ - return seamcall(TDH_VP_INIT | (1ULL << TDX_VERSION_SHIFT), &args); + return seamcall(TDH_VP_INIT, &args); } EXPORT_SYMBOL_FOR_KVM(tdh_vp_init); @@ -2005,6 +2053,269 @@ u64 tdh_phymem_page_wbinvd_hkid(u64 hkid, kvm_pfn_t pfn) } EXPORT_SYMBOL_FOR_KVM(tdh_phymem_page_wbinvd_hkid); +bool tdx_supports_dynamic_pamt(const struct tdx_sys_info *sysinfo) +{ + return sysinfo->features.tdx_features0 & TDX_FEATURES0_DYNAMIC_PAMT; +} +EXPORT_SYMBOL_FOR_KVM(tdx_supports_dynamic_pamt); + +static struct page *tdx_alloc_page_pamt_cache(struct tdx_pamt_cache *cache) +{ + struct page *page; + + page = list_first_entry_or_null(&cache->page_list, struct page, lru); + if (page) { + list_del(&page->lru); + cache->cnt--; + } + + return page; +} + +static struct page *alloc_dpamt_page(struct tdx_pamt_cache *cache) +{ + if (cache) + return tdx_alloc_page_pamt_cache(cache); + + return alloc_page(GFP_KERNEL_ACCOUNT); +} + +static int alloc_pamt_array(struct page **pamt_pages, struct tdx_pamt_cache *cache) +{ + int i, j; + + for (i = 0; i < TDX_DPAMT_ENTRY_PAGE_CNT; i++) { + pamt_pages[i] = alloc_dpamt_page(cache); + if (!pamt_pages[i]) + goto err; + } + + return 0; + +err: + for (j = 0; j < i; j++) + __free_page(pamt_pages[j]); + + return -ENOMEM; +} + +static void free_pamt_array(struct page **pamt_pages) +{ + int i; + + for (i = 0; i < TDX_DPAMT_ENTRY_PAGE_CNT; i++) { + /* + * Reset pages unconditionally to cover cases + * where they were passed to the TDX module. + */ + tdx_quirk_reset_paddr(page_to_phys(pamt_pages[i]), PAGE_SIZE); + + __free_page(pamt_pages[i]); + } +} + +/* Helper for building DPAMT seamcall() arguments. */ +static u64 pamt_2mb_arg(kvm_pfn_t pfn) +{ + /* Find the 2MB-wide DPAMT region for 'pfn': */ + unsigned long hpa_2mb = ALIGN_DOWN(pfn << PAGE_SHIFT, PMD_SIZE); + + /* + * TDX ABI requires specifying the page level the installed DPAMT + * backing will cover, even though today only 2MB is supported. + */ + return hpa_2mb | TDX_PS_2M; +} + +/* Add DPAMT backing for the 2MB region surrounding the given pfn. */ +static u64 tdh_phymem_pamt_add(kvm_pfn_t pfn, struct page **pamt_pages) +{ + struct tdx_module_args args = { + .rcx = pamt_2mb_arg(pfn), + .rdx = page_to_phys(pamt_pages[0]), + .r8 = page_to_phys(pamt_pages[1]), + }; + + return seamcall(TDH_PHYMEM_PAMT_ADD, &args); +} + +/* Remove DPAMT backing for the 2MB region surrounding the given pfn. */ +static u64 tdh_phymem_pamt_remove(kvm_pfn_t pfn, struct page **pamt_pages) +{ + struct tdx_module_args args = { + .rcx = pamt_2mb_arg(pfn), + }; + u64 ret; + + ret = seamcall_ret(TDH_PHYMEM_PAMT_REMOVE, &args); + if (ret) + return ret; + + /* Copy PAMT pages out of the struct per the TDX ABI */ + pamt_pages[0] = phys_to_page(args.rdx); + pamt_pages[1] = phys_to_page(args.r8); + + return 0; +} + +/* Serializes adding/removing DPAMT memory */ +static DEFINE_SPINLOCK(dpamt_lock); + +/* Bump DPAMT refcount for the given pfn and allocate DPAMT backing if needed. */ +int tdx_pamt_get(kvm_pfn_t pfn, struct tdx_pamt_cache *cache) +{ + struct page *pamt_pages[TDX_DPAMT_ENTRY_PAGE_CNT]; + atomic_t *dpamt_refcount; + u64 tdx_status; + int ret; + + if (!tdx_supports_dynamic_pamt(&tdx_sysinfo)) + return 0; + + ret = alloc_pamt_array(pamt_pages, cache); + if (ret) + return ret; + + dpamt_refcount = tdx_find_dpamt_refcount(pfn); + + spin_lock(&dpamt_lock); + + /* + * If the DPAMT entry is already added (i.e. refcount >= 1), + * then just increment the refcount. + */ + if (atomic_inc_not_zero(dpamt_refcount)) + goto out_free; + + /* Try to add the PAMT page and take the refcount 0->1. */ + tdx_status = tdh_phymem_pamt_add(pfn, pamt_pages); + if (WARN_ON_ONCE(tdx_status != TDX_SUCCESS)) { + ret = -EIO; + goto out_free; + } + + atomic_set(dpamt_refcount, 1); + spin_unlock(&dpamt_lock); + return 0; + +out_free: + spin_unlock(&dpamt_lock); + free_pamt_array(pamt_pages); + + return ret; +} +EXPORT_SYMBOL_FOR_KVM(tdx_pamt_get); + +/* Drop DPAMT refcount for the given pfn and free DPAMT backing if needed. */ +void tdx_pamt_put(kvm_pfn_t pfn) +{ + struct page *pamt_pages[TDX_DPAMT_ENTRY_PAGE_CNT] = {}; + atomic_t *dpamt_refcount; + u64 tdx_status; + + if (!tdx_supports_dynamic_pamt(&tdx_sysinfo)) + return; + + dpamt_refcount = tdx_find_dpamt_refcount(pfn); + + spin_lock(&dpamt_lock); + /* + * If there is more than 1 reference on the DPAMT entry, don't + * remove it yet. Just decrement the refcount. + */ + if (atomic_read(dpamt_refcount) > 1) { + atomic_dec(dpamt_refcount); + goto out_unlock; + } + + /* Try to remove the pamt page and take the refcount 1->0. */ + tdx_status = tdh_phymem_pamt_remove(pfn, pamt_pages); + + /* + * Don't free pamt_pages as it could hold garbage when + * tdh_phymem_pamt_remove() fails. Don't panic/BUG_ON(), as + * there is no risk of data corruption, but do yell loudly as + * failure indicates a kernel bug, memory is being leaked, and + * the dangling DPAMT entry may cause future operations to fail. + */ + if (WARN_ON_ONCE(tdx_status != TDX_SUCCESS)) + goto out_unlock; + + atomic_set(dpamt_refcount, 0); + spin_unlock(&dpamt_lock); + free_pamt_array(pamt_pages); + return; +out_unlock: + spin_unlock(&dpamt_lock); +} +EXPORT_SYMBOL_FOR_KVM(tdx_pamt_put); + +void tdx_free_pamt_cache(struct tdx_pamt_cache *cache) +{ + struct page *page; + + while ((page = tdx_alloc_page_pamt_cache(cache))) + __free_page(page); +} +EXPORT_SYMBOL_FOR_KVM(tdx_free_pamt_cache); + +int tdx_topup_pamt_cache(struct tdx_pamt_cache *cache, unsigned long npages) +{ + if (WARN_ON_ONCE(!tdx_supports_dynamic_pamt(&tdx_sysinfo))) + return 0; + + npages *= TDX_DPAMT_ENTRY_PAGE_CNT; + + while (cache->cnt < npages) { + struct page *page = alloc_page(GFP_KERNEL_ACCOUNT); + + if (!page) + return -ENOMEM; + + list_add(&page->lru, &cache->page_list); + cache->cnt++; + } + + return 0; +} +EXPORT_SYMBOL_FOR_KVM(tdx_topup_pamt_cache); + +/* + * Return a page that can be gifted to the TDX module for use as a "control" + * page, i.e. pages that are used for control structures for a given TDX + * guest, and thus obtain TDX protections, including DPAMT tracking. + */ +struct page *tdx_alloc_control_page(void) +{ + struct page *page; + + page = alloc_page(GFP_KERNEL_ACCOUNT); + if (!page) + return NULL; + + if (tdx_pamt_get(page_to_pfn(page), NULL)) { + __free_page(page); + return NULL; + } + + return page; +} +EXPORT_SYMBOL_FOR_KVM(tdx_alloc_control_page); + +/* + * Free a page that was gifted to the TDX module for use as a control + * page. After this, the page is no longer protected by TDX. + */ +void tdx_free_control_page(struct page *page) +{ + if (!page) + return; + + tdx_pamt_put(page_to_pfn(page)); + __free_page(page); +} +EXPORT_SYMBOL_FOR_KVM(tdx_free_control_page); + void tdx_sys_disable(void) { struct tdx_module_args args = {}; diff --git a/arch/x86/virt/vmx/tdx/tdx.h b/arch/x86/virt/vmx/tdx/tdx.h index bdfd0e1e337a..db209541d3cd 100644 --- a/arch/x86/virt/vmx/tdx/tdx.h +++ b/arch/x86/virt/vmx/tdx/tdx.h @@ -48,16 +48,10 @@ #define TDH_SYS_CONFIG 45 #define TDH_SYS_SHUTDOWN 52 #define TDH_SYS_UPDATE 53 +#define TDH_PHYMEM_PAMT_ADD 58 +#define TDH_PHYMEM_PAMT_REMOVE 59 #define TDH_SYS_DISABLE 69 -/* - * SEAMCALL leaf: - * - * Bit 15:0 Leaf number - * Bit 23:16 Version number - */ -#define TDX_VERSION_SHIFT 16 - /* TDX page types */ #define PT_NDA 0x0 #define PT_RSVD 0x1 diff --git a/arch/x86/virt/vmx/tdx/tdx_global_metadata.c b/arch/x86/virt/vmx/tdx/tdx_global_metadata.c index e49c300f23d4..98ebf17aab1c 100644 --- a/arch/x86/virt/vmx/tdx/tdx_global_metadata.c +++ b/arch/x86/virt/vmx/tdx/tdx_global_metadata.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * Automatically generated functions to read TDX global metadata. + * Functions to read TDX global metadata. * * This file doesn't compile on its own as it lacks of inclusion * of SEAMCALL wrapper primitive which reads global metadata. @@ -33,6 +33,18 @@ static __init int get_tdx_sys_info_features(struct tdx_sys_info_features *sysinf return ret; } +static __init int get_tdx_sys_info_tdmr_dpamt(struct tdx_sys_info_tdmr *sysinfo_tdmr) +{ + int ret; + u64 val; + + ret = read_sys_metadata_field(0x9100000000000013, &val); + if (!ret) + sysinfo_tdmr->pamt_page_bitmap_entry_bits = val; + + return ret; +} + static __init int get_tdx_sys_info_tdmr(struct tdx_sys_info_tdmr *sysinfo_tdmr) { int ret = 0; @@ -129,5 +141,14 @@ static __init int get_tdx_sys_info(struct tdx_sys_info *sysinfo) ret = ret ?: get_tdx_sys_info_td_ctrl(&sysinfo->td_ctrl); ret = ret ?: get_tdx_sys_info_td_conf(&sysinfo->td_conf); + /* + * The kernel supports using TDX without DPAMT, so + * avoid reporting failure if it's not supported. Don't + * try to support buggy TDX modules that advertise + * DPAMT but don't expose the metadata. + */ + if (!ret && tdx_supports_dynamic_pamt(sysinfo)) + ret = get_tdx_sys_info_tdmr_dpamt(&sysinfo->tdmr); + return ret; } diff --git a/arch/x86/virt/vmx/tdx/tdxcall.S b/arch/x86/virt/vmx/tdx/tdxcall.S index 016a2a1ec1d6..a194e83613e7 100644 --- a/arch/x86/virt/vmx/tdx/tdxcall.S +++ b/arch/x86/virt/vmx/tdx/tdxcall.S @@ -45,8 +45,14 @@ .macro TDX_MODULE_CALL host:req ret=0 saved=0 FRAME_BEGIN - /* Move Leaf ID to RAX */ - mov %rdi, %rax + /* Leaf ABI version -> RAX[23:16]. Zero rest of RAX. */ + movzbl TDX_MODULE_version(%rsi), %eax + shl $16, %eax + /* + * Combine leaf number arg and leaf ABI version into RAX, they don't + * overlap. + */ + or %rdi, %rax /* Move other input regs from 'struct tdx_module_args' */ movq TDX_MODULE_rcx(%rsi), %rcx diff --git a/arch/x86/xen/pmu.c b/arch/x86/xen/pmu.c index 5f50a3ee08f5..3f4dd3f50f56 100644 --- a/arch/x86/xen/pmu.c +++ b/arch/x86/xen/pmu.c @@ -456,12 +456,14 @@ static void xen_convert_regs(const struct xen_pmu_regs *xen_regs, } } +static DEFINE_PER_CPU(struct x86_perf_regs, x86_xen_intr_regs); irqreturn_t xen_pmu_irq_handler(int irq, void *dev_id) { int err, ret = IRQ_NONE; struct pt_regs regs = {0}; const struct xen_pmu_data *xenpmu_data = get_xenpmu_data(); uint8_t xenpmu_flags = get_xenpmu_flags(); + struct x86_perf_regs *x86_regs = this_cpu_ptr(&x86_xen_intr_regs); if (!xenpmu_data) { pr_warn_once("%s: pmudata not initialized\n", __func__); @@ -472,7 +474,8 @@ irqreturn_t xen_pmu_irq_handler(int irq, void *dev_id) xenpmu_flags | XENPMU_IRQ_PROCESSING; xen_convert_regs(&xenpmu_data->pmu.r.regs, ®s, xenpmu_data->pmu.pmu_flags); - if (x86_pmu.handle_irq(®s)) + x86_regs->regs = regs; + if (x86_pmu.handle_irq(&x86_regs->regs)) ret = IRQ_HANDLED; /* Write out cached context to HW */ diff --git a/drivers/base/cpu.c b/drivers/base/cpu.c index 69e52fed4241..747915ff974f 100644 --- a/drivers/base/cpu.c +++ b/drivers/base/cpu.c @@ -391,6 +391,15 @@ static int cpu_uevent(const struct device *dev, struct kobj_uevent_env *env) } #endif +#ifdef CONFIG_PREFERRED_CPU +static ssize_t preferred_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return sysfs_emit(buf, "%*pbl\n", cpumask_pr_args(cpu_preferred_mask)); +} +static DEVICE_ATTR_RO(preferred); +#endif + const struct bus_type cpu_subsys = { .name = "cpu", .dev_name = "cpu", @@ -532,6 +541,9 @@ static struct attribute *cpu_root_attrs[] = { #ifdef CONFIG_GENERIC_CPU_AUTOPROBE &dev_attr_modalias.attr, #endif +#ifdef CONFIG_PREFERRED_CPU + &dev_attr_preferred.attr, +#endif NULL }; diff --git a/drivers/clocksource/timer-clint.c b/drivers/clocksource/timer-clint.c index 0bdd9d7ec545..e56eee7e3781 100644 --- a/drivers/clocksource/timer-clint.c +++ b/drivers/clocksource/timer-clint.c @@ -243,7 +243,7 @@ static int __init clint_timer_init_dt(struct device_node *np) } #ifdef CONFIG_SMP - rc = ipi_mux_create(BITS_PER_BYTE, clint_send_ipi); + rc = ipi_mux_create(IPI_MAX, clint_send_ipi); if (rc <= 0) { pr_err("unable to create muxed IPIs\n"); rc = (rc < 0) ? rc : -ENODEV; @@ -251,7 +251,7 @@ static int __init clint_timer_init_dt(struct device_node *np) } irq_set_chained_handler(clint_ipi_irq, clint_ipi_interrupt); - riscv_ipi_set_virq_range(rc, BITS_PER_BYTE); + riscv_ipi_set_virq_range(rc, IPI_MAX); clint_clear_ipi(); #endif diff --git a/drivers/crypto/ccp/sev-dev.c b/drivers/crypto/ccp/sev-dev.c index f833cb7e4da3..5c996ab63895 100644 --- a/drivers/crypto/ccp/sev-dev.c +++ b/drivers/crypto/ccp/sev-dev.c @@ -1663,6 +1663,8 @@ static int __sev_snp_init_locked(int *error, unsigned int max_snp_asid) sev_es_tmr_size = SNP_TMR_SIZE; + snp_enable_rmpopt(); + return 0; } diff --git a/drivers/firmware/efi/libstub/x86-stub.c b/drivers/firmware/efi/libstub/x86-stub.c index cef32e2c82d8..80556a7e7552 100644 --- a/drivers/firmware/efi/libstub/x86-stub.c +++ b/drivers/firmware/efi/libstub/x86-stub.c @@ -10,6 +10,7 @@ #include <linux/pci.h> #include <linux/stddef.h> +#include <asm/cpuid/api.h> #include <asm/efi.h> #include <asm/e820/types.h> #include <asm/setup.h> @@ -17,6 +18,7 @@ #include <asm/boot.h> #include <asm/kaslr.h> #include <asm/sev.h> +#include <asm/shared/tdx.h> #include "efistub.h" #include "x86-stub.h" @@ -1068,3 +1070,40 @@ void efi64_stub_entry(efi_handle_t handle, efi_system_table_t *sys_table_arg, struct boot_params *boot_params); #endif #endif + +#ifdef CONFIG_UNACCEPTED_MEMORY +/* + * process_unaccepted_memory() is called after ExitBootServices(), and so these + * memory acceptance routines cannot rely on EFI protocols for detecting the + * presence of TDX or SEV-SNP, or emit any kind of output if any error + * conditions are detected. + */ +static bool early_is_tdx_guest(void) +{ + static bool once; + static bool is_tdx; + + if (!IS_ENABLED(CONFIG_INTEL_TDX_GUEST)) + return false; + + if (!once) { + u32 eax = TDX_CPUID_LEAF_ID, sig[3] = {}; + + native_cpuid(&eax, &sig[0], &sig[2], &sig[1]); + is_tdx = !memcmp(TDX_IDENT, sig, sizeof(sig)); + once = true; + } + + return is_tdx; +} + +void arch_accept_memory(phys_addr_t start, phys_addr_t end) +{ + if (early_is_tdx_guest()) { + if (!tdx_accept_memory(start, end)) + tdx_panic("Failed to accept memory"); + } else if (early_is_sevsnp_guest()) { + snp_accept_memory(start, end); + } +} +#endif diff --git a/drivers/irqchip/Kconfig b/drivers/irqchip/Kconfig index 20b77fbc51ee..a8f8c423b795 100644 --- a/drivers/irqchip/Kconfig +++ b/drivers/irqchip/Kconfig @@ -522,17 +522,19 @@ config GOLDFISH_PIC config QCOM_PDC tristate "Qualcomm PDC" - depends on ARCH_QCOM + depends on ARCH_QCOM || COMPILE_TEST select IRQ_DOMAIN_HIERARCHY + default ARCH_QCOM help Power Domain Controller driver to manage and configure wakeup IRQs for Qualcomm Technologies Inc (QTI) mobile chips. config QCOM_MPM tristate "Qualcomm MPM" - depends on ARCH_QCOM + depends on ARCH_QCOM || COMPILE_TEST depends on MAILBOX select IRQ_DOMAIN_HIERARCHY + default ARCH_QCOM if ARM64 help MSM Power Manager driver to manage and configure wakeup IRQs for Qualcomm Technologies Inc (QTI) mobile chips. diff --git a/drivers/irqchip/irq-aclint-sswi.c b/drivers/irqchip/irq-aclint-sswi.c index ca06efd86fa1..5e010fc401f7 100644 --- a/drivers/irqchip/irq-aclint-sswi.c +++ b/drivers/irqchip/irq-aclint-sswi.c @@ -138,7 +138,7 @@ static int __init aclint_sswi_probe(struct fwnode_handle *fwnode) } /* Register SSWI irq and handler */ - virq = ipi_mux_create(BITS_PER_BYTE, aclint_sswi_ipi_send); + virq = ipi_mux_create(IPI_MAX, aclint_sswi_ipi_send); if (virq <= 0) { pr_err("unable to create muxed IPIs\n"); irq_dispose_mapping(sswi_ipi_virq); @@ -152,7 +152,7 @@ static int __init aclint_sswi_probe(struct fwnode_handle *fwnode) aclint_sswi_starting_cpu, aclint_sswi_dying_cpu); - riscv_ipi_set_virq_range(virq, BITS_PER_BYTE); + riscv_ipi_set_virq_range(virq, IPI_MAX); return 0; } diff --git a/drivers/irqchip/irq-al-fic.c b/drivers/irqchip/irq-al-fic.c index d10ac9b63c99..760317db9dba 100644 --- a/drivers/irqchip/irq-al-fic.c +++ b/drivers/irqchip/irq-al-fic.c @@ -180,7 +180,7 @@ err_domain_remove: * @name: name of the fic * @parent_irq: interrupt of parent * - * This API will configure the fic hardware to to work in wire mode. + * This API will configure the fic hardware to work in wire mode. * In wire mode, fic hardware is generating a wire ("wired") interrupt. * Interrupt can be generated based on positive edge or level - configuration is * to be determined based on connected hardware to this fic. diff --git a/drivers/irqchip/irq-gic-v3-its.c b/drivers/irqchip/irq-gic-v3-its.c index e9807af23537..b6509b495d6d 100644 --- a/drivers/irqchip/irq-gic-v3-its.c +++ b/drivers/irqchip/irq-gic-v3-its.c @@ -2248,7 +2248,8 @@ out: static void its_lpi_free(unsigned long *bitmap, u32 base, u32 nr_ids) { - WARN_ON(free_lpi_range(base, nr_ids)); + if (free_lpi_range(base, nr_ids)) + pr_err_ratelimited("ITS: failed to free LPI range %u:%u\n", base, nr_ids); bitmap_free(bitmap); } @@ -4895,6 +4896,9 @@ static bool __maybe_unused its_enable_quirk_hip09_162100801(void *data) } static const char * const dma_32bit_impaired_platforms[] = { +#ifdef CONFIG_ALTERA_ERRATUM_AGILEX5_2_1_23 + "intel,socfpga-agilex5", +#endif #ifdef CONFIG_RENESAS_ERRATUM_GEN4GICITS1 "renesas,r8a779f0", "renesas,r8a779g0", diff --git a/drivers/irqchip/irq-gic-v3.c b/drivers/irqchip/irq-gic-v3.c index 6e1fa5b247fc..19a15c2ea61b 100644 --- a/drivers/irqchip/irq-gic-v3.c +++ b/drivers/irqchip/irq-gic-v3.c @@ -687,7 +687,7 @@ static void gic_eoimode1_eoi_irq(struct irq_data *d) { /* * No need to deactivate an LPI, or an interrupt that - * is is getting forwarded to a vcpu. + * is getting forwarded to a vcpu. */ if (irqd_to_hwirq(d) >= 8192 || irqd_is_forwarded_to_vcpu(d)) return; @@ -2276,7 +2276,6 @@ static struct bool single_redist; int enabled_rdists; u32 maint_irq; - int maint_irq_mode; phys_addr_t vcpu_base; } acpi_data __initdata; @@ -2454,21 +2453,19 @@ static int __init gic_acpi_parse_virt_madt_gicc(union acpi_subtable_headers *hea { struct acpi_madt_generic_interrupt *gicc = (struct acpi_madt_generic_interrupt *)header; - int maint_irq_mode; static int first_madt = true; if (!(gicc->flags & (ACPI_MADT_ENABLED | ACPI_MADT_GICC_ONLINE_CAPABLE))) return 0; - maint_irq_mode = (gicc->flags & ACPI_MADT_VGIC_IRQ_MODE) ? - ACPI_EDGE_SENSITIVE : ACPI_LEVEL_SENSITIVE; + if (gicc->flags & ACPI_MADT_VGIC_IRQ_MODE) + pr_warn_once(FW_BUG "MI wrongly advertised as Edge-triggered\n"); if (first_madt) { first_madt = false; acpi_data.maint_irq = gicc->vgic_interrupt; - acpi_data.maint_irq_mode = maint_irq_mode; acpi_data.vcpu_base = gicc->gicv_base_address; return 0; @@ -2478,7 +2475,6 @@ static int __init gic_acpi_parse_virt_madt_gicc(union acpi_subtable_headers *hea * The maintenance interrupt and GICV should be the same for every CPU */ if ((acpi_data.maint_irq != gicc->vgic_interrupt) || - (acpi_data.maint_irq_mode != maint_irq_mode) || (acpi_data.vcpu_base != gicc->gicv_base_address)) return -EINVAL; @@ -2511,7 +2507,7 @@ static void __init gic_acpi_setup_kvm_info(void) gic_v3_kvm_info.type = GIC_V3; irq = acpi_register_gsi(NULL, acpi_data.maint_irq, - acpi_data.maint_irq_mode, + ACPI_LEVEL_SENSITIVE, ACPI_ACTIVE_HIGH); if (irq <= 0) return; diff --git a/drivers/irqchip/irq-gic-v5.c b/drivers/irqchip/irq-gic-v5.c index 5f2551cf077d..74be71be479a 100644 --- a/drivers/irqchip/irq-gic-v5.c +++ b/drivers/irqchip/irq-gic-v5.c @@ -1168,21 +1168,18 @@ static int __init gicv5_init_common(struct fwnode_handle *parent_domain) if (ret) goto out_int; - ret = set_handle_irq(gicv5_handle_irq); + ret = gicv5_irs_enable(); if (ret) goto out_int; - ret = gicv5_irs_enable(); - if (ret) - goto out_handle; + if (set_handle_irq(gicv5_handle_irq)) + panic("GICv5: unable to install root IRQ handler\n"); gicv5_smp_init(); gicv5_irs_its_probe(); return 0; -out_handle: - set_handle_irq(NULL); out_int: gicv5_cpu_disable_interrupts(); gicv5_free_domains(); diff --git a/drivers/irqchip/irq-gic.c b/drivers/irqchip/irq-gic.c index f6bc29f515fb..b2926a3ddaf1 100644 --- a/drivers/irqchip/irq-gic.c +++ b/drivers/irqchip/irq-gic.c @@ -1527,7 +1527,6 @@ static struct { phys_addr_t cpu_phys_base; u32 maint_irq; - int maint_irq_mode; phys_addr_t vctrl_base; phys_addr_t vcpu_base; } acpi_data __initdata; @@ -1553,10 +1552,11 @@ gic_acpi_parse_madt_cpu(union acpi_subtable_headers *header, if (cpu_base_assigned && gic_cpu_base != acpi_data.cpu_phys_base) return -EINVAL; + if (processor->flags & ACPI_MADT_VGIC_IRQ_MODE) + pr_warn_once(FW_BUG "MI wrongly advertised as Edge-triggered\n"); + acpi_data.cpu_phys_base = gic_cpu_base; acpi_data.maint_irq = processor->vgic_interrupt; - acpi_data.maint_irq_mode = (processor->flags & ACPI_MADT_VGIC_IRQ_MODE) ? - ACPI_EDGE_SENSITIVE : ACPI_LEVEL_SENSITIVE; acpi_data.vctrl_base = processor->gich_base_address; acpi_data.vcpu_base = processor->gicv_base_address; @@ -1616,7 +1616,7 @@ static void __init gic_acpi_setup_kvm_info(void) vcpu_res->end = vcpu_res->start + ACPI_GICV2_VCPU_MEM_SIZE - 1; irq = acpi_register_gsi(NULL, acpi_data.maint_irq, - acpi_data.maint_irq_mode, + ACPI_LEVEL_SENSITIVE, ACPI_ACTIVE_HIGH); if (irq <= 0) return; diff --git a/drivers/irqchip/irq-lan966x-oic.c b/drivers/irqchip/irq-lan966x-oic.c index 8af08d0e4182..5122f3f1b353 100644 --- a/drivers/irqchip/irq-lan966x-oic.c +++ b/drivers/irqchip/irq-lan966x-oic.c @@ -220,7 +220,6 @@ static int lan966x_oic_probe(struct platform_device *pdev) }; struct irq_domain_info d_info = { .fwnode = of_fwnode_handle(pdev->dev.of_node), - .domain_flags = IRQ_DOMAIN_FLAG_DESTROY_GC, .size = LAN966X_OIC_NR_IRQ, .hwirq_max = LAN966X_OIC_NR_IRQ, .ops = &irq_generic_chip_ops, diff --git a/drivers/irqchip/irq-mtk-cirq.c b/drivers/irqchip/irq-mtk-cirq.c index 914d1d639fe3..d30c34ce0f56 100644 --- a/drivers/irqchip/irq-mtk-cirq.c +++ b/drivers/irqchip/irq-mtk-cirq.c @@ -247,7 +247,7 @@ static int mtk_cirq_suspend(void *data) writel_relaxed(mask, reg); } - /* set edge_only mode, record edge-triggerd interrupts */ + /* set edge_only mode, record edge-triggered interrupts */ /* enable cirq */ reg = mtk_cirq_reg(cirq_data, CIRQ_CONTROL); value = readl_relaxed(reg); diff --git a/drivers/irqchip/irq-pruss-intc.c b/drivers/irqchip/irq-pruss-intc.c index 81078d56f38d..5a3e9e5bccbe 100644 --- a/drivers/irqchip/irq-pruss-intc.c +++ b/drivers/irqchip/irq-pruss-intc.c @@ -181,7 +181,7 @@ static void pruss_intc_map(struct pruss_intc *intc, unsigned long hwirq) u8 ch, host, reg_idx; u32 val; - mutex_lock(&intc->lock); + guard(mutex)(&intc->lock); intc->event_channel[hwirq].ref_count++; @@ -206,8 +206,6 @@ static void pruss_intc_map(struct pruss_intc *intc, unsigned long hwirq) dev_dbg(dev, "mapped system_event = %lu channel = %d host = %d", hwirq, ch, host); - - mutex_unlock(&intc->lock); } /** @@ -224,7 +222,7 @@ static void pruss_intc_unmap(struct pruss_intc *intc, unsigned long hwirq) u8 ch, host, reg_idx; u32 val; - mutex_lock(&intc->lock); + guard(mutex)(&intc->lock); ch = intc->event_channel[hwirq].value; host = intc->channel_host[ch].value; @@ -251,8 +249,6 @@ static void pruss_intc_unmap(struct pruss_intc *intc, unsigned long hwirq) dev_dbg(intc->dev, "unmapped system_event = %lu channel = %d host = %d\n", hwirq, ch, host); - - mutex_unlock(&intc->lock); } static void pruss_intc_init(struct pruss_intc *intc) @@ -376,17 +372,15 @@ static int pruss_intc_validate_mapping(struct pruss_intc *intc, int event, int channel, int host) { struct device *dev = intc->dev; - int ret = 0; - mutex_lock(&intc->lock); + guard(mutex)(&intc->lock); /* check if sysevent already assigned */ if (intc->event_channel[event].ref_count > 0 && intc->event_channel[event].value != channel) { dev_err(dev, "event %d (req. ch %d) already assigned to channel %d\n", event, channel, intc->event_channel[event].value); - ret = -EBUSY; - goto unlock; + return -EBUSY; } /* check if channel already assigned */ @@ -394,16 +388,13 @@ static int pruss_intc_validate_mapping(struct pruss_intc *intc, int event, intc->channel_host[channel].value != host) { dev_err(dev, "channel %d (req. host %d) already assigned to host %d\n", channel, host, intc->channel_host[channel].value); - ret = -EBUSY; - goto unlock; + return -EBUSY; } intc->event_channel[event].value = channel; intc->channel_host[channel].value = host; -unlock: - mutex_unlock(&intc->lock); - return ret; + return 0; } static int @@ -516,24 +507,21 @@ static const char * const irq_names[MAX_NUM_HOST_IRQS] = { static int pruss_intc_probe(struct platform_device *pdev) { - const struct pruss_intc_match_data *data; struct device *dev = &pdev->dev; struct pruss_intc *intc; struct pruss_host_irq_data *host_data; int i, irq, ret; u8 max_system_events, irqs_reserved = 0; - data = of_device_get_match_data(dev); - if (!data) - return -ENODEV; - - max_system_events = data->num_system_events; - intc = devm_kzalloc(dev, sizeof(*intc), GFP_KERNEL); if (!intc) return -ENOMEM; - intc->soc_config = data; + intc->soc_config = of_device_get_match_data(dev); + if (!intc->soc_config) + return -ENODEV; + max_system_events = intc->soc_config->num_system_events; + intc->dev = dev; platform_set_drvdata(pdev, intc); @@ -553,7 +541,9 @@ static int pruss_intc_probe(struct platform_device *pdev) pruss_intc_init(intc); - mutex_init(&intc->lock); + ret = devm_mutex_init(dev, &intc->lock); + if (ret) + return ret; intc->domain = irq_domain_create_linear(dev_fwnode(dev), max_system_events, &pruss_intc_irq_domain_ops, intc); diff --git a/drivers/irqchip/irq-riscv-imsic-early.c b/drivers/irqchip/irq-riscv-imsic-early.c index 12efd241ce88..823f5f2ecb3d 100644 --- a/drivers/irqchip/irq-riscv-imsic-early.c +++ b/drivers/irqchip/irq-riscv-imsic-early.c @@ -67,12 +67,12 @@ static int __init imsic_ipi_domain_init(void) return 0; /* Create IMSIC IPI multiplexing */ - virq = ipi_mux_create(IMSIC_NR_IPI, imsic_ipi_send); + virq = ipi_mux_create(IPI_MAX, imsic_ipi_send); if (virq <= 0) return virq < 0 ? virq : -ENOMEM; /* Set vIRQ range */ - riscv_ipi_set_virq_range(virq, IMSIC_NR_IPI); + riscv_ipi_set_virq_range(virq, IPI_MAX); /* Announce that IMSIC is providing IPIs */ pr_info("%pfwP: providing IPIs using interrupt %d\n", imsic->fwnode, IMSIC_IPI_ID); diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c index b8d1bbbf42f7..9505ddbd9eec 100644 --- a/drivers/irqchip/irq-riscv-imsic-state.c +++ b/drivers/irqchip/irq-riscv-imsic-state.c @@ -7,6 +7,7 @@ #define pr_fmt(fmt) "riscv-imsic: " fmt #include <linux/acpi.h> #include <linux/cpu.h> +#include <linux/bits.h> #include <linux/bitmap.h> #include <linux/interrupt.h> #include <linux/irq.h> @@ -769,9 +770,9 @@ static int __init imsic_parse_fwnode(struct fwnode_handle *fwnode, return -EINVAL; } global->base_addr = res.start; - global->base_addr &= ~(BIT(global->guest_index_bits + - global->hart_index_bits + - IMSIC_MMIO_PAGE_SHIFT) - 1); + global->base_addr &= ~GENMASK(global->guest_index_bits + + global->hart_index_bits + + IMSIC_MMIO_PAGE_SHIFT - 1, 0); global->base_addr &= ~((BIT(global->group_index_bits) - 1) << global->group_index_shift); @@ -850,9 +851,9 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque) } base_addr = mmios[i].start; - base_addr &= ~(BIT(global->guest_index_bits + - global->hart_index_bits + - IMSIC_MMIO_PAGE_SHIFT) - 1); + base_addr &= ~GENMASK(global->guest_index_bits + + global->hart_index_bits + + IMSIC_MMIO_PAGE_SHIFT - 1, 0); base_addr &= ~((BIT(global->group_index_bits) - 1) << global->group_index_shift); if (base_addr != global->base_addr) { diff --git a/drivers/irqchip/irq-riscv-imsic-state.h b/drivers/irqchip/irq-riscv-imsic-state.h index c42ee180b305..878cc192ccec 100644 --- a/drivers/irqchip/irq-riscv-imsic-state.h +++ b/drivers/irqchip/irq-riscv-imsic-state.h @@ -13,7 +13,6 @@ #include <linux/timer.h> #define IMSIC_IPI_ID 1 -#define IMSIC_NR_IPI 8 struct imsic_vector { /* Fixed details of the vector */ diff --git a/drivers/irqchip/irq-sifive-plic.c b/drivers/irqchip/irq-sifive-plic.c index 5b0dac104814..a7cddadddf40 100644 --- a/drivers/irqchip/irq-sifive-plic.c +++ b/drivers/irqchip/irq-sifive-plic.c @@ -452,7 +452,7 @@ static irq_hw_number_t cp100_get_hwirq(struct plic_handler *handler, void __iome return 0; /* - * Interrupts delievered to hardware still become pending, but only + * Interrupts delivered to hardware still become pending, but only * interrupts that are both pending and enabled can be claimed. * Clearing the enable bit for all interrupts but the first pending * one avoids a hardware bug that occurs during read from the claim diff --git a/drivers/irqchip/irq-vic.c b/drivers/irqchip/irq-vic.c index e38104c5064e..607e3284f700 100644 --- a/drivers/irqchip/irq-vic.c +++ b/drivers/irqchip/irq-vic.c @@ -478,7 +478,7 @@ static void __init __vic_init(void __iomem *base, int parent_irq, int irq_start, /** * vic_init() - initialise a vectored interrupt controller * @base: iomem base address - * @irq_start: starting interrupt number, must be muliple of 32 + * @irq_start: starting interrupt number, must be multiple of 32 * @vic_sources: bitmask of interrupt sources to allow * @resume_sources: bitmask of interrupt sources to allow for resume */ diff --git a/drivers/irqchip/qcom-pdc.c b/drivers/irqchip/qcom-pdc.c index ce6d80c7f17a..29025a212ece 100644 --- a/drivers/irqchip/qcom-pdc.c +++ b/drivers/irqchip/qcom-pdc.c @@ -715,7 +715,10 @@ static int qcom_pdc_probe(struct platform_device *pdev, struct device_node *pare } pdc->x1e_quirk = true; + } + if (of_device_is_compatible(node, "qcom,x1e80100-pdc") || + of_device_is_compatible(node, "qcom,x1p42100-pdc")) { if (!qcom_scm_is_available()) return -EPROBE_DEFER; diff --git a/drivers/resctrl/mpam_resctrl.c b/drivers/resctrl/mpam_resctrl.c index 9d223057953a..f2c651e9f6ea 100644 --- a/drivers/resctrl/mpam_resctrl.c +++ b/drivers/resctrl/mpam_resctrl.c @@ -148,6 +148,11 @@ bool resctrl_arch_get_cdp_enabled(enum resctrl_res_level rid) return mpam_resctrl_controls[rid].cdp_enabled; } +u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val) +{ + return val; +} + /** * resctrl_reset_task_closids() - Reset the PARTID/PMG values for all tasks. * diff --git a/drivers/soc/fsl/qe/qe_ports_ic.c b/drivers/soc/fsl/qe/qe_ports_ic.c index 7375f92f528b..fb3b92039547 100644 --- a/drivers/soc/fsl/qe/qe_ports_ic.c +++ b/drivers/soc/fsl/qe/qe_ports_ic.c @@ -140,7 +140,6 @@ static int qepic_probe(struct platform_device *pdev) }; struct irq_domain_info d_info = { .fwnode = of_fwnode_handle(pdev->dev.of_node), - .domain_flags = IRQ_DOMAIN_FLAG_DESTROY_GC, .size = 32, .hwirq_max = 32, .ops = &irq_generic_chip_ops, diff --git a/drivers/virt/Kconfig b/drivers/virt/Kconfig index 52eb7e4ba71f..eeb84e578ddf 100644 --- a/drivers/virt/Kconfig +++ b/drivers/virt/Kconfig @@ -41,6 +41,23 @@ config FSL_HV_MANAGER 4) A kernel interface for receiving callbacks when a managed partition shuts down. +config STEAL_GOVERNOR + tristate "Dynamic vCPU management based on steal time" + depends on PARAVIRT && SMP + select PREFERRED_CPU + default m + help + This driver helps to reduce the steal time in paravirtualized + environments, thereby reducing vCPU preemption costs. + + By default preferred CPUs will be same as active CPUs. Depending + on the steal time when steal_governor driver is enabled, + preferred CPUs could become subset of active CPUs. + More details are at: Documentation/driver-api/steal-governor.rst + + It is recommended to build it as module and load the module + to enable it. + source "drivers/virt/vboxguest/Kconfig" source "drivers/virt/nitro_enclaves/Kconfig" diff --git a/drivers/virt/Makefile b/drivers/virt/Makefile index f29901bd7820..05fb075ef5b8 100644 --- a/drivers/virt/Makefile +++ b/drivers/virt/Makefile @@ -5,6 +5,7 @@ obj-$(CONFIG_FSL_HV_MANAGER) += fsl_hypervisor.o obj-$(CONFIG_VMGENID) += vmgenid.o +obj-$(CONFIG_STEAL_GOVERNOR) += steal_governor.o obj-y += vboxguest/ obj-$(CONFIG_NITRO_ENCLAVES) += nitro_enclaves/ diff --git a/drivers/virt/coco/tdx-guest/tdx-guest.c b/drivers/virt/coco/tdx-guest/tdx-guest.c index d0303e31e816..a21bd0376b74 100644 --- a/drivers/virt/coco/tdx-guest/tdx-guest.c +++ b/drivers/virt/coco/tdx-guest/tdx-guest.c @@ -265,7 +265,7 @@ static int wait_for_quote_completion(struct tdx_quote_buf *quote_buf, u32 timeou return (i == timeout) ? -ETIMEDOUT : 0; } -static int tdx_report_new_locked(struct tsm_report *report, void *data) +static int tdx_report_new_locked(struct tsm_report *report) { u8 *buf; struct tdx_quote_buf *quote_buf = quote_data; @@ -333,10 +333,10 @@ static int tdx_report_new_locked(struct tsm_report *report, void *data) return ret; } -static int tdx_report_new(struct tsm_report *report, void *data) +static int tdx_report_new(struct tsm_report *report, void *unused) { scoped_cond_guard(mutex_intr, return -EINTR, "e_lock) - return tdx_report_new_locked(report, data); + return tdx_report_new_locked(report); } static bool tdx_report_attr_visible(int n) diff --git a/drivers/virt/steal_governor.c b/drivers/virt/steal_governor.c new file mode 100644 index 000000000000..6e31f9923dea --- /dev/null +++ b/drivers/virt/steal_governor.c @@ -0,0 +1,296 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Steal time governor driver periodically computes steal time. + * Based on the thresholds it either reduce/increase the preferred + * CPUs which can be used by the workload to avoid vCPU preemption + * to an extent possible in paravirtualized environment. + * + * Available with CONFIG_STEAL_GOVERNOR + * + * Copyright (C) 2026 IBM + * Author: Shrikanth Hegde <sshegde@linux.ibm.com> + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt + +#include <linux/cleanup.h> +#include <linux/cpuhplock.h> +#include <linux/cpumask.h> +#include <linux/init.h> +#include <linux/kernel.h> +#include <linux/kernel_stat.h> +#include <linux/kconfig.h> +#include <linux/ktime.h> +#include <linux/math64.h> +#include <linux/module.h> +#include <linux/sched/isolation.h> +#include <linux/topology.h> +#include <linux/types.h> +#include <linux/workqueue.h> +#ifdef CONFIG_XEN +#include <xen/xen.h> +#endif + +#if !IS_ENABLED(CONFIG_PREFERRED_CPU) +#error "Steal Governor requires CONFIG_PREFERRED_CPU" +#endif + +struct steal_governor { + ktime_t time; + u64 steal; + unsigned long delay; + unsigned int interval_ms; + unsigned int high_threshold; + unsigned int low_threshold; + struct delayed_work work; +}; + +static struct steal_governor sg_ctx = { + .interval_ms = 1000, /* 1 second */ + .high_threshold = 500, /* 5% */ + .low_threshold = 200, /* 2% */ +}; + +static void restore_preferred_to_active(void) +{ + int cpu; + + guard(cpus_read_lock)(); + for_each_cpu(cpu, cpu_active_mask) + set_cpu_preferred(cpu, true); +} + +static int param_set_interval_ms(const char *val, const struct kernel_param *kp) +{ + unsigned int interval; + int ret; + + ret = kstrtouint(val, 0, &interval); + if (ret) + return ret; + + if (interval < 100 || interval > 100000) { + pr_err("interval_ms must be between 100 and 100000\n"); + return -EINVAL; + } + + return param_set_uint(val, kp); +} + +static const struct kernel_param_ops interval_ms_ops = { + .set = param_set_interval_ms, + .get = param_get_uint, +}; + +module_param_cb(interval_ms, &interval_ms_ops, &sg_ctx.interval_ms, 0444); +MODULE_PARM_DESC(interval_ms, + "Sampling frequency in milliseconds. default: 1000"); + +static int param_set_high_threshold(const char *val, const struct kernel_param *kp) +{ + unsigned int threshold; + int ret; + + ret = kstrtouint(val, 0, &threshold); + if (ret) + return ret; + + if (threshold >= 100 * 100) { + pr_err("high_threshold (%u) can't be more than 99.99%%\n", threshold); + return -EINVAL; + } + + return param_set_uint(val, kp); +} + +static const struct kernel_param_ops high_threshold_ops = { + .set = param_set_high_threshold, + .get = param_get_uint, +}; + +module_param_cb(high_threshold, &high_threshold_ops, &sg_ctx.high_threshold, 0444); +MODULE_PARM_DESC(high_threshold, + "High steal threshold. default: 500 i.e 5%. Must be > low_threshold"); + +module_param_named(low_threshold, sg_ctx.low_threshold, uint, 0444); +MODULE_PARM_DESC(low_threshold, + "Low steal threshold. default: 200 i.e 2%. Must be < high_threshold"); + +/* Return collective steal time across system. */ +static u64 get_system_steal_time(void) +{ + return kcpustat_field_total(CPUTIME_STEAL, cpu_possible_mask); +} + +/* Return number of CPUs to consider for steal ratio. */ +static unsigned int get_system_cpus(void) +{ + return num_active_cpus(); +} + +/* + * Called when the steal governor detects high physical CPU contention. + * It finds the last active core in the preferred mask and mark those + * CPUs as non-preferred. + * + * Must ensure: + * - at least one core is always kept as preferred + * - preferred is always subset of active. + */ +static void decrease_preferred_cpus(void) +{ + const struct cpumask *first_hk_core; + int target_cpu = nr_cpu_ids; + int cpu; + + guard(cpus_read_lock)(); + cpu = cpumask_first_and(housekeeping_cpumask(HK_TYPE_KERNEL_NOISE), + cpu_preferred_mask); + if (cpu >= nr_cpu_ids) + return; + + /* Always leave first housekeeping core as preferred. */ + first_hk_core = topology_sibling_cpumask(cpu); + cpu = cpumask_last(cpu_preferred_mask); + if (cpu >= nr_cpu_ids) + return; + + /* Find the last CPU which doesn't belong to that first hk_core. */ + if (!cpumask_test_cpu(cpu, first_hk_core)) { + target_cpu = cpu; + } else { + for_each_cpu_andnot(cpu, cpu_preferred_mask, first_hk_core) + target_cpu = cpu; + } + + /* Only the first housekeeping core remains */ + if (target_cpu >= nr_cpu_ids) + return; + + for_each_cpu_and(cpu, topology_sibling_cpumask(target_cpu), + cpu_preferred_mask) + set_cpu_preferred(cpu, false); +} + +/* + * Called when the steal governor detects no/low physical CPU contention. + * It finds the first active core outside of preferred mask and mark + * those CPUs as preferred. + * + * Must ensure preferred is subset of active. + */ +static void increase_preferred_cpus(void) +{ + int first_cpu, cpu; + + guard(cpus_read_lock)(); + first_cpu = cpumask_first_andnot(cpu_active_mask, cpu_preferred_mask); + + /* All CPUs are preferred. Nothing to increase further */ + if (first_cpu >= nr_cpu_ids) + return; + + for_each_cpu_and(cpu, topology_sibling_cpumask(first_cpu), + cpu_active_mask) + set_cpu_preferred(cpu, true); +} + +static bool preferred_cpus_valid(void) +{ + if (cpumask_empty(cpu_preferred_mask)) { + pr_err("empty preferred mask. stopping\n"); + return false; + } + + if (!cpumask_subset(cpu_preferred_mask, cpu_active_mask)) { + pr_err("preferred: %*pbl is not subset of active: %*pbl, stopping\n", + cpumask_pr_args(cpu_preferred_mask), + cpumask_pr_args(cpu_active_mask)); + return false; + } + + return true; +} + +static void steal_governor_loop(struct work_struct *work) +{ + u64 curr_steal, delta_steal, delta_ns, steal_ratio; + ktime_t now; + + now = ktime_get(); + delta_ns = ktime_to_ns(ktime_sub(now, sg_ctx.time)); + + if (unlikely(delta_ns < NSEC_PER_MSEC)) { + pr_err_ratelimited("work scheduled too soon delta_ns: %llu\n", delta_ns); + goto requeue_work; + } + + curr_steal = get_system_steal_time(); + delta_steal = curr_steal > sg_ctx.steal ? curr_steal - sg_ctx.steal : 0; + sg_ctx.steal = curr_steal; + sg_ctx.time = now; + + /* + * steal_ratio = (delta_steal * 100*100)/(delta_ns * num_cpus()) + * To avoid possible overflow, divide the denominator early. + * Note minimum interval is 100ms. + */ + delta_ns = max_t(u64, div_u64(delta_ns * get_system_cpus(), 10000), 1); + steal_ratio = div64_u64(delta_steal, delta_ns); + + if (steal_ratio > sg_ctx.high_threshold) + decrease_preferred_cpus(); + else if (steal_ratio <= sg_ctx.low_threshold) + increase_preferred_cpus(); + /* + * else: steal ratio is within bounds. Still do design checks so that + * module restores to active if CPU hotplug breaks those assumptions. + */ + if (!preferred_cpus_valid()) { + restore_preferred_to_active(); + return; + } + +requeue_work: + schedule_delayed_work(&sg_ctx.work, sg_ctx.delay); +} + +static int __init steal_governor_init(void) +{ +#ifdef CONFIG_XEN + if (xen_initial_domain()) { + pr_err("Cannot load in Xen Dom0 (Host OS). Driver is for guests only.\n"); + return -ENODEV; + } +#endif + + if (sg_ctx.low_threshold >= sg_ctx.high_threshold) { + pr_err("low_threshold (%u) must be less than high_threshold (%u)\n", + sg_ctx.low_threshold, sg_ctx.high_threshold); + return -EINVAL; + } + + sg_ctx.delay = msecs_to_jiffies(sg_ctx.interval_ms); + INIT_DELAYED_WORK(&sg_ctx.work, steal_governor_loop); + sg_ctx.steal = get_system_steal_time(); + sg_ctx.time = ktime_get(); + schedule_delayed_work(&sg_ctx.work, sg_ctx.delay); + pr_info("enabled. interval: %ums, high_threshold: %u, low_threshold: %u\n", + sg_ctx.interval_ms, sg_ctx.high_threshold, sg_ctx.low_threshold); + + return 0; +} + +static void __exit steal_governor_exit(void) +{ + disable_delayed_work_sync(&sg_ctx.work); + restore_preferred_to_active(); + pr_info("disabled\n"); +} + +module_init(steal_governor_init); +module_exit(steal_governor_exit); + +MODULE_LICENSE("GPL"); +MODULE_AUTHOR("IBM Corporation"); +MODULE_DESCRIPTION("Virtualization Steal Time Governor"); diff --git a/drivers/xen/time.c b/drivers/xen/time.c index a2be0a4d45b0..a02d48a2aa68 100644 --- a/drivers/xen/time.c +++ b/drivers/xen/time.c @@ -169,7 +169,7 @@ void __init xen_time_setup_guest(void) static_call_update(pv_steal_clock, xen_steal_clock); - static_key_slow_inc(¶virt_steal_enabled); + static_branch_inc(¶virt_steal_enabled); if (xen_runstate_remote) - static_key_slow_inc(¶virt_steal_rq_enabled); + static_branch_inc(¶virt_steal_rq_enabled); } @@ -1402,7 +1402,7 @@ static long read_events(struct kioctx *ctx, long min_nr, long nr, w.min_nr = min_nr - ret; ret2 = prepare_to_wait_event(&ctx->wait, &w.w, TASK_INTERRUPTIBLE); - if (!ret2 && !t.task) + if (!ret2 && !hrtimer_sleeper_task_get(&t)) ret2 = -ETIME; if (aio_read_events(ctx, min_nr, nr, event, &ret) || ret2) diff --git a/fs/exec.c b/fs/exec.c index aba8902d9423..8d2b14e1d455 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1116,17 +1116,6 @@ static struct file *bprm_identity_file(const struct linux_binprm *bprm) return bprm->file; } -static void posixtimer_exec(struct task_struct *me) -{ -#ifdef CONFIG_POSIX_TIMERS - spin_lock_irq(&me->sighand->siglock); - posix_cpu_timers_exit(me); - spin_unlock_irq(&me->sighand->siglock); - exit_itimers(me); - flush_itimer_signals(); -#endif -} - /* * Calling this is the point of no return. None of the failures will be * seen by userspace since either the process is already taking a fatal @@ -1180,7 +1169,7 @@ int begin_new_exec(struct linux_binprm * bprm) * timer would not remove an enqueued timer because the TID lookup * of the old TID fails. */ - posixtimer_exec(me); + posixtimer_exec(); /* see the comment in check_unsafe_exec() */ current->fs->in_exec = 0; diff --git a/fs/proc/uptime.c b/fs/proc/uptime.c index 433aa947cd57..53143c66cbe1 100644 --- a/fs/proc/uptime.c +++ b/fs/proc/uptime.c @@ -15,12 +15,8 @@ static int uptime_proc_show(struct seq_file *m, void *v) struct timespec64 idle; u64 idle_nsec; u32 rem; - int i; - - idle_nsec = 0; - for_each_possible_cpu(i) - idle_nsec += kcpustat_field(CPUTIME_IDLE, i); + idle_nsec = kcpustat_field_total(CPUTIME_IDLE, cpu_possible_mask); ktime_get_boottime_ts64(&uptime); timens_add_boottime(&uptime); diff --git a/fs/resctrl/ctrlmondata.c b/fs/resctrl/ctrlmondata.c index 18ec9f564b5a..cafebdff70dc 100644 --- a/fs/resctrl/ctrlmondata.c +++ b/fs/resctrl/ctrlmondata.c @@ -37,8 +37,8 @@ typedef int (ctrlval_parser_t)(struct rdt_parse_data *data, /* * Check whether MBA bandwidth percentage value is correct. The value is * checked against the minimum and max bandwidth values specified by the - * hardware. The allocated bandwidth percentage is rounded to the next - * control step available on the hardware. + * hardware. The allocated bandwidth percentage is converted as appropriate + * for consumption by the specific hardware driver. */ static bool bw_validate(char *buf, u32 *data, struct rdt_resource *r) { @@ -71,7 +71,7 @@ static bool bw_validate(char *buf, u32 *data, struct rdt_resource *r) return false; } - *data = roundup(bw, (unsigned long)r->membw.bw_gran); + *data = resctrl_arch_preconvert_bw(r, bw); return true; } diff --git a/fs/resctrl/pseudo_lock.c b/fs/resctrl/pseudo_lock.c index dea2b4bf966f..56ab63f19bad 100644 --- a/fs/resctrl/pseudo_lock.c +++ b/fs/resctrl/pseudo_lock.c @@ -750,17 +750,10 @@ static ssize_t pseudo_lock_measure_trigger(struct file *file, size_t count, loff_t *ppos) { struct rdtgroup *rdtgrp = file->private_data; - size_t buf_size; - char buf[32]; int ret; int sel; - buf_size = min(count, (sizeof(buf) - 1)); - if (copy_from_user(buf, user_buf, buf_size)) - return -EFAULT; - - buf[buf_size] = '\0'; - ret = kstrtoint(buf, 10, &sel); + ret = kstrtoint_from_user(user_buf, count, 10, &sel); if (ret == 0) { if (sel != 1 && sel != 2 && sel != 3) return -EINVAL; diff --git a/fs/resctrl/rdtgroup.c b/fs/resctrl/rdtgroup.c index 5dcbb0a964e8..68be9b903ac6 100644 --- a/fs/resctrl/rdtgroup.c +++ b/fs/resctrl/rdtgroup.c @@ -2858,7 +2858,7 @@ static int schemata_list_add(struct rdt_resource *r, enum resctrl_conf_type type { struct resctrl_schema *s; const char *suffix = ""; - int ret, cl; + int cl; s = kzalloc_obj(*s); if (!s) @@ -2882,14 +2882,12 @@ static int schemata_list_add(struct rdt_resource *r, enum resctrl_conf_type type break; } - ret = snprintf(s->name, sizeof(s->name), "%s%s", r->name, suffix); - if (ret >= sizeof(s->name)) { + cl = snprintf(s->name, sizeof(s->name), "%s%s", r->name, suffix); + if (cl >= sizeof(s->name)) { kfree(s); return -EINVAL; } - cl = strlen(s->name); - /* * If CDP is supported by this resource, but not enabled, * include the suffix. This ensures the tabular format of the diff --git a/include/asm-generic/preempt.h b/include/asm-generic/preempt.h index c8683c046615..1adddeab8545 100644 --- a/include/asm-generic/preempt.h +++ b/include/asm-generic/preempt.h @@ -96,19 +96,9 @@ static __always_inline bool should_resched(int preempt_offset) extern asmlinkage void preempt_schedule(void); extern asmlinkage void preempt_schedule_notrace(void); -#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) - -void dynamic_preempt_schedule(void); -void dynamic_preempt_schedule_notrace(void); -#define __preempt_schedule() dynamic_preempt_schedule() -#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace() - -#else /* !CONFIG_PREEMPT_DYNAMIC || !CONFIG_HAVE_PREEMPT_DYNAMIC_KEY*/ - #define __preempt_schedule() preempt_schedule() #define __preempt_schedule_notrace() preempt_schedule_notrace() -#endif /* CONFIG_PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_KEY*/ #endif /* CONFIG_PREEMPTION */ #endif /* __ASM_PREEMPT_H */ diff --git a/include/linux/bitmap.h b/include/linux/bitmap.h index 7df1573a409c..adafbcf2016b 100644 --- a/include/linux/bitmap.h +++ b/include/linux/bitmap.h @@ -52,6 +52,7 @@ struct device; * bitmap_complement(dst, src, nbits) *dst = ~(*src) * bitmap_equal(src1, src2, nbits) Are *src1 and *src2 equal? * bitmap_intersects(src1, src2, nbits) Do *src1 and *src2 overlap? + * bitmap_intersects_and(src1, src2, src3, nbits) Do *src1, *src2 and *src3 overlap? * bitmap_subset(src1, src2, nbits) Is *src1 a subset of *src2? * bitmap_empty(src, nbits) Are all bits zero in *src? * bitmap_full(src, nbits) Are all bits set in *src? @@ -181,6 +182,9 @@ void __bitmap_replace(unsigned long *dst, const unsigned long *mask, unsigned int nbits); bool __bitmap_intersects(const unsigned long *bitmap1, const unsigned long *bitmap2, unsigned int nbits); +bool __bitmap_intersects_and(const unsigned long *bitmap1, + const unsigned long *bitmap2, + const unsigned long *bitmap3, unsigned int nbits); bool __bitmap_subset(const unsigned long *bitmap1, const unsigned long *bitmap2, unsigned int nbits); unsigned int __bitmap_weight(const unsigned long *bitmap, unsigned int nbits); @@ -446,6 +450,16 @@ bool bitmap_intersects(const unsigned long *src1, const unsigned long *src2, uns } static __always_inline +bool bitmap_intersects_and(const unsigned long *src1, const unsigned long *src2, + const unsigned long *src3, unsigned int nbits) +{ + if (small_const_nbits(nbits)) + return ((*src1 & *src2 & *src3) & BITMAP_LAST_WORD_MASK(nbits)) != 0; + else + return __bitmap_intersects_and(src1, src2, src3, nbits); +} + +static __always_inline bool bitmap_subset(const unsigned long *src1, const unsigned long *src2, unsigned int nbits) { if (small_const_nbits(nbits)) diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h index 4c8bb6953107..bf89bb3f30f6 100644 --- a/include/linux/cpumask.h +++ b/include/linux/cpumask.h @@ -121,12 +121,20 @@ extern struct cpumask __cpu_enabled_mask; extern struct cpumask __cpu_present_mask; extern struct cpumask __cpu_active_mask; extern struct cpumask __cpu_dying_mask; + +#ifdef CONFIG_PREFERRED_CPU +extern struct cpumask __cpu_preferred_mask; +#else +#define __cpu_preferred_mask __cpu_active_mask +#endif + #define cpu_possible_mask ((const struct cpumask *)&__cpu_possible_mask) #define cpu_online_mask ((const struct cpumask *)&__cpu_online_mask) #define cpu_enabled_mask ((const struct cpumask *)&__cpu_enabled_mask) #define cpu_present_mask ((const struct cpumask *)&__cpu_present_mask) #define cpu_active_mask ((const struct cpumask *)&__cpu_active_mask) #define cpu_dying_mask ((const struct cpumask *)&__cpu_dying_mask) +#define cpu_preferred_mask ((const struct cpumask *)&__cpu_preferred_mask) extern atomic_t __num_online_cpus; extern unsigned int __num_possible_cpus; @@ -825,6 +833,24 @@ bool cpumask_intersects(const struct cpumask *src1p, const struct cpumask *src2p } /** + * cpumask_intersects_and - (*src1p & *src2p & *src3p) != 0 + * @src1p: the first input + * @src2p: the second input + * @src3p: the third input + * + * Return: true if AND of the three cpumasks is non-empty, + * otherwise false + */ +static __always_inline +bool cpumask_intersects_and(const struct cpumask *src1p, + const struct cpumask *src2p, + const struct cpumask *src3p) +{ + return bitmap_intersects_and(cpumask_bits(src1p), cpumask_bits(src2p), + cpumask_bits(src3p), small_cpumask_bits); +} + +/** * cpumask_subset - (*src1p & ~*src2p) == 0 * @src1p: the first input * @src2p: the second input @@ -1163,6 +1189,12 @@ void init_cpu_possible(const struct cpumask *src); #define set_cpu_active(cpu, active) assign_cpu((cpu), &__cpu_active_mask, (active)) #define set_cpu_dying(cpu, dying) assign_cpu((cpu), &__cpu_dying_mask, (dying)) +#ifdef CONFIG_PREFERRED_CPU +#define set_cpu_preferred(cpu, preferred) assign_cpu((cpu), &__cpu_preferred_mask, (preferred)) +#else +#define set_cpu_preferred(cpu, preferred) do { } while (0) +#endif + void set_cpu_online(unsigned int cpu, bool online); void set_cpu_possible(unsigned int cpu, bool possible); @@ -1257,6 +1289,11 @@ static __always_inline bool cpu_dying(unsigned int cpu) return cpumask_test_cpu(cpu, cpu_dying_mask); } +static __always_inline bool cpu_preferred(unsigned int cpu) +{ + return cpumask_test_cpu(cpu, cpu_preferred_mask); +} + #else #define num_online_cpus() 1U @@ -1295,6 +1332,11 @@ static __always_inline bool cpu_dying(unsigned int cpu) return false; } +static __always_inline bool cpu_preferred(unsigned int cpu) +{ + return cpu == 0; +} + #endif /* NR_CPUS > 1 */ #define cpu_is_offline(cpu) unlikely(!cpu_online(cpu)) diff --git a/include/linux/hrtimer.h b/include/linux/hrtimer.h index 29072d89e5cb..cad8482337cb 100644 --- a/include/linux/hrtimer.h +++ b/include/linux/hrtimer.h @@ -72,7 +72,7 @@ enum hrtimer_mode { */ struct hrtimer_sleeper { struct hrtimer timer; - struct task_struct *task; + struct task_struct *__private task; }; static inline void hrtimer_set_expires(struct hrtimer *timer, ktime_t time) @@ -320,6 +320,14 @@ extern int schedule_hrtimeout_range_clock(ktime_t *expires, const enum hrtimer_mode mode, clockid_t clock_id); extern int schedule_hrtimeout(ktime_t *expires, const enum hrtimer_mode mode); +static inline struct task_struct *hrtimer_sleeper_task_get(struct hrtimer_sleeper *sl) +{ + return READ_ONCE(ACCESS_PRIVATE(sl, task)); +} +static inline void hrtimer_sleeper_task_set(struct hrtimer_sleeper *sl, struct task_struct *t) +{ + WRITE_ONCE(ACCESS_PRIVATE(sl, task), t); +} /* Soft interrupt function to run the hrtimer queues: */ extern void hrtimer_run_queues(void); diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h index 3bf969ad8fe0..52bb684090c0 100644 --- a/include/linux/interrupt.h +++ b/include/linux/interrupt.h @@ -573,9 +573,11 @@ enum * _ IRQ_POLL: irq_poll_cpu_dead() migrates the queue * * _ (HR)TIMER_SOFTIRQ: (hr)timers_dead_cpu() migrates the queue + * + * _ BLOCK_SOFTIRQ: blk_softirq_cpu_dead() completes the remaining requests */ -#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(IRQ_POLL_SOFTIRQ) |\ - BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ)) +#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(BLOCK_SOFTIRQ) |\ + BIT(IRQ_POLL_SOFTIRQ) | BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ)) /* map softirq index to softirq name. update 'softirq_to_name' in diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h index 0bb6c03481fa..2be273bb68f0 100644 --- a/include/linux/irq-entry-common.h +++ b/include/linux/irq-entry-common.h @@ -346,22 +346,7 @@ typedef struct irqentry_state { * * Conditional reschedule with additional sanity checks. */ -void raw_irqentry_exit_cond_resched(void); - -#ifdef CONFIG_PREEMPT_DYNAMIC -#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched -#define irqentry_exit_cond_resched_dynamic_disabled NULL -DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched); -#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)() -#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched); -void dynamic_irqentry_exit_cond_resched(void); -#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched() -#endif -#else /* CONFIG_PREEMPT_DYNAMIC */ -#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched() -#endif /* CONFIG_PREEMPT_DYNAMIC */ +void irqentry_exit_cond_resched(void); /** * irqentry_enter_from_kernel_mode - Establish state before invoking the irq handler diff --git a/include/linux/kernel.h b/include/linux/kernel.h index 24414c79e59a..4b11d1dc0a67 100644 --- a/include/linux/kernel.h +++ b/include/linux/kernel.h @@ -43,30 +43,10 @@ struct completion; struct user; #ifdef CONFIG_PREEMPT_VOLUNTARY_BUILD - extern int __cond_resched(void); # define might_resched() __cond_resched() - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) - -extern int __cond_resched(void); - -DECLARE_STATIC_CALL(might_resched, __cond_resched); - -static __always_inline void might_resched(void) -{ - static_call_mod(might_resched)(); -} - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) - -extern int dynamic_might_resched(void); -# define might_resched() dynamic_might_resched() - #else - # define might_resched() do { } while (0) - #endif /* CONFIG_PREEMPT_* */ #ifdef CONFIG_DEBUG_ATOMIC_SLEEP diff --git a/include/linux/kernel_stat.h b/include/linux/kernel_stat.h index 9ca6c2259dfe..c1e85550bf12 100644 --- a/include/linux/kernel_stat.h +++ b/include/linux/kernel_stat.h @@ -196,6 +196,17 @@ static inline void kcpustat_cpu_fetch(struct kernel_cpustat *dst, int cpu) } #endif /* !CONFIG_VIRT_CPU_ACCOUNTING_GEN */ +static inline u64 kcpustat_field_total(enum cpu_usage_stat usage, const struct cpumask *cpus) +{ + u64 total = 0; + int cpu; + + for_each_cpu(cpu, cpus) + total += kcpustat_field(usage, cpu); + + return total; +} + extern void account_user_time(struct task_struct *, u64); extern void account_guest_time(struct task_struct *, u64); extern void account_system_time(struct task_struct *, int, u64); diff --git a/include/linux/list.h b/include/linux/list.h index 77fb62f79928..e3753695e76c 100644 --- a/include/linux/list.h +++ b/include/linux/list.h @@ -1172,7 +1172,7 @@ static inline void hlist_move_list(struct hlist_head *old, { new->first = old->first; if (new->first) - new->first->pprev = &new->first; + WRITE_ONCE(new->first->pprev, &new->first); old->first = NULL; } @@ -1189,10 +1189,10 @@ static inline void hlist_splice_init(struct hlist_head *from, struct hlist_head *to) { if (to->first) - to->first->pprev = &last->next; + WRITE_ONCE(to->first->pprev, &last->next); last->next = to->first; to->first = from->first; - from->first->pprev = &to->first; + WRITE_ONCE(from->first->pprev, &to->first); from->first = NULL; } diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h index 915c6fd3f084..7797ce207555 100644 --- a/include/linux/perf_event.h +++ b/include/linux/perf_event.h @@ -306,6 +306,7 @@ struct perf_event_pmu_context; #define PERF_PMU_CAP_AUX_PAUSE 0x0200 #define PERF_PMU_CAP_AUX_PREFER_LARGE 0x0400 #define PERF_PMU_CAP_MEDIATED_VPMU 0x0800 +#define PERF_PMU_CAP_SIMD_REGS 0x1000 /** * pmu::scope @@ -1467,6 +1468,7 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data, return size; } +extern u64 perf_update_xregs_size(struct perf_event *event, bool intr); extern void perf_output_sample(struct perf_output_handle *handle, struct perf_event_header *header, struct perf_sample_data *data, @@ -1517,6 +1519,27 @@ perf_event__output_id_sample(struct perf_event *event, extern void perf_log_lost_samples(struct perf_event *event, u64 lost); +static inline bool event_has_simd_regs(struct perf_event *event) +{ + struct perf_event_attr *attr = &event->attr; + + if (!(event->attr.sample_type & + (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER))) + return false; + + return attr->sample_simd_regs_enabled != 0; +} + +static inline bool event_has_extended_regs(struct perf_event *event) +{ + struct perf_event_attr *attr = &event->attr; + + return ((attr->sample_type & PERF_SAMPLE_REGS_USER) && + (attr->sample_regs_user & PERF_REG_EXTENDED_MASK)) || + ((attr->sample_type & PERF_SAMPLE_REGS_INTR) && + (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK)); +} + static inline bool event_has_any_exclude_flag(struct perf_event *event) { struct perf_event_attr *attr = &event->attr; diff --git a/include/linux/perf_regs.h b/include/linux/perf_regs.h index f632c5725f16..09dbc2fc3859 100644 --- a/include/linux/perf_regs.h +++ b/include/linux/perf_regs.h @@ -9,6 +9,16 @@ struct perf_regs { struct pt_regs *regs; }; +u64 perf_reg_value(struct pt_regs *regs, int idx); +int perf_reg_validate(u64 mask, bool simd_enabled); +u64 perf_reg_abi(struct task_struct *task); +void perf_get_regs_user(struct perf_regs *regs_user, + struct pt_regs *regs); +int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask, + u16 pred_qwords, u32 pred_mask); +u64 perf_simd_reg_value(struct pt_regs *regs, int idx, + u16 qwords_idx, bool pred); + #ifdef CONFIG_HAVE_PERF_REGS #include <asm/perf_regs.h> @@ -16,35 +26,9 @@ struct perf_regs { #define PERF_REG_EXTENDED_MASK 0 #endif -u64 perf_reg_value(struct pt_regs *regs, int idx); -int perf_reg_validate(u64 mask); -u64 perf_reg_abi(struct task_struct *task); -void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs); #else #define PERF_REG_EXTENDED_MASK 0 -static inline u64 perf_reg_value(struct pt_regs *regs, int idx) -{ - return 0; -} - -static inline int perf_reg_validate(u64 mask) -{ - return mask ? -ENOSYS : 0; -} - -static inline u64 perf_reg_abi(struct task_struct *task) -{ - return PERF_SAMPLE_REGS_ABI_NONE; -} - -static inline void perf_get_regs_user(struct perf_regs *regs_user, - struct pt_regs *regs) -{ - regs_user->regs = task_pt_regs(current); - regs_user->abi = perf_reg_abi(current); -} #endif /* CONFIG_HAVE_PERF_REGS */ #endif /* _LINUX_PERF_REGS_H */ diff --git a/include/linux/posix-timers.h b/include/linux/posix-timers.h index 9a1a0c61361c..00767acbc111 100644 --- a/include/linux/posix-timers.h +++ b/include/linux/posix-timers.h @@ -66,38 +66,6 @@ struct cpu_timer { struct task_struct __rcu *handling; }; -static inline bool cpu_timer_enqueue(struct timerqueue_head *head, - struct cpu_timer *ctmr) -{ - ctmr->head = head; - return timerqueue_add(head, &ctmr->node); -} - -static inline bool cpu_timer_queued(struct cpu_timer *ctmr) -{ - return !!ctmr->head; -} - -static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr) -{ - if (cpu_timer_queued(ctmr)) { - timerqueue_del(ctmr->head, &ctmr->node); - ctmr->head = NULL; - return true; - } - return false; -} - -static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr) -{ - return ctmr->node.expires; -} - -static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp) -{ - ctmr->node.expires = exp; -} - static inline void posix_cputimers_init(struct posix_cputimers *pct) { memset(pct, 0, sizeof(*pct)); @@ -224,14 +192,15 @@ struct k_itimer { } ____cacheline_aligned_in_smp; void run_posix_cpu_timers(void); -void posix_cpu_timers_exit(struct task_struct *task); -void posix_cpu_timers_exit_group(struct task_struct *task); void set_process_cpu_timer(struct task_struct *task, unsigned int clock_idx, u64 *newval, u64 *oldval); int update_rlimit_cpu(struct task_struct *task, unsigned long rlim_new); #ifdef CONFIG_POSIX_TIMERS +void posixtimer_exec(void); +void posixtimer_exit(bool group_dead); + static inline void posixtimer_putref(struct k_itimer *tmr) { if (rcuref_put(&tmr->rcuref)) @@ -259,6 +228,8 @@ static inline bool posixtimer_valid(const struct k_itimer *timer) return !(val & 0x1UL); } #else /* CONFIG_POSIX_TIMERS */ +static inline void posixtimer_exec(void) { } +static inline void posixtimer_exit(bool group_dead) { } static inline void posixtimer_sigqueue_getref(struct sigqueue *q) { } static inline void posixtimer_sigqueue_putref(struct sigqueue *q) { } #endif /* !CONFIG_POSIX_TIMERS */ diff --git a/include/linux/preempt.h b/include/linux/preempt.h index 2e689de7b29a..06ff44a4b8b6 100644 --- a/include/linux/preempt.h +++ b/include/linux/preempt.h @@ -498,21 +498,11 @@ DEFINE_LOCK_GUARD_0(preempt_notrace, preempt_disable_notrace(), preempt_enable_n #ifdef CONFIG_PREEMPT_DYNAMIC -extern bool preempt_model_none(void); -extern bool preempt_model_voluntary(void); extern bool preempt_model_full(void); extern bool preempt_model_lazy(void); #else -static inline bool preempt_model_none(void) -{ - return IS_ENABLED(CONFIG_PREEMPT_NONE); -} -static inline bool preempt_model_voluntary(void) -{ - return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY); -} static inline bool preempt_model_full(void) { return IS_ENABLED(CONFIG_PREEMPT); @@ -525,6 +515,16 @@ static inline bool preempt_model_lazy(void) #endif +static inline bool preempt_model_none(void) +{ + return IS_ENABLED(CONFIG_PREEMPT_NONE); +} + +static inline bool preempt_model_voluntary(void) +{ + return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY); +} + static inline bool preempt_model_rt(void) { return IS_ENABLED(CONFIG_PREEMPT_RT); diff --git a/include/linux/resctrl.h b/include/linux/resctrl.h index dd09c2ce9a0f..10dfdca7f4bf 100644 --- a/include/linux/resctrl.h +++ b/include/linux/resctrl.h @@ -505,6 +505,25 @@ bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r); */ int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable); +/** + * resctrl_arch_preconvert_bw() - Prepare bandwidth control value for arch use. + * @r: Resource whose schema was written. + * @val: Bandwidth control value written to the schemata file by userspace. + * + * Convert the user provided bandwidth control value to an appropriate form for + * consumption by the hardware driver for resource @r. Converted value is stored + * in rdt_ctrl_domain::staged_config[] for later consumption by + * resctrl_arch_update_domains(). Is not called when MBA software controller is + * enabled. + * + * Architectures for which this pre-conversion hook is not useful should supply + * an implementation of this function that just returns @val unmodified. + * + * Return: + * The converted value. + */ +u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val); + /* * Update the ctrl_val and apply this config right now. * Must be called on one of the domain's CPUs. diff --git a/include/linux/sched.h b/include/linux/sched.h index 64b7581e2012..d828f0fb1896 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -554,6 +554,7 @@ struct sched_statistics { u64 nr_failed_migrations_running; u64 nr_failed_migrations_hot; u64 nr_forced_migrations; + u64 nr_migrations_cpu_non_preferred; u64 nr_wakeups; u64 nr_wakeups_sync; @@ -2139,44 +2140,19 @@ static inline void set_need_resched_current(void) * value indicates whether a reschedule was done in fact. * cond_resched_lock() will drop the spinlock before scheduling, */ -#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC) +#if !defined(CONFIG_PREEMPTION) extern int __cond_resched(void); -#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) - -DECLARE_STATIC_CALL(cond_resched, __cond_resched); - -static __always_inline int _cond_resched(void) -{ - return static_call_mod(cond_resched)(); -} - -#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) - -extern int dynamic_cond_resched(void); - -static __always_inline int _cond_resched(void) -{ - return dynamic_cond_resched(); -} - -#else /* !CONFIG_PREEMPTION */ - static inline int _cond_resched(void) { return __cond_resched(); } - -#endif /* PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_CALL */ - -#else /* CONFIG_PREEMPTION && !CONFIG_PREEMPT_DYNAMIC */ - +#else static inline int _cond_resched(void) { return 0; } - -#endif /* !CONFIG_PREEMPTION || CONFIG_PREEMPT_DYNAMIC */ +#endif #define cond_resched() ({ \ __might_resched(__FILE__, __LINE__, 0); \ diff --git a/include/linux/sched/cputime.h b/include/linux/sched/cputime.h index e90efaf6d26e..694126411dfe 100644 --- a/include/linux/sched/cputime.h +++ b/include/linux/sched/cputime.h @@ -182,9 +182,9 @@ extern unsigned long long task_sched_runtime(struct task_struct *task); #ifdef CONFIG_PARAVIRT -struct static_key; -extern struct static_key paravirt_steal_enabled; -extern struct static_key paravirt_steal_rq_enabled; +#include <linux/jump_label.h> +DECLARE_STATIC_KEY_FALSE(paravirt_steal_enabled); +DECLARE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled); #ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN u64 dummy_steal_clock(int cpu); diff --git a/include/linux/sched/task.h b/include/linux/sched/task.h index e0c1ca8c6a18..90ed5bc3c7af 100644 --- a/include/linux/sched/task.h +++ b/include/linux/sched/task.h @@ -94,7 +94,6 @@ static inline void exit_thread(struct task_struct *tsk) extern __noreturn void do_group_exit(int); extern void exit_files(struct task_struct *); -extern void exit_itimers(struct task_struct *); extern pid_t kernel_clone(struct kernel_clone_args *kargs); struct task_struct *copy_process(struct pid *pid, int trace, int node, diff --git a/include/linux/wait.h b/include/linux/wait.h index 7e215330199c..c2af98b0074d 100644 --- a/include/linux/wait.h +++ b/include/linux/wait.h @@ -556,7 +556,7 @@ do { \ } \ \ __ret = ___wait_event(wq_head, condition, state, 0, 0, \ - if (!__t.task) { \ + if (!hrtimer_sleeper_task_get(&__t)) { \ __ret = -ETIME; \ break; \ } \ diff --git a/include/uapi/linux/perf_event.h b/include/uapi/linux/perf_event.h index fd10aa8d697f..c49fc76292f7 100644 --- a/include/uapi/linux/perf_event.h +++ b/include/uapi/linux/perf_event.h @@ -314,8 +314,9 @@ enum { */ enum perf_sample_regs_abi { PERF_SAMPLE_REGS_ABI_NONE = 0, - PERF_SAMPLE_REGS_ABI_32 = 1, - PERF_SAMPLE_REGS_ABI_64 = 2, + PERF_SAMPLE_REGS_ABI_32 = (1 << 0), + PERF_SAMPLE_REGS_ABI_64 = (1 << 1), + PERF_SAMPLE_REGS_ABI_SIMD = (1 << 2), }; /* @@ -383,6 +384,7 @@ enum perf_event_read_format { #define PERF_ATTR_SIZE_VER7 128 /* Add: sig_data */ #define PERF_ATTR_SIZE_VER8 136 /* Add: config3 */ #define PERF_ATTR_SIZE_VER9 144 /* add: config4 */ +#define PERF_ATTR_SIZE_VER10 176 /* Add: sample_simd_{vec|pred}_reg_* */ /* * 'struct perf_event_attr' contains various attributes that define @@ -547,6 +549,29 @@ struct perf_event_attr { __u64 config3; /* extension of config2 */ __u64 config4; /* extension of config3 */ + + /* + * Defines the sampling SIMD/PRED(predicate) registers bitmap and + * qwords (8 bytes) length. + * + * sample_simd_regs_enabled != 0 indicates there are SIMD/PRED + * registers to be sampled, the SIMD/PRED registers bitmap and + * qwords length are represented in + * sample_simd_{vec|pred}_reg_{intr|user} and + * sample_simd_{vec|pred}_reg_qwords fields separately. + * + * sample_simd_regs_enabled == 0 indicates no SIMD/PRED registers + * are sampled. + */ + __u16 sample_simd_regs_enabled; + __u16 sample_simd_pred_reg_qwords; + __u16 sample_simd_vec_reg_qwords; + __u16 __reserved_4; + + __u32 sample_simd_pred_reg_intr; + __u32 sample_simd_pred_reg_user; + __u64 sample_simd_vec_reg_intr; + __u64 sample_simd_vec_reg_user; }; /* @@ -1020,7 +1045,15 @@ enum perf_event_type { * } && PERF_SAMPLE_BRANCH_STACK * * { u64 abi; # enum perf_sample_regs_abi - * u64 regs[weight(mask)]; } && PERF_SAMPLE_REGS_USER + * u64 regs[weight(mask)]; + * struct { + * u64 nr_vectors; # 0 ... weight(sample_simd_vec_reg_user) + * u64 vector_qwords; # 0 ... sample_simd_vec_reg_qwords + * u64 nr_pred; # 0 ... weight(sample_simd_pred_reg_user) + * u64 pred_qwords; # 0 ... sample_simd_pred_reg_qwords + * u64 data[nr_vectors * vector_qwords + nr_pred * pred_qwords]; + * } && (abi & PERF_SAMPLE_REGS_ABI_SIMD) + * } && PERF_SAMPLE_REGS_USER * * { u64 size; * char data[size]; @@ -1047,7 +1080,15 @@ enum perf_event_type { * { u64 data_src; } && PERF_SAMPLE_DATA_SRC * { u64 transaction; } && PERF_SAMPLE_TRANSACTION * { u64 abi; # enum perf_sample_regs_abi - * u64 regs[weight(mask)]; } && PERF_SAMPLE_REGS_INTR + * u64 regs[weight(mask)]; + * struct { + * u64 nr_vectors; # 0 ... weight(sample_simd_vec_reg_intr) + * u64 vector_qwords; # 0 ... sample_simd_vec_reg_qwords + * u64 nr_pred; # 0 ... weight(sample_simd_pred_reg_intr) + * u64 pred_qwords; # 0 ... sample_simd_pred_reg_qwords + * u64 data[nr_vectors * vector_qwords + nr_pred * pred_qwords]; + * } && (abi & PERF_SAMPLE_REGS_ABI_SIMD) + * } && PERF_SAMPLE_REGS_INTR * { u64 phys_addr;} && PERF_SAMPLE_PHYS_ADDR * { u64 cgroup;} && PERF_SAMPLE_CGROUP * { u64 data_page_size;} && PERF_SAMPLE_DATA_PAGE_SIZE diff --git a/include/uapi/linux/sched.h b/include/uapi/linux/sched.h index 33a4624285cd..19ffeba89428 100644 --- a/include/uapi/linux/sched.h +++ b/include/uapi/linux/sched.h @@ -53,7 +53,7 @@ */ #define UNSHARE_EMPTY_MNTNS 0x00100000 /* Unshare an empty mount namespace. */ -#ifndef __ASSEMBLY__ +#ifndef __ASSEMBLER__ /** * struct clone_args - arguments for the clone3 syscall * @flags: Flags for the new process as listed above. diff --git a/include/vdso/math64.h b/include/vdso/math64.h index 22ae212f8b28..55b45f5cf615 100644 --- a/include/vdso/math64.h +++ b/include/vdso/math64.h @@ -2,15 +2,36 @@ #ifndef __VDSO_MATH64_H #define __VDSO_MATH64_H -static __always_inline u32 -__iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder) +static __always_inline u32 __iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder) { u32 ret = 0; while (dividend >= divisor) { - /* The following asm() prevents the compiler from - optimising this loop into a modulo operation. */ - asm("" : "+rm"(dividend)); + /* + * Prevent the compiler from optimising this loop into a + * modulo operation. + */ + OPTIMIZER_HIDE_VAR(dividend); + + dividend -= divisor; + ret++; + } + + *remainder = dividend; + + return ret; +} + +static __always_inline u32 __iter_div64_u64_rem(u64 dividend, u64 divisor, u64 *remainder) +{ + u32 ret = 0; + + while (dividend >= divisor) { + /* + * Prevent the compiler from optimising this loop into a + * modulo operation. + */ + OPTIMIZER_HIDE_VAR(dividend); dividend -= divisor; ret++; diff --git a/io_uring/rw.c b/io_uring/rw.c index 0c9494fd21be..755166e90746 100644 --- a/io_uring/rw.c +++ b/io_uring/rw.c @@ -1296,7 +1296,7 @@ static u64 io_hybrid_iopoll_delay(struct io_ring_ctx *ctx, struct io_kiocb *req) set_current_state(TASK_INTERRUPTIBLE); hrtimer_sleeper_start_expires(&timer, mode); - if (timer.task) + if (hrtimer_sleeper_task_get(&timer)) io_schedule(); hrtimer_cancel(&timer.timer); diff --git a/kernel/Kconfig.kexec b/kernel/Kconfig.kexec index 15632358bcf7..a97ed9605602 100644 --- a/kernel/Kconfig.kexec +++ b/kernel/Kconfig.kexec @@ -167,7 +167,7 @@ config CRASH_MAX_MEMORY_RANGES memory regions that the elfcorehdr buffer/segment can accommodate. These regions are obtained via walk_system_ram_res(); eg. the 'System RAM' entries in /proc/iomem. - This value is combined with NR_CPUS_DEFAULT and multiplied by + This value is combined with NR_CPUS and multiplied by sizeof(Elf64_Phdr) to determine the final elfcorehdr memory buffer/ segment size. The value 8192, for example, covers a (sparsely populated) 1TiB system diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt index f294dad43bd7..edc067a0c422 100644 --- a/kernel/Kconfig.preempt +++ b/kernel/Kconfig.preempt @@ -132,10 +132,9 @@ config PREEMPTION config PREEMPT_DYNAMIC bool "Preemption behaviour defined on boot" - depends on HAVE_PREEMPT_DYNAMIC - select JUMP_LABEL if HAVE_PREEMPT_DYNAMIC_KEY + depends on ARCH_HAS_PREEMPT_LAZY select PREEMPT_BUILD - default y if HAVE_PREEMPT_DYNAMIC_CALL + default y help This option allows to define the preemption model on the kernel command line parameter and thus override the default preemption @@ -145,9 +144,7 @@ config PREEMPT_DYNAMIC provide a pre-built kernel binary to reduce the number of kernel flavors they offer while still offering different usecases. - The runtime overhead is negligible with HAVE_STATIC_CALL_INLINE enabled - but if runtime patching is not available for the specific architecture - then the potential overhead should be considered. + The runtime overhead is negligible. Interesting if you want the same pre-built kernel should be used for both Server and Desktop workloads. @@ -197,3 +194,7 @@ config SCHED_CLASS_EXT For more information: Documentation/scheduler/sched-ext.rst https://github.com/sched-ext/scx + +config PREFERRED_CPU + bool + depends on SMP && PARAVIRT diff --git a/kernel/cpu.c b/kernel/cpu.c index b3c8553d7bd6..376d297a6292 100644 --- a/kernel/cpu.c +++ b/kernel/cpu.c @@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask); atomic_t __num_online_cpus __read_mostly; EXPORT_SYMBOL(__num_online_cpus); +#ifdef CONFIG_PREFERRED_CPU +struct cpumask __cpu_preferred_mask __read_mostly; +EXPORT_SYMBOL_GPL(__cpu_preferred_mask); +#endif + void init_cpu_present(const struct cpumask *src) { cpumask_copy(&__cpu_present_mask, src); @@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void) /* Mark the boot cpu "present", "online" etc for SMP and UP case */ set_cpu_online(cpu, true); set_cpu_active(cpu, true); + set_cpu_preferred(cpu, true); set_cpu_present(cpu, true); set_cpu_possible(cpu, true); diff --git a/kernel/crash_core.c b/kernel/crash_core.c index 2b36aa9fade0..d0bd2d0cf899 100644 --- a/kernel/crash_core.c +++ b/kernel/crash_core.c @@ -648,7 +648,7 @@ int crash_check_hotplug_support(void) * new list of CPUs and memory. To make changes to the elfcorehdr, it * should be large enough to permit a growing number of CPU and Memory * resources. One can estimate the elfcorehdr memory size based on - * NR_CPUS_DEFAULT and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is + * NR_CPUS and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is * excluded from SHA verification by default if the architecture * supports crash hotplug. */ diff --git a/kernel/entry/common.c b/kernel/entry/common.c index e3d381fd3d25..e234b04373fe 100644 --- a/kernel/entry/common.c +++ b/kernel/entry/common.c @@ -123,7 +123,7 @@ noinstr irqentry_state_t irqentry_enter(struct pt_regs *regs) /** * arch_irqentry_exit_need_resched - Architecture specific need resched function * - * Invoked from raw_irqentry_exit_cond_resched() to check if resched is needed. + * Invoked from irqentry_exit_cond_resched() to check if resched is needed. * Defaults return true. * * The main purpose is to permit arch to avoid preemption of a task from an IRQ. @@ -134,7 +134,7 @@ static inline bool arch_irqentry_exit_need_resched(void); static inline bool arch_irqentry_exit_need_resched(void) { return true; } #endif -void raw_irqentry_exit_cond_resched(void) +void irqentry_exit_cond_resched(void) { if (!preempt_count()) { /* Sanity check RCU and thread stack */ @@ -145,19 +145,6 @@ void raw_irqentry_exit_cond_resched(void) preempt_schedule_irq(); } } -#ifdef CONFIG_PREEMPT_DYNAMIC -#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -DEFINE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched); -#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -DEFINE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched); -void dynamic_irqentry_exit_cond_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_irqentry_exit_cond_resched)) - return; - raw_irqentry_exit_cond_resched(); -} -#endif -#endif noinstr void irqentry_exit(struct pt_regs *regs, irqentry_state_t state) { diff --git a/kernel/events/core.c b/kernel/events/core.c index 601e8d944c24..a34ff4cb410d 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -7828,22 +7828,82 @@ unsigned long perf_instruction_pointer(struct perf_event *event, 0 : perf_arch_instruction_pointer(regs); } +u64 __weak perf_reg_value(struct pt_regs *regs, int idx) +{ + return 0; +} + +int __weak perf_reg_validate(u64 mask, bool simd_enabled) +{ + return mask ? -ENOSYS : 0; +} + +u64 __weak perf_reg_abi(struct task_struct *task) +{ + return PERF_SAMPLE_REGS_ABI_NONE; +} + +void __weak perf_get_regs_user(struct perf_regs *regs_user, + struct pt_regs *regs) +{ + regs_user->regs = task_pt_regs(current); + regs_user->abi = perf_reg_abi(current); +} + +#define word_for_each_set_bit(bit, val) \ + for (unsigned long long __v = (val); \ + __v && ((bit = __builtin_ctzll(__v)), 1); \ + __v &= __v - 1) + static void perf_output_sample_regs(struct perf_output_handle *handle, struct pt_regs *regs, u64 mask) { int bit; - DECLARE_BITMAP(_mask, 64); - bitmap_from_u64(_mask, mask); - for_each_set_bit(bit, _mask, sizeof(mask) * BITS_PER_BYTE) { - u64 val; - - val = perf_reg_value(regs, bit); + word_for_each_set_bit(bit, mask) { + u64 val = perf_reg_value(regs, bit); perf_output_put(handle, val); } } +static void +perf_output_sample_simd_regs(struct perf_output_handle *handle, + struct perf_event *event, + struct pt_regs *regs, + u64 mask, u32 pred_mask) +{ + u64 pred_qwords = event->attr.sample_simd_pred_reg_qwords; + u64 vec_qwords = event->attr.sample_simd_vec_reg_qwords; + u64 nr_vectors = hweight64(mask); + u64 nr_pred = hweight32(pred_mask); + int bit; + + perf_output_put(handle, nr_vectors); + perf_output_put(handle, vec_qwords); + perf_output_put(handle, nr_pred); + perf_output_put(handle, pred_qwords); + + if (nr_vectors) { + word_for_each_set_bit(bit, mask) { + for (int i = 0; i < vec_qwords; i++) { + u64 val = perf_simd_reg_value(regs, bit, + i, false); + perf_output_put(handle, val); + } + } + } + if (nr_pred) { + word_for_each_set_bit(bit, pred_mask) { + for (int i = 0; i < pred_qwords; i++) { + u64 val = perf_simd_reg_value(regs, bit, + i, true); + perf_output_put(handle, val); + } + } + } +} + static void perf_sample_regs_user(struct perf_regs *regs_user, struct pt_regs *regs) { @@ -7877,6 +7937,17 @@ static void perf_sample_regs_intr(struct perf_regs *regs_intr, } } +int __weak perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask, + u16 pred_qwords, u32 pred_mask) +{ + return -EINVAL; +} + +u64 __weak perf_simd_reg_value(struct pt_regs *regs, int idx, + u16 qwords_idx, bool pred) +{ + return 0; +} /* * Get remaining task size from user stack pointer. @@ -8407,10 +8478,17 @@ void perf_output_sample(struct perf_output_handle *handle, perf_output_put(handle, abi); if (abi) { - u64 mask = event->attr.sample_regs_user; + struct perf_event_attr *attr = &event->attr; + u64 mask = attr->sample_regs_user; perf_output_sample_regs(handle, data->regs_user.regs, mask); + if (abi & PERF_SAMPLE_REGS_ABI_SIMD) { + perf_output_sample_simd_regs(handle, event, + data->regs_user.regs, + attr->sample_simd_vec_reg_user, + attr->sample_simd_pred_reg_user); + } } } @@ -8438,11 +8516,18 @@ void perf_output_sample(struct perf_output_handle *handle, perf_output_put(handle, abi); if (abi) { - u64 mask = event->attr.sample_regs_intr; + struct perf_event_attr *attr = &event->attr; + u64 mask = attr->sample_regs_intr; perf_output_sample_regs(handle, data->regs_intr.regs, mask); + if (abi & PERF_SAMPLE_REGS_ABI_SIMD) { + perf_output_sample_simd_regs(handle, event, + data->regs_intr.regs, + attr->sample_simd_vec_reg_intr, + attr->sample_simd_pred_reg_intr); + } } } @@ -8645,6 +8730,29 @@ static __always_inline u64 __cond_set(u64 flags, u64 s, u64 d) return d * !!(flags & s); } +u64 perf_update_xregs_size(struct perf_event *event, bool intr) +{ + u16 pred_qwords = event->attr.sample_simd_pred_reg_qwords; + u16 vec_qwords = event->attr.sample_simd_vec_reg_qwords; + u64 pred_mask; + u64 mask; + int size; + + if (intr) { + mask = event->attr.sample_simd_vec_reg_intr; + pred_mask = event->attr.sample_simd_pred_reg_intr; + } else { + mask = event->attr.sample_simd_vec_reg_user; + pred_mask = event->attr.sample_simd_pred_reg_user; + } + + size = sizeof(u64) * 4; + size += (hweight64(mask) * vec_qwords + + hweight64(pred_mask) * pred_qwords) * sizeof(u64); + + return size; +} + void perf_prepare_sample(struct perf_sample_data *data, struct perf_event *event, struct pt_regs *regs) @@ -8707,7 +8815,12 @@ void perf_prepare_sample(struct perf_sample_data *data, if (data->regs_user.regs) { u64 mask = event->attr.sample_regs_user; + size += hweight64(mask) * sizeof(u64); + if (event_has_simd_regs(event)) { + size += perf_update_xregs_size(event, false); + data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } } data->dyn_size += size; @@ -8772,6 +8885,10 @@ void perf_prepare_sample(struct perf_sample_data *data, u64 mask = event->attr.sample_regs_intr; size += hweight64(mask) * sizeof(u64); + if (event_has_simd_regs(event)) { + size += perf_update_xregs_size(event, true); + data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } } data->dyn_size += size; @@ -13116,12 +13233,6 @@ int perf_pmu_unregister(struct pmu *pmu) } EXPORT_SYMBOL_GPL(perf_pmu_unregister); -static inline bool has_extended_regs(struct perf_event *event) -{ - return (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK) || - (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK); -} - static int perf_try_init_event(struct pmu *pmu, struct perf_event *event) { struct perf_event_context *ctx = NULL; @@ -13155,8 +13266,14 @@ static int perf_try_init_event(struct pmu *pmu, struct perf_event *event) if (ret) goto err_pmu; + if (!(pmu->capabilities & PERF_PMU_CAP_SIMD_REGS) && + event_has_simd_regs(event)) { + ret = -EOPNOTSUPP; + goto err_destroy; + } + if (!(pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS) && - has_extended_regs(event)) { + event_has_extended_regs(event)) { ret = -EOPNOTSUPP; goto err_destroy; } @@ -13650,7 +13767,8 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, attr->size = size; - if (attr->__reserved_1 || attr->__reserved_2 || attr->__reserved_3) + if (attr->__reserved_1 || attr->__reserved_2 || + attr->__reserved_3 || attr->__reserved_4) return -EINVAL; if (attr->sample_type & ~(PERF_SAMPLE_MAX-1)) @@ -13696,9 +13814,18 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, } if (attr->sample_type & PERF_SAMPLE_REGS_USER) { - ret = perf_reg_validate(attr->sample_regs_user); + ret = perf_reg_validate(attr->sample_regs_user, + attr->sample_simd_regs_enabled); if (ret) return ret; + if (attr->sample_simd_regs_enabled) { + ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords, + attr->sample_simd_vec_reg_user, + attr->sample_simd_pred_reg_qwords, + attr->sample_simd_pred_reg_user); + if (ret) + return ret; + } } if (attr->sample_type & PERF_SAMPLE_STACK_USER) { @@ -13719,8 +13846,20 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, if (!attr->sample_max_stack) attr->sample_max_stack = sysctl_perf_event_max_stack; - if (attr->sample_type & PERF_SAMPLE_REGS_INTR) - ret = perf_reg_validate(attr->sample_regs_intr); + if (attr->sample_type & PERF_SAMPLE_REGS_INTR) { + ret = perf_reg_validate(attr->sample_regs_intr, + attr->sample_simd_regs_enabled); + if (ret) + return ret; + if (attr->sample_simd_regs_enabled) { + ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords, + attr->sample_simd_vec_reg_intr, + attr->sample_simd_pred_reg_qwords, + attr->sample_simd_pred_reg_intr); + if (ret) + return ret; + } + } #ifndef CONFIG_CGROUP_PERF if (attr->sample_type & PERF_SAMPLE_CGROUP) diff --git a/kernel/exit.c b/kernel/exit.c index 29e853a36602..9ff1fa7b30ea 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -168,12 +168,6 @@ static void __exit_signal(struct release_task_post *post, struct task_struct *ts lockdep_tasklist_lock_is_held()); spin_lock(&sighand->siglock); -#ifdef CONFIG_POSIX_TIMERS - posix_cpu_timers_exit(tsk); - if (group_dead) - posix_cpu_timers_exit_group(tsk); -#endif - if (group_dead) { tty = sig->tty; sig->tty = NULL; @@ -940,13 +934,12 @@ void __noreturn do_exit(long code) panic("Attempted to kill init! exitcode=0x%08x\n", tsk->signal->group_exit_code ?: (int)code); -#ifdef CONFIG_POSIX_TIMERS - hrtimer_cancel(&tsk->signal->real_timer); - exit_itimers(tsk); -#endif if (tsk->mm) setmax_mm_hiwater_rss(&tsk->signal->maxrss, tsk->mm); } + + posixtimer_exit(group_dead); + acct_collect(code, group_dead); if (group_dead) tty_audit_exit(); diff --git a/kernel/futex/requeue.c b/kernel/futex/requeue.c index b3f4a4bccb12..842d852302dd 100644 --- a/kernel/futex/requeue.c +++ b/kernel/futex/requeue.c @@ -744,7 +744,7 @@ int handle_early_requeue_pi_wakeup(struct futex_hash_bucket *hb, /* Handle spurious wakeups gracefully */ ret = -EWOULDBLOCK; - if (timeout && !timeout->task) + if (timeout && !hrtimer_sleeper_task_get(timeout)) ret = -ETIMEDOUT; else if (signal_pending(current)) ret = -ERESTARTNOINTR; diff --git a/kernel/futex/waitwake.c b/kernel/futex/waitwake.c index d4483d15d30a..cf18309e5770 100644 --- a/kernel/futex/waitwake.c +++ b/kernel/futex/waitwake.c @@ -383,7 +383,7 @@ void futex_do_wait(struct futex_q *q, struct hrtimer_sleeper *timeout) * flagged for rescheduling. Only call schedule if there * is no timeout, or if it has yet to expire. */ - if (!timeout || timeout->task) + if (!timeout || hrtimer_sleeper_task_get(timeout)) schedule(); } __set_current_state(TASK_RUNNING); @@ -539,7 +539,7 @@ retry: static void futex_sleep_multiple(struct futex_vector *vs, unsigned int count, struct hrtimer_sleeper *to) { - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return; for (; count; count--, vs++) { @@ -590,7 +590,7 @@ int futex_wait_multiple(struct futex_vector *vs, unsigned int count, if (ret >= 0) return ret; - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return -ETIMEDOUT; else if (signal_pending(current)) return -ERESTARTSYS; @@ -725,7 +725,7 @@ retry: if (!futex_unqueue(&q)) return 0; - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return -ETIMEDOUT; /* diff --git a/kernel/irq/irqdomain.c b/kernel/irq/irqdomain.c index 57c819da30c2..4fdcb6df5306 100644 --- a/kernel/irq/irqdomain.c +++ b/kernel/irq/irqdomain.c @@ -344,6 +344,7 @@ static struct irq_domain *__irq_domain_instantiate(const struct irq_domain_info err = irq_domain_alloc_generic_chips(domain, info->dgc_info); if (err) goto err_domain_free; + domain->flags |= IRQ_DOMAIN_FLAG_DESTROY_GC; } if (info->init) { diff --git a/kernel/locking/rtmutex.c b/kernel/locking/rtmutex.c index 4728631ae719..5a9534c715b8 100644 --- a/kernel/locking/rtmutex.c +++ b/kernel/locking/rtmutex.c @@ -1644,7 +1644,7 @@ static int __sched rt_mutex_slowlock_block(struct rt_mutex_base *lock, break; } - if (timeout && !timeout->task) { + if (timeout && !hrtimer_sleeper_task_get(timeout)) { ret = -ETIMEDOUT; break; } diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 1fe40de6ebe3..84313c9c4ba9 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf) /* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_rq_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled); #endif static void update_rq_clock_task(struct rq *rq, s64 delta) @@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta) } #endif #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING - if (static_key_false((¶virt_steal_rq_enabled))) { + if (static_branch_unlikely(¶virt_steal_rq_enabled)) { u64 prev_steal; steal = prev_steal = paravirt_steal_clock(cpu_of(rq)); @@ -2252,7 +2252,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags) dequeue_task(rq, p, flags); } -static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +static bool dequeue_block_task(struct rq *rq, struct task_struct *p, + unsigned long task_state) { int flags = DEQUEUE_NOCLOCK; @@ -2273,9 +2274,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_ * * Where __schedule() and ttwu() have matching control dependencies. * - * After this, schedule() must not care about p->state any more. + * Once the caller invokes __block_task(), schedule() must not care about + * p->state any more. */ - if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags)) + return dequeue_task(rq, p, DEQUEUE_SLEEP | flags); +} + +static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +{ + if (dequeue_block_task(rq, p, task_state)) __block_task(rq, p); } @@ -2504,6 +2511,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq) return rq->nr_pinned; } +static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu) +{ + /* No need to migrate from a preferred CPU */ + if (cpu_preferred(cpu)) + return false; + + /* Only FAIR tasks honor preferred CPU state */ + if (unlikely(p->sched_class != &fair_sched_class)) + return false; + + /* Ignore preferred state if task affinity is changing */ + if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr))) + return false; + + return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask, + task_cpu_possible_mask(p)); +} + /* * Per-CPU kthreads are allowed to run on !active && online CPUs, see * __set_cpus_allowed_ptr() and select_fallback_rq(). @@ -2519,8 +2544,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) return cpu_online(cpu); /* Non kernel threads are not allowed during either online or offline. */ - if (!(p->flags & PF_KTHREAD)) + if (!(p->flags & PF_KTHREAD)) { + /* Try to use preferred CPU if task's affinity allows */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; return cpu_active(cpu); + } /* KTHREAD_IS_PER_CPU is always allowed. */ if (kthread_is_per_cpu(p)) @@ -2530,7 +2559,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) if (cpu_dying(cpu)) return false; - /* But are allowed during online. */ + /* Try to keep unbound kthreads on a preferred CPU if possible. */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; + + /* Otherwise, they are allowed to run on online CPU. */ return cpu_online(cpu); } @@ -3773,6 +3806,7 @@ static inline void proxy_reset_donor(struct rq *rq) WARN_ON_ONCE(rq->donor == rq->curr); put_prev_set_next_task(rq, rq->donor, rq->curr); + rq->next_class = rq->curr->sched_class; rq_set_donor(rq, rq->curr); zap_balance_callbacks(rq); resched_curr(rq); @@ -3787,6 +3821,8 @@ static inline void proxy_reset_donor(struct rq *rq) */ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) { + bool dequeued; + /* * Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu. * @@ -3809,12 +3845,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) /* If already current, don't need to return migrate */ if (task_current(rq, p)) return false; - - /* If we're return migrating the rq->donor, switch it out for idle */ - if (task_current_donor(rq, p)) - proxy_reset_donor(rq); } - block_task(rq, p, TASK_WAKING); + + dequeued = dequeue_block_task(rq, p, TASK_WAKING); + + /* + * Dequeue @p from its scheduling class before resetting rq->donor. + * In particular, sched_ext needs to end the donor's running session + * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by + * proxy_reset_donor(); otherwise it would reenqueue the blocked donor. + * + * Keep on_rq set until all donor references have been replaced. + */ + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (dequeued) + __block_task(rq, p); return true; } #else /* !CONFIG_SCHED_PROXY_EXEC */ @@ -3905,7 +3952,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags) * When on_rq && !on_cpu the task is preempted, see if * it should preempt the task that is current now. */ - wakeup_preempt(rq, p, wake_flags); + wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ); } ttwu_do_wakeup(p); return 1; @@ -5149,7 +5196,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head) lockdep_assert_rq_held(rq); while (head) { - func = (void (*)(struct rq *))head->func; + func = head->func; next = head->next; head->next = NULL; head = next; @@ -5789,6 +5836,9 @@ void sched_tick(void) unsigned long hw_pressure; u64 resched_latency; + if (!cpu_preferred(cpu)) + sched_push_current_non_preferred_cpu(rq); + if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE)) arch_scale_freq_tick(); @@ -6283,10 +6333,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf) * selection. In this case, do a core-wide selection. */ if (rq->core->core_pick_seq == rq->core->core_task_seq && - rq->core->core_pick_seq != rq->core_sched_seq && rq->core_pick) { - WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq); - next = rq->core_pick; rq->dl_server = rq->core_dl_server; rq->core_pick = NULL; @@ -6318,11 +6365,13 @@ restart: } /* - * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq + * core->core_task_seq, core->core_pick_seq * * @task_seq guards the task state ({en,de}queues) * @pick_seq is the @task_seq we did a selection on - * @sched_seq is the @pick_seq we scheduled + * + * Once a core-wide selection is committed, a non-NULL core_pick denotes + * a pick which still needs to be consumed on this CPU. * * However, preemptions can cause multiple picks on the same task set. * 'Fix' this by also increasing @task_seq for every pick. @@ -6429,7 +6478,6 @@ restart: rq->core->core_pick_seq = rq->core->core_task_seq; next = rq->core_pick; - rq->core_sched_seq = rq->core->core_pick_seq; /* Something should have been selected for current CPU */ WARN_ON_ONCE(!next); @@ -6517,7 +6565,10 @@ static bool try_steal_cookie(int this, int that) return false; do { - if (p == src->core_pick || p == src->curr) + if (p == src->core_pick || p == src->curr || p == src->donor) + goto next; + + if (task_is_blocked(p)) goto next; if (!is_cpu_allowed(p, this)) @@ -6820,6 +6871,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor) block_task(rq, donor, state); } +/* + * Remove a retained proxy donor before changing its scheduler ownership. + * The caller holds p->pi_lock, so p cannot wake and migrate if block_task() + * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task + * queued, switching_from_fair() completes the dequeue in the immediately + * following sched_change_begin(). + */ +void sched_proxy_block_task(struct rq *rq, struct task_struct *p) +{ + unsigned long state = READ_ONCE(p->__state); + + lockdep_assert_held(&p->pi_lock); + lockdep_assert_rq_held(rq); + + if (!p->is_blocked || !task_on_rq_queued(p)) + return; + if (WARN_ON_ONCE(state == TASK_RUNNING)) + return; + + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (!p->se.sched_delayed) + block_task(rq, p, state); + + WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed); +} + static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf) __releases(__rq_lockp(rq)) { @@ -6865,9 +6944,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, __must_hold(__rq_lockp(rq)) { struct rq *target_rq = cpu_rq(target_cpu); + LIST_HEAD(migrate_list); lockdep_assert_rq_held(rq); - WARN_ON(p == rq->curr); /* * Since we are migrating a blocked donor, it could be rq->donor, * and we want to make sure there aren't any references from this @@ -6880,13 +6959,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, * before we release the lock. */ proxy_resched_idle(rq); - - deactivate_task(rq, p, DEQUEUE_NOCLOCK); - proxy_set_task_cpu(p, target_cpu); - + for (; p; p = p->blocked_donor) { + WARN_ON(p == rq->curr); + deactivate_task(rq, p, DEQUEUE_NOCLOCK); + proxy_set_task_cpu(p, target_cpu); + /* + * We can re-use se.group_node to migrate the thing, + * because @p is deactivated (won't be balanced) and + * we hold the rq_lock. + */ + list_add(&p->se.group_node, &migrate_list); + } proxy_release_rq_lock(rq, rf); - attach_one_task(target_rq, p); + __attach_tasks(target_rq, &migrate_list); proxy_reacquire_rq_lock(rq, rf); } @@ -6979,7 +7065,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) { /* XXX Don't handle blocked owners/delayed dequeue yet */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; __clear_task_blocked_on(p, NULL); goto deactivate; } @@ -6991,7 +7077,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * and leave that CPU to sort things out. */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; goto migrate_task; } @@ -7004,7 +7090,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * case we should end up back in find_proxy_task(), this time * hopefully with all relevant tasks already enqueued. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* @@ -7041,7 +7127,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * So schedule rq->idle so that ttwu_runnable() can get the rq * lock and mark owner as running. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* * OK, now we're absolutely sure @owner is on this @@ -7051,8 +7137,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) owner->blocked_donor = p; } WARN_ON_ONCE(owner && !owner->on_rq); + + if (owner && !sched_cpu_cookie_match(rq, owner)) { + if (curr_in_chain) + return proxy_resched_idle(rq); + p = donor; /* Deactivate the donor, not the runnable owner */ + clear_task_blocked_on(p, NULL); + goto deactivate; + } return owner; +resched_idle: + return proxy_resched_idle(rq); deactivate: proxy_deactivate(rq, p); return NULL; @@ -7184,13 +7280,12 @@ static void __sched notrace __schedule(int sched_mode) } } else if (!preempt && prev_state) { /* - * We pass task_is_blocked() as the should_block arg - * in order to keep mutex-blocked tasks on the runqueue - * for slection with proxy-exec (without proxy-exec - * task_is_blocked() will always be false). + * Keep mutex-blocked tasks on the runqueue for proxy execution + * only when their scheduling class allows it. Without proxy + * execution, task_is_blocked() always returns false. */ try_to_block_task(rq, prev, &prev_state, - !task_is_blocked(prev)); + !task_is_blocked(prev) || !scx_allow_proxy_exec(prev)); switch_count = &prev->nvcsw; } @@ -7211,6 +7306,7 @@ pick_again: } if (next == rq->idle) { zap_balance_callbacks(rq); + scx_proxy_reenqueue_retry(rq, next); goto keep_resched; } } @@ -7229,8 +7325,10 @@ pick_again: * on_cpu. */ donor->sched_class->put_prev_task(rq, donor, donor); - donor->sched_class->set_next_task(rq, donor, true); + donor->sched_class->set_next_task(rq, donor, SNT_PICK); } + scx_proxy_donor_start(rq); + scx_proxy_reenqueue_retry(rq, next); } else { rq_set_donor(rq, next); } @@ -7489,27 +7587,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void) NOKPROBE_SYMBOL(preempt_schedule); EXPORT_SYMBOL(preempt_schedule); -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# ifndef preempt_schedule_dynamic_enabled -# define preempt_schedule_dynamic_enabled preempt_schedule -# define preempt_schedule_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule); -void __sched notrace dynamic_preempt_schedule(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule)) - return; - preempt_schedule(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule); -EXPORT_SYMBOL(dynamic_preempt_schedule); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /** * preempt_schedule_notrace - preempt_schedule called by tracing * @@ -7562,27 +7639,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void) } EXPORT_SYMBOL_GPL(preempt_schedule_notrace); -#ifdef CONFIG_PREEMPT_DYNAMIC -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# ifndef preempt_schedule_notrace_dynamic_enabled -# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace -# define preempt_schedule_notrace_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace); -void __sched notrace dynamic_preempt_schedule_notrace(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace)) - return; - preempt_schedule_notrace(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace); -EXPORT_SYMBOL(dynamic_preempt_schedule_notrace); -# endif -#endif - #endif /* CONFIG_PREEMPTION */ /* @@ -7799,7 +7855,7 @@ out_unlock: } #endif /* CONFIG_RT_MUTEXES */ -#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC) +#if !defined(CONFIG_PREEMPTION) int __sched __cond_resched(void) { if (should_resched(0) && !irqs_disabled()) { @@ -7827,38 +7883,6 @@ int __sched __cond_resched(void) EXPORT_SYMBOL(__cond_resched); #endif -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# define cond_resched_dynamic_enabled __cond_resched -# define cond_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(cond_resched); - -# define might_resched_dynamic_enabled __cond_resched -# define might_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(might_resched); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched); -int __sched dynamic_cond_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_cond_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_cond_resched); - -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched); -int __sched dynamic_might_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_might_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_might_resched); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /* * __cond_resched_lock() - if a reschedule is pending, drop the given lock, * call schedule, and on return reacquire the lock. @@ -7928,50 +7952,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write); # endif /* - * SC:cond_resched - * SC:might_resched - * SC:preempt_schedule - * SC:preempt_schedule_notrace - * SC:irqentry_exit_cond_resched - * - * * NONE: - * cond_resched <- __cond_resched - * might_resched <- RET0 - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * VOLUNTARY: - * cond_resched <- __cond_resched - * might_resched <- __cond_resched - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * FULL: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- false * * LAZY: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- true */ enum { preempt_dynamic_undefined = -1, - preempt_dynamic_none, - preempt_dynamic_voluntary, preempt_dynamic_full, preempt_dynamic_lazy, }; @@ -7980,21 +7975,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined; int sched_dynamic_mode(const char *str) { -# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY)) - if (!strcmp(str, "none")) - return preempt_dynamic_none; - - if (!strcmp(str, "voluntary")) - return preempt_dynamic_voluntary; -# endif - if (!strcmp(str, "full")) return preempt_dynamic_full; -# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY if (!strcmp(str, "lazy")) return preempt_dynamic_lazy; -# endif return -EINVAL; } @@ -8002,71 +7987,18 @@ int sched_dynamic_mode(const char *str) # define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key) # define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key) -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled) -# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled) -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f) -# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f) -# else -# error "Unsupported PREEMPT_DYNAMIC mechanism" -# endif - static DEFINE_MUTEX(sched_dynamic_mutex); static void __sched_dynamic_update(int mode) { - /* - * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in - * the ZERO state, which is invalid. - */ - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - switch (mode) { - case preempt_dynamic_none: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: none\n"); - break; - - case preempt_dynamic_voluntary: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: voluntary\n"); - break; - case preempt_dynamic_full: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_disable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: full\n"); break; case preempt_dynamic_lazy: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_enable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: lazy\n"); @@ -8099,11 +8031,7 @@ __setup("preempt=", setup_preempt_mode); static void __init preempt_dynamic_init(void) { if (preempt_dynamic_mode == preempt_dynamic_undefined) { - if (IS_ENABLED(CONFIG_PREEMPT_NONE)) { - sched_dynamic_update(preempt_dynamic_none); - } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) { - sched_dynamic_update(preempt_dynamic_voluntary); - } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { + if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { sched_dynamic_update(preempt_dynamic_lazy); } else { /* Default static call setting, nothing to do */ @@ -8123,8 +8051,6 @@ static void __init preempt_dynamic_init(void) } \ EXPORT_SYMBOL_GPL(preempt_model_##mode) -PREEMPT_MODEL_ACCESSOR(none); -PREEMPT_MODEL_ACCESSOR(voluntary); PREEMPT_MODEL_ACCESSOR(full); PREEMPT_MODEL_ACCESSOR(lazy); @@ -8137,7 +8063,7 @@ static inline void preempt_dynamic_init(void) { } #endif /* CONFIG_PREEMPT_DYNAMIC */ const char *preempt_modes[] = { - "none", "voluntary", "full", "lazy", NULL, + "full", "lazy", NULL, }; const char *preempt_model_str(void) @@ -8759,6 +8685,9 @@ int sched_cpu_activate(unsigned int cpu) */ sched_set_rq_online(rq, cpu); + /* preferred is subset of active and follows its state */ + set_cpu_preferred(cpu, true); + return 0; } @@ -8772,6 +8701,8 @@ int sched_cpu_deactivate(unsigned int cpu) if (ret) return ret; + set_cpu_preferred(cpu, false); + /* * Remove CPU from nohz.idle_cpus_mask to prevent participating in * load balancing when not active @@ -11349,3 +11280,88 @@ void sched_change_end(struct sched_change_ctx *ctx) p->sched_class->prio_changed(rq, p, ctx->prio); } } + +#ifdef CONFIG_PREFERRED_CPU +static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work); + +static int sched_non_preferred_cpu_push_stop(void *arg) +{ + struct task_struct *p = arg; + struct rq *rq = this_rq(); + struct rq_flags rf; + int cpu; + + if (cpu_preferred(rq->cpu)) { + scoped_guard(rq_lock_irqsave, rq) + rq->npc_push_work_pending = false; + put_task_struct(p); + return 0; + } + + scoped_guard (raw_spinlock_irq, &p->pi_lock) { + /* + * select_fallback_rq() may acquire the rq lock in case of + * fallback. So call it before grabbing rq lock. If the task + * migrates to another CPU before the rq lock is acquired, + * subsequent validation of task's current rq will help to + * safely bail out. + */ + cpu = select_fallback_rq(rq->cpu, p); + rq_lock(rq, &rf); + rq->npc_push_work_pending = false; + update_rq_clock(rq); + context_unsafe_alias(rq); + + if (task_rq(p) == rq && task_on_rq_queued(p)) { + struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu); + + if (rq != dest_rq) + schedstat_inc(p->stats.nr_migrations_cpu_non_preferred); + rq = dest_rq; + } + rq_unlock(rq, &rf); + } + + put_task_struct(p); + return 0; +} + +/* + * Push the current task running on non-preferred CPU(npc). + * Using this non preferred CPU will lead to more contention + * in the host. So it is better not to use this CPU. + * + * Since task is running, call a stopper to push the task out. This is + * similar to how task moves during hotplug. In select_fallback_rq() a + * preferred CPU will be chosen and henceforth task shouldn't come back to + * this CPU again. + * + * Works for FAIR class only. + * + * If task is affined only on non-preferred CPUs, no point in moving it out. + */ +void sched_push_current_non_preferred_cpu(struct rq *rq) +{ + struct task_struct *push_task = rq->curr; + + scoped_guard(rq_lock, rq) { + /* Push the task if its explicit affinity allows */ + if (!task_can_migrate_to_preferred(push_task, rq->cpu)) + return; + + /* There is already a stopper thread. Don't race with it. */ + if (rq->npc_push_work_pending) + return; + + if (is_migration_disabled(push_task)) + return; + + rq->npc_push_work_pending = true; + } + + /* sched_tick runs with interrupts disabled. */ + get_task_struct(push_task); + stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop, + push_task, this_cpu_ptr(&npc_push_task_work)); +} +#endif diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c index 06bddaa738e5..f16970ca81d0 100644 --- a/kernel/sched/cputime.c +++ b/kernel/sched/cputime.c @@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta) * occasion account more time than the calling functions think elapsed. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled); #ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN static u64 native_steal_clock(int cpu) @@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock); static __always_inline u64 steal_account_process_time(u64 maxtime) { #ifdef CONFIG_PARAVIRT - if (static_key_false(¶virt_steal_enabled)) { + if (static_branch_unlikely(¶virt_steal_enabled)) { u64 steal; steal = paravirt_steal_clock(smp_processor_id()); diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 0663c00c41c0..c0ebdcde5fe5 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se) * chosen as the deadline is too small, don't even try to * start the timer in the past! */ - if (ktime_us_delta(act, now) < 0) + if (ktime_before(act, now)) return 0; /* @@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se) * DL keeps current in tree, because ->deadline is not typically changed while * a task is runnable. */ -static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_dl_entity *dl_se = &p->dl; struct dl_rq *dl_rq = &rq->dl; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_dl_rq(&p->dl)) update_stats_wait_end_dl(dl_rq, dl_se); @@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(dl_rq->curr); dl_rq->curr = dl_se; - if (!first) + if (type != SNT_PICK) return; if (rq->donor->sched_class != &dl_sched_class) diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c index 72236db67983..e6a3b516c703 100644 --- a/kernel/sched/debug.c +++ b/kernel/sched/debug.c @@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v) #ifdef CONFIG_JUMP_LABEL -#define jump_label_key__true STATIC_KEY_INIT_TRUE -#define jump_label_key__false STATIC_KEY_INIT_FALSE +#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT } +#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT } #define SCHED_FEAT(name, enabled) \ jump_label_key__##enabled , -struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { +union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = { #include "features.h" }; @@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { static void sched_feat_disable(int i) { - static_key_disable_cpuslocked(&sched_feat_keys[i]); + static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true); } static void sched_feat_enable(int i) { - static_key_enable_cpuslocked(&sched_feat_keys[i]); + static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false); } #else /* !CONFIG_JUMP_LABEL: */ static void sched_feat_disable(int i) { }; @@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf, static int sched_dynamic_show(struct seq_file *m, void *v) { - int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2; int mode = READ_ONCE(preempt_dynamic_mode); - int j; - /* Count entries in NULL terminated preempt_modes */ - for (j = 0; preempt_modes[j]; j++) - ; - j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY); - - for (; i < j; i++) { + /* Stop at NULL terminator */ + for (int i = 0; preempt_modes[i]; i++) { if (mode == i) seq_puts(m, "("); seq_puts(m, preempt_modes[i]); @@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns, P_SCHEDSTAT(nr_failed_migrations_running); P_SCHEDSTAT(nr_failed_migrations_hot); P_SCHEDSTAT(nr_forced_migrations); + P_SCHEDSTAT(nr_migrations_cpu_non_preferred); P_SCHEDSTAT(nr_wakeups); P_SCHEDSTAT(nr_wakeups_sync); P_SCHEDSTAT(nr_wakeups_migrate); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index e56c3c95018f..aed5286b82aa 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -24,6 +24,11 @@ DEFINE_RAW_SPINLOCK(scx_sched_lock); +bool scx_allow_proxy_exec(const struct task_struct *p) +{ + return true; +} + /* * NOTE: sched_ext is in the process of growing multiple scheduler support and * scx_root usage is in a transitional state. Naked dereferences are safe if the @@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq) schedule_deferred(rq); } +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next) +{ +} + void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq, u64 reenq_flags, struct rq *locked_rq) { @@ -3021,10 +3030,13 @@ has_tasks: return verdict; } -static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type) { struct scx_sched *sch = scx_task_sched(p); + if (type == SNT_REPICK) + return; + if (p->scx.flags & SCX_TASK_QUEUED) { /* * Core-sched might decide to execute @p before it is @@ -3082,6 +3094,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) } } +void scx_proxy_donor_start(struct rq *rq) +{ +} + static enum scx_cpu_preempt_reason preempt_reason_from_class(const struct sched_class *class) { diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h index 0b7fc46aee08..3cfbfeb1bf9d 100644 --- a/kernel/sched/ext/ext.h +++ b/kernel/sched/ext/ext.h @@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq); int scx_check_setscheduler(struct task_struct *p, int policy); bool task_should_scx(int policy); bool scx_allow_ttwu_queue(const struct task_struct *p); +bool scx_allow_proxy_exec(const struct task_struct *p); +void scx_proxy_donor_start(struct rq *rq); +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next); void init_sched_ext_class(void); static inline u32 scx_cpuperf_target(s32 cpu) @@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {} static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; } static inline bool task_on_scx(const struct task_struct *p) { return false; } static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; } +static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; } +static inline void scx_proxy_donor_start(struct rq *rq) {} +static inline void scx_proxy_reenqueue_retry(struct rq *rq, + struct task_struct *next) {} static inline void init_sched_ext_class(void) {} #endif /* CONFIG_SCHED_CLASS_EXT */ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 8d38c3b7d792..56f4ab6d9ada 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -23,6 +23,7 @@ #include <linux/energy_model.h> #include <linux/mmap_lock.h> #include <linux/jiffies.h> +#include <linux/math.h> #include <linux/mm_api.h> #include <linux/highmem.h> #include <linux/hrtimer.h> @@ -819,12 +820,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq) if (curr && !curr->on_rq) curr = NULL; - /* - * This is called from set_next_task_fair(.first=true) / - * set_protect_slice() so curr had better be set and on_rq. - */ - WARN_ON_ONCE(!curr); - if (weight) { s64 runtime = cfs_rq->sum_w_vruntime; @@ -1136,10 +1131,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity /* If there are shorter slices than se's one */ if (slice != se->slice) { + vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); if (sched_feat(PREEMPT_SHORT)) vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); - else - vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); } se->vprot = vprot; @@ -1147,10 +1141,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se) { - u64 slice = cfs_rq_min_slice(cfs_rq); u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq)); + u64 slice = normalized_sysctl_sched_base_slice; + u64 vprot; - se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + if (sched_feat(RUN_TO_PARITY)) + slice = cfs_rq_min_slice(cfs_rq); + + vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + + if (sched_feat(PREEMPT_SHORT) && slice != se->slice) + vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); + + se->vprot = vprot; } static inline bool protect_slice(struct sched_entity *se) @@ -3712,7 +3715,7 @@ static void update_task_scan_period(struct task_struct *p, p->mm->numa_next_scan = jiffies + msecs_to_jiffies(p->numa_scan_period); - return; + goto out; } /* @@ -3756,7 +3759,10 @@ static void update_task_scan_period(struct task_struct *p, p->numa_scan_period = clamp(p->numa_scan_period + diff, task_scan_min(p), task_scan_max(p)); - memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality)); + +out: + memset(p->numa_faults_locality, 0, + sizeof(p->numa_faults_locality)); } /* @@ -8208,7 +8214,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) struct sched_entity *se = &p->se; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight; - bool curr; if (task_is_throttled(p) && enqueue_throttled_task(p)) return; @@ -8237,23 +8242,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) if (p->in_iowait) cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT); - /* - * XXX comment on the curr thing - */ - curr = (cfs_rq->curr == se); - if (curr) - place_entity(cfs_rq, se, flags); if (se->on_rq && se->sched_delayed) requeue_delayed_entity(cfs_rq, se); weight = enqueue_hierarchy(p, flags); - - if (!curr) { - reweight_eevdf(cfs_rq, se, weight, false); - place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); - __enqueue_entity(cfs_rq, se); - } + reweight_eevdf(cfs_rq, se, weight, false); + place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); + __enqueue_entity(cfs_rq, se); if (!rq_h_nr_queued && rq->cfs.h_nr_queued) dl_server_start(&rq->fair_server); @@ -8673,8 +8669,8 @@ static int sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu) { unsigned long load, min_load = ULONG_MAX; - unsigned int min_exit_latency = UINT_MAX; - u64 latest_idle_timestamp = 0; + u64 min_exit_latency = U64_MAX; + unsigned int nr_candidates = 0; int least_loaded_cpu = this_cpu; int shallowest_idle_cpu = -1; int i; @@ -8695,24 +8691,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct * if (available_idle_cpu(i)) { struct cpuidle_state *idle = idle_get_state(rq); - if (idle && idle->exit_latency < min_exit_latency) { - /* - * We give priority to a CPU whose idle state - * has the smallest exit latency irrespective - * of any idle timestamp. - */ - min_exit_latency = idle->exit_latency; - latest_idle_timestamp = rq->idle_stamp; - shallowest_idle_cpu = i; - } else if ((!idle || idle->exit_latency == min_exit_latency) && - rq->idle_stamp > latest_idle_timestamp) { - /* - * If equal or no active idle state, then - * the most recently idled CPU might have - * a warmer cache. - */ - latest_idle_timestamp = rq->idle_stamp; + u64 exit_latency = idle ? idle->exit_latency : U64_MAX; + + if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) { + min_exit_latency = exit_latency; shallowest_idle_cpu = i; + nr_candidates = 1; + } else if (exit_latency == min_exit_latency) { + nr_candidates++; + if (!reciprocal_scale(sched_rng(), nr_candidates)) + shallowest_idle_cpu = i; } } else if (shallowest_idle_cpu == -1) { load = cpu_load(cpu_rq(i)); @@ -10071,8 +10059,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse) { - if (cfs_rq->next && cfs_rq->next->slice < pse->slice) - return false; + if (cfs_rq->next) { + if (cfs_rq->next->slice < pse->slice) + return false; + + if (cfs_rq->next->slice == pse->slice && + entity_before(cfs_rq->next, pse)) + return false; + } set_next_buddy(cfs_rq, pse); return true; @@ -11438,21 +11432,7 @@ next: */ static void attach_tasks(struct lb_env *env) { - struct list_head *tasks = &env->tasks; - struct task_struct *p; - struct rq_flags rf; - - rq_lock(env->dst_rq, &rf); - update_rq_clock(env->dst_rq); - - while (!list_empty(tasks)) { - p = list_first_entry(tasks, struct task_struct, se.group_node); - list_del_init(&p->se.group_node); - - attach_task(env->dst_rq, p); - } - - rq_unlock(env->dst_rq, &rf); + __attach_tasks(env->dst_rq, &env->tasks); } #ifdef CONFIG_NO_HZ_COMMON @@ -13745,7 +13725,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq, }; bool need_unlock = false; - cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask); + cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask); schedstat_inc(sd->lb_count[idle]); @@ -14870,10 +14850,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf) */ this_rq->idle_stamp = rq_clock(this_rq); - /* - * Do not pull tasks towards !active CPUs... - */ - if (!cpu_active(this_cpu)) + /* Do not pull tasks towards !preferred CPUs */ + if (!cpu_preferred(this_cpu)) return 0; /* @@ -15513,14 +15491,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p) } } -static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_entity *se = &p->se; - bool throttled = false; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight = NICE_0_LOAD; + bool first = type == SNT_PICK; + bool throttled = false; bool on_rq = se->on_rq; + if (type == SNT_REPICK) + goto repick; + clear_buddies(cfs_rq, se); if (on_rq) @@ -15564,11 +15546,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(se->sched_delayed); - if (hrtick_enabled_fair(rq)) - hrtick_start_fair(rq, p); - update_misfit_status(p, rq); sched_fair_update_stop_tick(rq, p); + +repick: + /* + * A same-task repick skips put_prev_task_fair(), but + * pick_task_fair() refreshed the entity hrtick_start_fair() reads + * before selecting it again. rq->cfs.curr identifies that entity, + * including with group scheduling. + */ + if (hrtick_enabled_fair(rq)) + hrtick_start_fair(rq, p); } void init_cfs_rq(struct cfs_rq *cfs_rq) diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c index eb73b65ce6c4..76f3c84ca684 100644 --- a/kernel/sched/idle.c +++ b/kernel/sched/idle.c @@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t update_rq_avg_idle(rq); } -static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first) +static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type) { + if (type == SNT_REPICK) + return; + update_idle_core(rq); scx_update_idle(rq, true, true); schedstat_inc(rq->sched_goidle); diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 85303add726d..1535046a23ff 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags) check_preempt_equal_prio(rq, p); } -static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first) +static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_rt_entity *rt_se = &p->rt; struct rt_rq *rt_rq = &rq->rt; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_rt_rq(&p->rt)) update_stats_wait_end_rt(rt_rq, rt_se); @@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f /* The running task is never eligible for pushing */ dequeue_pushable_task(rq, p); - if (!first) + if (type != SNT_PICK) return; /* diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index e656c7059bf8..7d2ec527b8a2 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -1326,6 +1326,9 @@ struct rq { #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING u64 prev_steal_time_rq; #endif +#ifdef CONFIG_PREFERRED_CPU + bool npc_push_work_pending; +#endif /* calc_load related fields */ unsigned long calc_load_update; @@ -1371,7 +1374,6 @@ struct rq { struct task_struct *core_pick; struct sched_dl_entity *core_dl_server; unsigned int core_enabled; - unsigned int core_sched_seq; struct rb_root core_tree; /* shared state -- careful with sched_core_cpu_deactivate() */ @@ -2447,16 +2449,25 @@ extern __read_mostly unsigned int sysctl_sched_features; #ifdef CONFIG_JUMP_LABEL -#define SCHED_FEAT(name, enabled) \ -static __always_inline bool static_branch_##name(struct static_key *key) \ -{ \ - return static_key_##enabled(key); \ +union sched_feat_key { + struct static_key_true key_true; + struct static_key_false key_false; +}; + +#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true) +#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false) + +#define SCHED_FEAT(name, enabled) \ +static __always_inline bool \ +static_branch_##name(union sched_feat_key *key) \ +{ \ + return sched_feat_branch_##enabled(key); \ } #include "features.h" #undef SCHED_FEAT -extern struct static_key sched_feat_keys[__SCHED_FEAT_NR]; +extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR]; #define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x])) #else /* !CONFIG_JUMP_LABEL: */ @@ -2508,6 +2519,12 @@ static inline bool task_is_blocked(struct task_struct *p) return !!p->blocked_on; } +#ifdef CONFIG_SCHED_PROXY_EXEC +void sched_proxy_block_task(struct rq *rq, struct task_struct *p); +#else +static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {} +#endif + static inline int task_on_cpu(struct rq *rq, struct task_struct *p) { return p->on_cpu; @@ -2527,11 +2544,17 @@ static inline int task_on_rq_migrating(struct task_struct *p) #define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */ #define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */ #define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */ - -#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */ +/* + * Hint that the caller expects the waker to sleep soon. + * Scheduler classes may use it for placement or preemption. + * Callers must not rely on it to prevent migration, + * preserve CPU locality or make the wakee run next. + */ +#define WF_SYNC 0x10 #define WF_MIGRATED 0x20 /* Internal use, task got migrated */ #define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */ #define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */ +#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */ static_assert(WF_EXEC == SD_BALANCE_EXEC); static_assert(WF_FORK == SD_BALANCE_FORK); @@ -2621,6 +2644,12 @@ struct affinity_context { extern s64 update_curr_common(struct rq *rq); +enum snt_e { + SNT_NORMAL, /* set_next_task() */ + SNT_PICK, /* put_prev_set_next_task(): prev != next */ + SNT_REPICK, /* put_prev_set_next_task(): prev == next */ +}; + struct sched_class { #ifdef CONFIG_UCLAMP_TASK @@ -2678,7 +2707,7 @@ struct sched_class { * __schedule: rq->lock */ void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next); - void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first); + void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type); /* * select_task_rq: p->pi_lock @@ -2781,7 +2810,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev) static inline void set_next_task(struct rq *rq, struct task_struct *next) { - next->sched_class->set_next_task(rq, next, false); + next->sched_class->set_next_task(rq, next, SNT_NORMAL); } static inline void @@ -2802,11 +2831,13 @@ static inline void put_prev_set_next_task(struct rq *rq, __put_prev_set_next_dl_server(rq, prev, next); - if (next == prev) + if (next == prev) { + next->sched_class->set_next_task(rq, next, SNT_REPICK); return; + } prev->sched_class->put_prev_task(rq, prev, next); - next->sched_class->set_next_task(rq, next, true); + next->sched_class->set_next_task(rq, next, SNT_PICK); } /* @@ -3139,6 +3170,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p) attach_task(rq, p); } +/* + * __attach_tasks() - attaches a list of tasks (using se.group_node) to + * the new rq + */ +static inline void __attach_tasks(struct rq *rq, struct list_head *tasks) +{ + guard(rq_lock)(rq); + update_rq_clock(rq); + + while (!list_empty(tasks)) { + struct task_struct *p; + + p = list_first_entry(tasks, struct task_struct, se.group_node); + list_del_init(&p->se.group_node); + + attach_task(rq, p); + } +} + #ifdef CONFIG_PREEMPT_RT # define SCHED_NR_MIGRATE_BREAK 8 #else @@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change) #include "ext/ext.h" +#ifdef CONFIG_PREFERRED_CPU +void sched_push_current_non_preferred_cpu(struct rq *rq); +#else /* !CONFIG_PREFERRED_CPU */ +static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { } +#endif + #endif /* _KERNEL_SCHED_SCHED_H */ diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index c909ca0d8c87..1e0109ec36b3 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags) /* we're never preempted */ } -static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first) +static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type) { + if (type == SNT_REPICK) + return; + stop->se.exec_start = rq_clock_task(rq); } diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c index d033f600f48c..477e4bf9c01e 100644 --- a/kernel/sched/wait.c +++ b/kernel/sched/wait.c @@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. + * Passes WF_SYNC to waitqueue wake functions. The default wake function + * forwards it to the scheduler; see WF_SYNC for the hint's semantics. * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * If this function wakes up a task, it executes a full memory barrier + * before accessing the task state. */ void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) @@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs in that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. - * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * Same as __wake_up_sync_key(), but called with @wq_head->lock held. */ void __wake_up_locked_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) diff --git a/kernel/time/hrtimer.c b/kernel/time/hrtimer.c index cbf1693c86b3..17dd38a6cee7 100644 --- a/kernel/time/hrtimer.c +++ b/kernel/time/hrtimer.c @@ -780,7 +780,8 @@ static void hrtimer_switch_to_hres(void) return; } base->hres_active = true; - hrtimer_resolution = HIGH_RES_NSEC; + if (hrtimer_resolution != HIGH_RES_NSEC) + hrtimer_resolution = HIGH_RES_NSEC; tick_setup_sched_timer(true); /* "Retrigger" the interrupt to get things going */ @@ -2003,7 +2004,7 @@ bool hrtimer_active(const struct hrtimer *timer) base = READ_ONCE(timer->base); seq = raw_read_seqcount_begin(&base->seq); - if (timer->is_queued || base->running == timer) + if (timer->is_queued || READ_ONCE(base->running) == timer) return true; } while (read_seqcount_retry(&base->seq, seq) || base != READ_ONCE(timer->base)); @@ -2040,7 +2041,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc lockdep_assert_held(&cpu_base->lock); debug_hrtimer_deactivate(timer); - base->running = timer; + WRITE_ONCE(base->running, timer); /* * Separate the ->running assignment from the ->is_queued assignment. @@ -2099,7 +2100,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc raw_write_seqcount_barrier(&base->seq); WARN_ON_ONCE(base->running != timer); - base->running = NULL; + WRITE_ONCE(base->running, NULL); } static void __hrtimer_run_queues(struct hrtimer_cpu_base *cpu_base, ktime_t now, @@ -2323,9 +2324,9 @@ void hrtimer_run_queues(void) static enum hrtimer_restart hrtimer_wakeup(struct hrtimer *timer) { struct hrtimer_sleeper *t = container_of(timer, struct hrtimer_sleeper, timer); - struct task_struct *task = t->task; + struct task_struct *task = hrtimer_sleeper_task_get(t); - t->task = NULL; + hrtimer_sleeper_task_set(t, NULL); if (task) wake_up_process(task); @@ -2354,7 +2355,7 @@ void hrtimer_sleeper_start_expires(struct hrtimer_sleeper *sl, enum hrtimer_mode /* If already expired, clear the task pointer and set current state to running */ if (!hrtimer_start_expires_user(&sl->timer, mode)) { - sl->task = NULL; + hrtimer_sleeper_task_set(sl, NULL); __set_current_state(TASK_RUNNING); } } @@ -2388,7 +2389,7 @@ static void __hrtimer_setup_sleeper(struct hrtimer_sleeper *sl, clockid_t clock_ } __hrtimer_setup(&sl->timer, hrtimer_wakeup, clock_id, mode); - sl->task = current; + hrtimer_sleeper_task_set(sl, current); } /** @@ -2432,17 +2433,17 @@ static int __sched do_nanosleep(struct hrtimer_sleeper *t, enum hrtimer_mode mod set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); hrtimer_sleeper_start_expires(t, mode); - if (likely(t->task)) + if (likely(hrtimer_sleeper_task_get(t))) schedule(); hrtimer_cancel(&t->timer); mode = HRTIMER_MODE_ABS; - } while (t->task && !signal_pending(current)); + } while (hrtimer_sleeper_task_get(t) && !signal_pending(current)); __set_current_state(TASK_RUNNING); - if (!t->task) + if (!hrtimer_sleeper_task_get(t)) return 0; restart = ¤t->restart_block; diff --git a/kernel/time/posix-cpu-timers.c b/kernel/time/posix-cpu-timers.c index 0bf4fcd969c8..cd75d4bb5b64 100644 --- a/kernel/time/posix-cpu-timers.c +++ b/kernel/time/posix-cpu-timers.c @@ -439,6 +439,38 @@ static void trigger_base_recalc_expires(struct k_itimer *timer, base->nextevt = 0; } +static inline bool cpu_timer_enqueue(struct timerqueue_head *head, + struct cpu_timer *ctmr) +{ + ctmr->head = head; + return timerqueue_add(head, &ctmr->node); +} + +static inline bool cpu_timer_queued(struct cpu_timer *ctmr) +{ + return !!ctmr->head; +} + +static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr) +{ + if (cpu_timer_queued(ctmr)) { + timerqueue_del(ctmr->head, &ctmr->node); + ctmr->head = NULL; + return true; + } + return false; +} + +static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr) +{ + return ctmr->node.expires; +} + +static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp) +{ + ctmr->node.expires = exp; +} + /* * Dequeue the timer and reset the base if it was its earliest expiration. * It makes sure the next tick recalculates the base next expiration so we @@ -607,6 +639,7 @@ static int posix_cpu_timer_del(struct k_itimer *timer) } if (!ret) { + WARN_ON_ONCE(cpu_timer_queued(&timer->it.cpu)); put_pid(timer->it.cpu.pid); timer->it_status = POSIX_TIMER_DISARMED; } @@ -639,18 +672,50 @@ static void cleanup_timers(struct posix_cputimers *pct) cleanup_timerqueue(&pct->bases[CPUCLOCK_SCHED].tqhead); } +static inline void posix_cpu_timers_exit_work(void); + /* - * These are both called with the siglock held, when the current thread - * is being reaped. When the final (leader) thread in the group is reaped, - * posix_cpu_timers_exit_group will be called after posix_cpu_timers_exit. + * Invoked from posixtimer_exit_task() after PF_EXITING was set in tsk::flags or + * from posixtimer_exec_cleanup(). */ -void posix_cpu_timers_exit(struct task_struct *tsk) +void posix_cpu_timers_exit_task(void) { - cleanup_timers(&tsk->posix_cputimers); + posix_cpu_timers_exit_work(); + + guard(spinlock_irq)(¤t->sighand->siglock); + cleanup_timers(¤t->posix_cputimers); } -void posix_cpu_timers_exit_group(struct task_struct *tsk) + +/* + * Invoked from posixtimer_exit_group() after PF_EXITING was set in tsk::flags. + */ +void posix_cpu_timers_exit_group(void) { - cleanup_timers(&tsk->signal->posix_cputimers); + posix_cpu_timers_exit_task(); + + guard(spinlock_irq)(¤t->sighand->siglock); + cleanup_timers(¤t->signal->posix_cputimers); +} + +/* + * This function validates that POSIX CPU timers can be safely enqueued on the + * target task. + * + * Enqueue is allowed when PF_EXITING is not set. If set then it is only allowed + * for process shared timers (type = PIDTYPE_TGID) as long as tsk::signal::flags + * does not have SIGNAL_GROUP_EXIT set. PIDTYPE_PID targets are not allowed at + * all when the task has PF_EXITING set. + * + * This guarantees that after the POSIX timer cleanup in posixtimer_exit() no + * POSIX CPU timers are queued on the task or in case of a group exit on the + * process. + */ +static inline bool task_can_enqueue_timer(struct task_struct *tsk, enum pid_type type) +{ + if (likely(!(tsk->flags & PF_EXITING))) + return true; + + return type == PIDTYPE_TGID && !(tsk->signal->flags & SIGNAL_GROUP_EXIT); } /* @@ -663,7 +728,13 @@ static void arm_timer(struct k_itimer *timer, struct task_struct *p) struct cpu_timer *ctmr = &timer->it.cpu; u64 newexp = cpu_timer_getexpires(ctmr); + lockdep_assert_held(&p->sighand->siglock); + timer->it_status = POSIX_TIMER_ARMED; + + if (unlikely(!task_can_enqueue_timer(p, clock_pid_type(timer->it_clock)))) + return; + if (!cpu_timer_enqueue(&base->tqhead, ctmr)) return; @@ -1201,6 +1272,20 @@ static void posix_cpu_timers_work(struct callback_head *work) mutex_unlock(&cw->mutex); } +static inline void posix_cpu_timers_exit_work(void) +{ + /* Canceling the work is only valid for exit() but not for exec() */ + if (!(current->flags & PF_EXITING)) + return; + /* + * current->flags has PF_EXITING set so this can be done lockless and + * with interrupts enabled as PF_EXITING prevents the interrupt from + * scheduling the work. + */ + if (current->posix_cputimers_work.scheduled) + task_work_cancel(current, ¤t->posix_cputimers_work.work); +} + /* * Invoked from the posix-timer core when a cancel operation failed because * the timer is marked firing. The caller holds rcu_read_lock(), which @@ -1331,6 +1416,8 @@ static inline void __run_posix_cpu_timers(struct task_struct *tsk) lockdep_posixtimer_exit(); } +static inline void posix_cpu_timers_exit_work(void) { } + static void posix_cpu_timer_wait_running(struct k_itimer *timr) { cpu_relax(); @@ -1477,7 +1564,7 @@ void run_posix_cpu_timers(void) * posix_cpu_timer_del() may fail to lock_task_sighand(tsk) and * miss timer->it.cpu.firing != 0. */ - if (tsk->exit_state) + if (tsk->flags & PF_EXITING) return; /* diff --git a/kernel/time/posix-timers.c b/kernel/time/posix-timers.c index 436ba794cc0b..188dbedbffca 100644 --- a/kernel/time/posix-timers.c +++ b/kernel/time/posix-timers.c @@ -1077,13 +1077,9 @@ SYSCALL_DEFINE1(timer_delete, timer_t, timer_id) return 0; } -/* - * Invoked from do_exit() when the last thread of a thread group exits. - * At that point no other task can access the timers of the dying - * task anymore. - */ -void exit_itimers(struct task_struct *tsk) +static void posixtimer_delete_timers(void) { + struct task_struct *tsk = current; struct hlist_head timers; struct hlist_node *next; struct k_itimer *timer; @@ -1120,6 +1116,24 @@ void exit_itimers(struct task_struct *tsk) } } +void posixtimer_exit(bool group_dead) +{ + if (group_dead) { + hrtimer_cancel(¤t->signal->real_timer); + posix_cpu_timers_exit_group(); + posixtimer_delete_timers(); + } else { + posix_cpu_timers_exit_task(); + } +} + +void posixtimer_exec(void) +{ + posix_cpu_timers_exit_task(); + posixtimer_delete_timers(); + flush_itimer_signals(); +} + SYSCALL_DEFINE2(clock_settime, const clockid_t, which_clock, const struct __kernel_timespec __user *, tp) { diff --git a/kernel/time/posix-timers.h b/kernel/time/posix-timers.h index 4ea9611dd716..79fd7ea71046 100644 --- a/kernel/time/posix-timers.h +++ b/kernel/time/posix-timers.h @@ -51,3 +51,6 @@ int common_timer_set(struct k_itimer *timr, int flags, struct itimerspec64 *old_setting); void posix_timer_set_common(struct k_itimer *timer, struct itimerspec64 *new_setting); int common_timer_del(struct k_itimer *timer); + +void posix_cpu_timers_exit_task(void); +void posix_cpu_timers_exit_group(void); diff --git a/kernel/time/sleep_timeout.c b/kernel/time/sleep_timeout.c index 3c90574bd904..ad8c415851ae 100644 --- a/kernel/time/sleep_timeout.c +++ b/kernel/time/sleep_timeout.c @@ -212,7 +212,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta, hrtimer_set_expires_range_ns(&t.timer, *expires, delta); hrtimer_sleeper_start_expires(&t, mode); - if (likely(t.task)) + if (likely(hrtimer_sleeper_task_get(&t))) schedule(); hrtimer_cancel(&t.timer); @@ -220,7 +220,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta, __set_current_state(TASK_RUNNING); - return !t.task ? 0 : -EINTR; + return !hrtimer_sleeper_task_get(&t) ? 0 : -EINTR; } EXPORT_SYMBOL_GPL(schedule_hrtimeout_range_clock); diff --git a/kernel/time/tick-sched.c b/kernel/time/tick-sched.c index 6c3fea386713..a7893a079a83 100644 --- a/kernel/time/tick-sched.c +++ b/kernel/time/tick-sched.c @@ -738,14 +738,11 @@ bool tick_nohz_tick_stopped_cpu(int cpu) */ static void tick_nohz_update_jiffies(ktime_t now) { - unsigned long flags; + /* Reached only from irq_enter_rcu(), i.e. hard interrupt entry. */ + lockdep_assert_irqs_disabled(); __this_cpu_write(tick_cpu_sched.idle_waketime, now); - - local_irq_save(flags); tick_do_update_jiffies64(now); - local_irq_restore(flags); - touch_softlockup_watchdog_sched(); } @@ -819,7 +816,7 @@ u64 get_jiffies_update(unsigned long *basej) */ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) { - u64 basemono, next_tick, delta, expires; + u64 basemono, next_tick, expires; unsigned long basejiff; int tick_cpu; @@ -859,8 +856,7 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) * If the tick is due in the next period, keep it ticking or * force prod the timer. */ - delta = next_tick - basemono; - if (delta <= (u64)TICK_NSEC) { + if (next_tick - basemono <= (u64)TICK_NSEC) { /* * We've not stopped the tick yet, and there's a timer in the * next period, so no point in stopping it either, bail. @@ -876,17 +872,19 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) * the sleep time to the timekeeping 'max_deferment' value. * Otherwise we can sleep as long as we want. */ - delta = timekeeping_max_deferment(); tick_cpu = READ_ONCE(tick_do_timer_cpu); if (tick_cpu != cpu && - (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) - delta = KTIME_MAX; - - /* Calculate the next expiry time */ - if (delta < (KTIME_MAX - basemono)) - expires = basemono + delta; - else + (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) { expires = KTIME_MAX; + } else { + expires = timekeeping_max_deferment(); + + /* Calculate the next expiry time */ + if (expires < (KTIME_MAX - basemono)) + expires += basemono; + else + expires = KTIME_MAX; + } ts->timer_expires = min_t(u64, expires, next_tick); diff --git a/kernel/time/time_test.c b/kernel/time/time_test.c index 1b99180da288..8b718767b3ba 100644 --- a/kernel/time/time_test.c +++ b/kernel/time/time_test.c @@ -87,8 +87,24 @@ static void time64_to_tm_test_date_range(struct kunit *test) } } +static void time64_to_tm_test_wide_day_count(struct kunit *test) +{ + /* 2^31 days: the first count that does not fit in a 32-bit long. */ + time64_t timestamp = (1LL << 31) * 86400; + struct tm result; + + time64_to_tm(timestamp, 0, &result); + + KUNIT_EXPECT_EQ(test, result.tm_year, 5879680); + KUNIT_EXPECT_EQ(test, result.tm_mon, 6); + KUNIT_EXPECT_EQ(test, result.tm_mday, 12); + KUNIT_EXPECT_EQ(test, result.tm_yday, 193); + KUNIT_EXPECT_EQ(test, result.tm_wday, 6); +} + static struct kunit_case time_test_cases[] = { KUNIT_CASE_SLOW(time64_to_tm_test_date_range), + KUNIT_CASE(time64_to_tm_test_wide_day_count), {} }; diff --git a/kernel/time/timeconv.c b/kernel/time/timeconv.c index 59b922c826e7..aed3af950fa0 100644 --- a/kernel/time/timeconv.c +++ b/kernel/time/timeconv.c @@ -49,8 +49,9 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result) u32 u32tmp, day_of_century, year_of_century, day_of_year, month, day; u64 u64tmp, udays, century, year; bool is_Jan_or_Feb, is_leap_year; - long days, rem; int remainder; + long rem; + s64 days; days = div_s64_rem(totalsecs, SECS_PER_DAY, &remainder); rem = remainder; @@ -70,7 +71,8 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result) result->tm_sec = rem % 60; /* January 1, 1970 was a Thursday. */ - result->tm_wday = (4 + days) % 7; + div_s64_rem(days + 4, 7, &remainder); + result->tm_wday = remainder; if (result->tm_wday < 0) result->tm_wday += 7; diff --git a/kernel/time/timekeeping.c b/kernel/time/timekeeping.c index ea2e6e55f37b..d54c4d303db6 100644 --- a/kernel/time/timekeeping.c +++ b/kernel/time/timekeeping.c @@ -861,8 +861,10 @@ static void timekeeping_update_from_shadow(struct tk_data *tkd, unsigned int act * * Write xtime_sec first so that even if the memcpy() tears the store * data integrity is provided for ktime_get_real_seconds(). + * The same goes for ktime_sec and ktime_get_seconds(). */ WRITE_ONCE(tkd->timekeeper.xtime_sec, tk->xtime_sec); + WRITE_ONCE(tkd->timekeeper.ktime_sec, tk->ktime_sec); memcpy(&tkd->timekeeper, tk, sizeof(*tk)); write_seqcount_end(&tkd->seq); } @@ -1169,7 +1171,7 @@ time64_t ktime_get_seconds(void) struct timekeeper *tk = &tk_core.timekeeper; WARN_ON(timekeeping_suspended); - return tk->ktime_sec; + return READ_ONCE(tk->ktime_sec); } EXPORT_SYMBOL_GPL(ktime_get_seconds); diff --git a/kernel/time/timer.c b/kernel/time/timer.c index ae9abf14688e..42afdcb229d8 100644 --- a/kernel/time/timer.c +++ b/kernel/time/timer.c @@ -890,7 +890,7 @@ static inline void detach_timer(struct timer_list *timer, bool clear_pending) __hlist_del(entry); if (clear_pending) - entry->pprev = NULL; + WRITE_ONCE(entry->pprev, NULL); entry->next = LIST_POISON2; } diff --git a/kernel/time/timer_migration.c b/kernel/time/timer_migration.c index 059d43355e65..f920e73fff51 100644 --- a/kernel/time/timer_migration.c +++ b/kernel/time/timer_migration.c @@ -715,7 +715,7 @@ static void __tmigr_cpu_activate(struct tmigr_cpu *tmc) trace_tmigr_cpu_active(tmc); - tmc->cpuevt.ignore = true; + WRITE_ONCE(tmc->cpuevt.ignore, true); WRITE_ONCE(tmc->wakeup, KTIME_MAX); walk_groups(&tmigr_active_up, &data, tmc); @@ -1258,7 +1258,7 @@ u64 tmigr_cpu_new_timer(u64 nextexp) ret = READ_ONCE(tmc->wakeup); if (nextexp != KTIME_MAX) { if (nextexp != tmc->cpuevt.nextevt.expires || - tmc->cpuevt.ignore) { + READ_ONCE(tmc->cpuevt.ignore)) { ret = tmigr_new_timer(tmc, nextexp); /* * Make sure the reevaluation of timers in idle path @@ -1362,7 +1362,7 @@ static u64 __tmigr_cpu_deactivate(struct tmigr_cpu *tmc, u64 nextexp) * or CPU goes offline. */ if (nextexp != KTIME_MAX) - tmc->cpuevt.ignore = false; + WRITE_ONCE(tmc->cpuevt.ignore, false); walk_groups(&tmigr_inactive_up, &data, tmc); return data.firstexp; diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c index aa59919b8f2c..0e4b499328c0 100644 --- a/kernel/time/vsyscall.c +++ b/kernel/time/vsyscall.c @@ -41,14 +41,12 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti nsec = tk->tkr_mono.xtime_nsec; nsec += ((u64)tk->wall_to_monotonic.tv_nsec << tk->tkr_mono.shift); - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* Copy MONOTONIC time for BOOTTIME */ sec = vdso_ts->sec; + nsec = vdso_ts->nsec; /* Add the boot offset */ sec += tk->monotonic_to_boot.tv_sec; nsec += (u64)tk->monotonic_to_boot.tv_nsec << tk->tkr_mono.shift; @@ -56,12 +54,8 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti /* CLOCK_BOOTTIME */ vdso_ts = &vc[CS_HRES_COARSE].basetime[CLOCK_BOOTTIME]; vdso_ts->sec = sec; - - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* CLOCK_MONOTONIC_RAW */ vdso_ts = &vc[CS_RAW].basetime[CLOCK_MONOTONIC_RAW]; @@ -161,11 +155,11 @@ void vdso_time_update_aux(struct timekeeper *tk) vdso_ts->sec = tk->xtime_sec + tk->monotonic_to_aux.tv_sec; - nsec = tk->tkr_mono.xtime_nsec >> tk->tkr_mono.shift; - nsec += tk->monotonic_to_aux.tv_nsec; - vdso_ts->sec += __iter_div_u64_rem(nsec, NSEC_PER_SEC, &nsec); - nsec = nsec << tk->tkr_mono.shift; - vdso_ts->nsec = nsec; + nsec = tk->tkr_mono.xtime_nsec; + nsec += (u64)tk->monotonic_to_aux.tv_nsec << tk->tkr_mono.shift; + vdso_ts->sec += __iter_div64_u64_rem(nsec, + (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); } __arch_update_vdso_clock(vc); diff --git a/lib/bitmap.c b/lib/bitmap.c index ed685127a107..d1cb8a507c60 100644 --- a/lib/bitmap.c +++ b/lib/bitmap.c @@ -308,6 +308,23 @@ bool __bitmap_intersects(const unsigned long *bitmap1, } EXPORT_SYMBOL(__bitmap_intersects); +bool __bitmap_intersects_and(const unsigned long *bitmap1, + const unsigned long *bitmap2, + const unsigned long *bitmap3, unsigned int bits) +{ + unsigned int k, lim = bits / BITS_PER_LONG; + + for (k = 0; k < lim; ++k) + if (bitmap1[k] & bitmap2[k] & bitmap3[k]) + return true; + + if (bits % BITS_PER_LONG) + if ((bitmap1[k] & bitmap2[k] & bitmap3[k]) & BITMAP_LAST_WORD_MASK(bits)) + return true; + return false; +} +EXPORT_SYMBOL(__bitmap_intersects_and); + bool __bitmap_subset(const unsigned long *bitmap1, const unsigned long *bitmap2, unsigned int bits) { diff --git a/lib/crc/x86/crc-pclmul-template.h b/lib/crc/x86/crc-pclmul-template.h index 02744831c6fa..893119bb7c07 100644 --- a/lib/crc/x86/crc-pclmul-template.h +++ b/lib/crc/x86/crc-pclmul-template.h @@ -27,16 +27,14 @@ DEFINE_STATIC_CALL(prefix##_pclmul, prefix##_pclmul_sse) static inline bool have_vpclmul(void) { return boot_cpu_has(X86_FEATURE_VPCLMULQDQ) && - boot_cpu_has(X86_FEATURE_AVX2) && - cpu_has_xfeatures(XFEATURE_MASK_YMM, NULL); + boot_cpu_has(X86_FEATURE_AVX2); } static inline bool have_avx512(void) { return boot_cpu_has(X86_FEATURE_AVX512BW) && boot_cpu_has(X86_FEATURE_AVX512VL) && - !boot_cpu_has(X86_FEATURE_PREFER_YMM) && - cpu_has_xfeatures(XFEATURE_MASK_AVX512, NULL); + !boot_cpu_has(X86_FEATURE_PREFER_YMM); } /* diff --git a/lib/crypto/x86/blake2s.h b/lib/crypto/x86/blake2s.h index f8eed6cb042e..0f7c51f055c8 100644 --- a/lib/crypto/x86/blake2s.h +++ b/lib/crypto/x86/blake2s.h @@ -55,8 +55,6 @@ static void blake2s_mod_init_arch(void) if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2) && boot_cpu_has(X86_FEATURE_AVX512F) && - boot_cpu_has(X86_FEATURE_AVX512VL) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM | - XFEATURE_MASK_AVX512, NULL)) + boot_cpu_has(X86_FEATURE_AVX512VL)) static_branch_enable(&blake2s_use_avx512); } diff --git a/lib/crypto/x86/chacha.h b/lib/crypto/x86/chacha.h index 10cf8f1c569d..c79562aac56b 100644 --- a/lib/crypto/x86/chacha.h +++ b/lib/crypto/x86/chacha.h @@ -165,8 +165,7 @@ static void chacha_mod_init_arch(void) static_branch_enable(&chacha_use_simd); if (boot_cpu_has(X86_FEATURE_AVX) && - boot_cpu_has(X86_FEATURE_AVX2) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) { + boot_cpu_has(X86_FEATURE_AVX2)) { static_branch_enable(&chacha_use_avx2); if (boot_cpu_has(X86_FEATURE_AVX512VL) && diff --git a/lib/crypto/x86/nh.h b/lib/crypto/x86/nh.h index 83361c2e9783..342636dcb750 100644 --- a/lib/crypto/x86/nh.h +++ b/lib/crypto/x86/nh.h @@ -37,9 +37,7 @@ static void nh_mod_init_arch(void) { if (boot_cpu_has(X86_FEATURE_XMM2)) { static_branch_enable(&have_sse2); - if (boot_cpu_has(X86_FEATURE_AVX2) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - NULL)) + if (boot_cpu_has(X86_FEATURE_AVX2)) static_branch_enable(&have_avx2); } } diff --git a/lib/crypto/x86/poly1305.h b/lib/crypto/x86/poly1305.h index ee92e3740a78..b061b9926fa5 100644 --- a/lib/crypto/x86/poly1305.h +++ b/lib/crypto/x86/poly1305.h @@ -143,15 +143,12 @@ static void poly1305_emit(const struct poly1305_state *ctx, #define poly1305_mod_init_arch poly1305_mod_init_arch static void poly1305_mod_init_arch(void) { - if (boot_cpu_has(X86_FEATURE_AVX) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) + if (boot_cpu_has(X86_FEATURE_AVX)) static_branch_enable(&poly1305_use_avx); - if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) + if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2)) static_branch_enable(&poly1305_use_avx2); if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2) && boot_cpu_has(X86_FEATURE_AVX512F) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM | XFEATURE_MASK_AVX512, NULL) && /* Skylake downclocks unacceptably much when using zmm, but later generations are fast. */ boot_cpu_data.x86_vfm != INTEL_SKYLAKE_X) static_branch_enable(&poly1305_use_avx512); diff --git a/lib/crypto/x86/sha1.h b/lib/crypto/x86/sha1.h index c48a0131fd12..6aff433466e7 100644 --- a/lib/crypto/x86/sha1.h +++ b/lib/crypto/x86/sha1.h @@ -59,9 +59,7 @@ static void sha1_mod_init_arch(void) { if (boot_cpu_has(X86_FEATURE_SHA_NI)) { static_call_update(sha1_blocks_x86, sha1_blocks_ni); - } else if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - NULL) && - boot_cpu_has(X86_FEATURE_AVX)) { + } else if (boot_cpu_has(X86_FEATURE_AVX)) { if (boot_cpu_has(X86_FEATURE_AVX2) && boot_cpu_has(X86_FEATURE_BMI1) && boot_cpu_has(X86_FEATURE_BMI2)) diff --git a/lib/crypto/x86/sha256.h b/lib/crypto/x86/sha256.h index 0ee69d8e39fe..e98ffdaf4b14 100644 --- a/lib/crypto/x86/sha256.h +++ b/lib/crypto/x86/sha256.h @@ -104,9 +104,7 @@ static void sha256_mod_init_arch(void) boot_cpu_has(X86_FEATURE_PHE_EN) && boot_cpu_data.x86 >= 0x07) { static_call_update(sha256_blocks_x86, sha256_blocks_phe); - } else if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, - NULL) && - boot_cpu_has(X86_FEATURE_AVX)) { + } else if (boot_cpu_has(X86_FEATURE_AVX)) { if (boot_cpu_has(X86_FEATURE_AVX2) && boot_cpu_has(X86_FEATURE_BMI2)) static_call_update(sha256_blocks_x86, diff --git a/lib/crypto/x86/sha512.h b/lib/crypto/x86/sha512.h index 0213c70cedd0..4e177b4606bd 100644 --- a/lib/crypto/x86/sha512.h +++ b/lib/crypto/x86/sha512.h @@ -37,8 +37,7 @@ static void sha512_blocks(struct sha512_block_state *state, #define sha512_mod_init_arch sha512_mod_init_arch static void sha512_mod_init_arch(void) { - if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL) && - boot_cpu_has(X86_FEATURE_AVX)) { + if (boot_cpu_has(X86_FEATURE_AVX)) { if (boot_cpu_has(X86_FEATURE_AVX2) && boot_cpu_has(X86_FEATURE_BMI2)) static_call_update(sha512_blocks_x86, diff --git a/lib/crypto/x86/sm3.h b/lib/crypto/x86/sm3.h index 3834780f2f6a..e06d4a22e4fa 100644 --- a/lib/crypto/x86/sm3.h +++ b/lib/crypto/x86/sm3.h @@ -33,7 +33,6 @@ static void sm3_blocks(struct sm3_block_state *state, #define sm3_mod_init_arch sm3_mod_init_arch static void sm3_mod_init_arch(void) { - if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_BMI2) && - cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) + if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_BMI2)) static_call_update(sm3_blocks_x86, sm3_blocks_avx); } diff --git a/lib/raid/xor/Makefile b/lib/raid/xor/Makefile index 9b0fad459cdb..e1e3455c219d 100644 --- a/lib/raid/xor/Makefile +++ b/lib/raid/xor/Makefile @@ -31,7 +31,7 @@ xor-$(CONFIG_SPARC32) += sparc/xor-sparc32.o xor-$(CONFIG_SPARC64) += sparc/xor-sparc64.o sparc/xor-sparc64-glue.o xor-$(CONFIG_S390) += s390/xor.o xor-$(CONFIG_X86_32) += x86/xor-avx.o x86/xor-sse.o x86/xor-mmx.o -xor-$(CONFIG_X86_64) += x86/xor-avx.o x86/xor-sse.o +xor-$(CONFIG_X86_64) += x86/xor-avx.o x86/xor-sse.o x86/xor-avx512.o obj-y += tests/ CFLAGS_xor-neon.o += $(CC_FLAGS_FPU) -I$(src)/$(SRCARCH) diff --git a/lib/raid/xor/x86/xor-avx512.c b/lib/raid/xor/x86/xor-avx512.c new file mode 100644 index 000000000000..c11d83441875 --- /dev/null +++ b/lib/raid/xor/x86/xor-avx512.c @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * AVX-512 optimized implementation of xor_gen() + * + * Copyright 2026 Google LLC + */ + +#include <linux/types.h> +#include <asm/fpu/api.h> +#include "xor_impl.h" +#include "xor_arch.h" + +/* + * Implementation notes: + * + * Unrolling by the number of buffers (2-5) is very important. + * + * Unrolling by length is less important, especially when using register-indexed + * addressing with negative indices from the end of the buffers. That approach + * results in just two loop control instructions being needed per iteration, + * regardless of the number of buffers. + * + * In fact, benchmarks showed that the 2 and 3 buffer cases require only 2x + * unrolling by length, while the 4 and 5 buffer cases don't require any + * unrolling by length. Benchmarks also showed that the register-indexed + * addressing isn't a bottleneck either; i.e., we can't do any better by + * incrementing the pointers as we go along, even with more unrolling. + */ + +static void xor_avx512_2(long bytes, u8 *p1, const u8 *p2) +{ + long i = -bytes; + + asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n" + "vmovdqa64 64(%1,%0), %%zmm1\n" + "vpxorq (%2,%0), %%zmm0, %%zmm0\n" + "vpxorq 64(%2,%0), %%zmm1, %%zmm1\n" + "vmovdqa64 %%zmm0, (%1,%0)\n" + "vmovdqa64 %%zmm1, 64(%1,%0)\n" + "add $128, %0\n" + "jnz 1b\n" + : "+&r"(i) + : "r"(p1 + bytes), "r"(p2 + bytes) + : "memory", "cc"); +} + +static void xor_avx512_3(long bytes, u8 *p1, const u8 *p2, const u8 *p3) +{ + long i = -bytes; + + asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n" + "vmovdqa64 64(%1,%0), %%zmm1\n" + "vmovdqa64 (%2,%0), %%zmm2\n" + "vmovdqa64 64(%2,%0), %%zmm3\n" + "vpternlogq $0x96, (%3,%0), %%zmm2, %%zmm0\n" + "vpternlogq $0x96, 64(%3,%0), %%zmm3, %%zmm1\n" + "vmovdqa64 %%zmm0, (%1,%0)\n" + "vmovdqa64 %%zmm1, 64(%1,%0)\n" + "add $128, %0\n" + "jnz 1b\n" + : "+&r"(i) + : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes) + : "memory", "cc"); +} + +static void xor_avx512_4(long bytes, u8 *p1, const u8 *p2, const u8 *p3, + const u8 *p4) +{ + long i = -bytes; + + asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n" + "vmovdqa64 (%2,%0), %%zmm1\n" + "vpxorq (%3,%0), %%zmm0, %%zmm0\n" + "vpternlogq $0x96, (%4,%0), %%zmm1, %%zmm0\n" + "vmovdqa64 %%zmm0, (%1,%0)\n" + "add $64, %0\n" + "jnz 1b\n" + : "+&r"(i) + : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes), + "r"(p4 + bytes) + : "memory", "cc"); +} + +static void xor_avx512_5(long bytes, u8 *p1, const u8 *p2, const u8 *p3, + const u8 *p4, const u8 *p5) +{ + long i = -bytes; + + asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n" + "vmovdqa64 (%2,%0), %%zmm1\n" + "vpternlogq $0x96, (%3,%0), %%zmm1, %%zmm0\n" + "vmovdqa64 (%4,%0), %%zmm1\n" + "vpternlogq $0x96, (%5,%0), %%zmm1, %%zmm0\n" + "vmovdqa64 %%zmm0, (%1,%0)\n" + "add $64, %0\n" + "jnz 1b\n" + : "+&r"(i) + : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes), + "r"(p4 + bytes), "r"(p5 + bytes) + : "memory", "cc"); +} + +DO_XOR_BLOCKS(avx512_inner, xor_avx512_2, xor_avx512_3, xor_avx512_4, + xor_avx512_5); + +/* + * Preconditions: bytes is a nonzero multiple of 512, and all buffers are + * 64-byte aligned. + */ +static void xor_gen_avx512(void *dest, void **srcs, unsigned int src_cnt, + unsigned int bytes) +{ + kernel_fpu_begin(); + xor_gen_avx512_inner(dest, srcs, src_cnt, bytes); + asm volatile("vzeroupper"); + kernel_fpu_end(); +} + +struct xor_block_template xor_block_avx512 = { + .name = "avx512", + .xor_gen = xor_gen_avx512, +}; diff --git a/lib/raid/xor/x86/xor_arch.h b/lib/raid/xor/x86/xor_arch.h index 99fe85a213c6..ed5921d2e2aa 100644 --- a/lib/raid/xor/x86/xor_arch.h +++ b/lib/raid/xor/x86/xor_arch.h @@ -6,22 +6,31 @@ extern struct xor_block_template xor_block_p5_mmx; extern struct xor_block_template xor_block_sse; extern struct xor_block_template xor_block_sse_pf64; extern struct xor_block_template xor_block_avx; +extern struct xor_block_template xor_block_avx512; -/* - * When SSE is available, use it as it can write around L2. We may also be able - * to load into the L1 only depending on how the cpu deals with a load to a line - * that is being prefetched. - * - * When AVX2 is available, force using it as it is better by all measures. - * - * 32-bit without MMX can fall back to the generic routines. - */ static __always_inline void __init arch_xor_init(void) { - if (boot_cpu_has(X86_FEATURE_AVX) && - boot_cpu_has(X86_FEATURE_OSXSAVE)) { + if (IS_ENABLED(CONFIG_X86_64) && boot_cpu_has(X86_FEATURE_AVX512F) && + !boot_cpu_has(X86_FEATURE_PREFER_YMM)) { + /* + * Use the AVX-512 code on CPUs that support AVX-512 without + * overly-eager downclocking. On such CPUs the AVX-512 code + * should always work at least as well as the AVX code, so + * runtime selection is unnecessary. + * + * The AVX-512 code can work on X86_32. However, due to lack of + * use case for that, for now it's built only for X86_64. + */ + xor_force(&xor_block_avx512); + } else if (boot_cpu_has(X86_FEATURE_AVX)) { + /* AVX will be the best; no need to try others. */ xor_force(&xor_block_avx); } else if (IS_ENABLED(CONFIG_X86_64) || boot_cpu_has(X86_FEATURE_XMM)) { + /* + * When SSE is available, use it as it can write around L2. We + * may also be able to load into the L1 only depending on how + * the cpu deals with a load to a line that is being prefetched. + */ xor_register(&xor_block_sse); xor_register(&xor_block_sse_pf64); } else if (boot_cpu_has(X86_FEATURE_MMX)) { diff --git a/lib/vdso/gettimeofday.c b/lib/vdso/gettimeofday.c index f7a591aba59f..ef4dcc614489 100644 --- a/lib/vdso/gettimeofday.c +++ b/lib/vdso/gettimeofday.c @@ -285,6 +285,7 @@ __cvdso_clock_gettime_common(const struct vdso_time_data *vd, clockid_t clock, * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ + BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (likely(msk & VDSO_HRES)) vc = &vc[CS_HRES_COARSE]; @@ -438,6 +439,7 @@ bool __cvdso_clock_getres_common(const struct vdso_time_data *vd, clockid_t cloc * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ + BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (msk & (VDSO_HRES | VDSO_RAW)) { /* diff --git a/net/core/pktgen.c b/net/core/pktgen.c index a89cb0760821..d63eeadddfcc 100644 --- a/net/core/pktgen.c +++ b/net/core/pktgen.c @@ -2344,11 +2344,11 @@ static void spin(struct pktgen_dev *pkt_dev, ktime_t spin_until) set_current_state(TASK_INTERRUPTIBLE); hrtimer_sleeper_start_expires(&t, HRTIMER_MODE_ABS); - if (likely(t.task)) + if (likely(hrtimer_sleeper_task_get(&t))) schedule(); hrtimer_cancel(&t.timer); - } while (t.task && pkt_dev->running && !signal_pending(current)); + } while (hrtimer_sleeper_task_get(&t) && pkt_dev->running && !signal_pending(current)); __set_current_state(TASK_RUNNING); end_time = ktime_get(); } diff --git a/tools/objtool/Documentation/klp-test-design.txt b/tools/objtool/Documentation/klp-test-design.txt new file mode 100644 index 000000000000..2082c277197f --- /dev/null +++ b/tools/objtool/Documentation/klp-test-design.txt @@ -0,0 +1,286 @@ +.. SPDX-License-Identifier: GPL-2.0 + +====================================== +Design of the objtool klp test harness +====================================== + +tools/objtool/tests/ holds unit tests for the klp subcommands of +objtool -- ``klp checksum``, ``klp diff``, ``klp post-link`` and +``--klp-symids`` -- which together turn two builds of the kernel into a +livepatch module. + +This document explains how the harness is built and why. For the rules to +follow when adding a test, see klp-write-tests.txt. + + +TL;DR +===== + +One run covers one compiler and one architecture; CI runs the combinations. +Build objtool first -- it needs libelf and libxxhash -- and the same ARCH is +used for both steps. + +Natively, with gcc:: + + make -C tools/objtool + make -C tools/objtool tests + +Natively, with clang -- LLVM=1 additionally selects the LLVM binutils:: + + CC=clang make -C tools/objtool tests + LLVM=1 make -C tools/objtool tests + +Cross, with gcc -- an arm64 host running the x86 tests:: + + ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool + ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool tests + +Cross, with clang. It defaults to the host triple however it is invoked, so +--target= is what makes it emit x86; OBJCOPY is needed because BFD's is +usually built for the host's target alone:: + + ARCH=x86_64 make -C tools/objtool + ARCH=x86_64 CC="clang --target=x86_64-linux-gnu" OBJCOPY=llvm-objcopy \ + make -C tools/objtool tests + +A run ends with a totals line; anything other than fail:0 is a real result:: + + # pass:48 fail:0 static-skip:1 probe-skip:0 xfail:0 xpass:0 + +Useful extras:: + + tools/objtool/tests/run-tests.sh basic # one test, by name + tools/objtool/tests/run-tests.sh --keep basic # and keep what it built + + +Why unit tests are possible at all +================================== + +klp-build is a pipeline: build the kernel twice, checksum both, diff them, +link the result. Testing that end to end means two kernel builds per case, +which is too slow to run often and too heavy to keep in the tree. + +Three properties make a much cheaper test possible. + +**objtool has no configuration-dependent logic.** It never reads ``.config``. +Every ``CONFIG_`` string in its source is a comment or one error message, and +its only build-time conditionals are driven by host libraries and the target +architecture. Configuration reaches objtool through exactly two channels: the +``objtool-args-$(CONFIG_*)`` lines in scripts/Makefile.lib, and the +contents of the object handed to it. + +**The klp subcommands use none of the first channel.** Of objtool's options +they consult three -- ``checksum``, ``debug_checksum``, ``dryrun`` -- all from +their own command line. So klp behaviour varies with configuration *only* +through the input object. + +**Therefore a test can reproduce any configuration's behaviour by reproducing +its input.** Compile a small freestanding fixture with the flags that +configuration would have used, and objtool cannot tell the difference. No +kernel, no ``.config``, no object cache. + +The whole suite runs in a few seconds. + + +Shape of a test +=============== + +Each test compiles one fixture twice -- once plain, once with ``-DPATCHED`` -- +runs ``klp checksum`` over both, diffs them, and asserts on properties of the +output object:: + + . "$(dirname "$0")/../lib.sh" + + setup + build_pair basic.c + + assert_input_symbol changed + run_diff + + assert_patched changed + assert_not_patched untouched + + pass "changed function cloned, unchanged function left alone" + +Assertions check properties, never recorded output. Codegen varies between +compilers and versions, so a golden file would report churn rather than +regressions. + + +Layout +====== + +:: + + tools/objtool/tests/ + lib.sh the harness: everything a test may call + run-tests.sh selects, runs and classifies + generic/ + test-*.sh + fixtures/*.c + x86/ + test-*.sh + fixtures/*.c + +Which architecture a test is for is expressed by where it lives. The runner +executes ``generic/`` plus the directory matching this architecture, so a test +which cannot apply is not run rather than running in order to report that it +did not. There is no ``x86_only`` helper, and no lookup letting an +architecture fixture shadow a generic one: an architecture-specific test +carries its own fixtures. + +Compilers cannot be expressed the same way, because CI varies ``CC`` over the +same tree. A compiler requirement stays a declaration inside the test +(``gcc_only``, ``clang_only``). + + +The environment is established once +=================================== + +Sourcing lib.sh runs ``klp_preflight``, which checks that objtool +exists and has klp support, that ``$CC`` works, that the binutils are present, +and which architecture this is. The answers are exported, so: + +* ``run-tests.sh`` sources lib.sh too, and therefore knows the + architecture before it chooses which tests to run; +* each test inherits the answers rather than repeating the work; +* a test run on its own establishes them for itself. + +Preflight answers only whether the suite can run at all. A suite which cannot +run must not exit 0 looking like one which passed, so a missing objtool fails +the whole run with a TAP ``Bail out!`` rather than skipping each test in turn. +What a *particular* compiler can do is a different question, left to the test +which cares. + + +Outcomes +======== + +Output is TAP. The distinction the harness cares most about is between kinds +of skip, because a skip is how a suite quietly stops testing anything: + +``declared`` + The test said in advance it does not apply -- ``gcc_only`` on a clang run. + Expected indefinitely. + +``probe`` + The construct did not turn up in the built object this time. Weaker: one + which becomes permanent is a fixture that has stopped testing anything. + +``undeclared`` + Counted as a **failure**. A test which gives up for a reason it never + declared is a hole, not an outcome. + +``xfail``/``xpass`` come with them, so a known failure is reported rather than +commented out, and one which starts passing says so instead of going quietly +green. + +The runner classifies the TAP result line, not everything a test printed: +objtool warns on stderr and that output is captured, so a stray line ahead of +the result would otherwise leave the exit status to decide -- and an expected +failure exits 0. + +A run ends with a totals line:: + + # pass:48 fail:0 static-skip:1 probe-skip:0 xfail:0 xpass:0 + +and reports what it left out:: + + # not run: 5 tests in x86/ (this run is arm64) + + +Working directories +=================== + +A run gets one directory; each test gets a subdirectory of it, mirroring the +source layout:: + + /tmp/klp-tests.XXXXXXXX/ + generic/test-basic/{orig.o,patched.o,out.o,Module.symvers,...} + x86/test-kcfi/... + +By default (``KEEP=failed``) only failing tests keep their directories; the +runner reports where they are. ``KEEP=all`` keeps every test's directory; +``KEEP=none`` removes them all. The runner ``rmdir``s the run directory when +it is empty -- which fails if anything was left behind unexpectedly, so a test +which dies without cleaning up is reported rather than silently leaking. + + +Running +======= + +:: + + make -C tools/objtool # needs libelf and libxxhash + make -C tools/objtool tests + + CC=clang make -C tools/objtool tests # the other toolchain + LLVM=1 make -C tools/objtool tests # and its binutils too + + make -C tools/objtool tests KEEP=all # keep every test's workdir + make -C tools/objtool tests KEEP=none # remove all workdirs + + tools/objtool/tests/run-tests.sh basic # one test; failures kept by default + +A run covers one compiler and one architecture; CI runs the combinations. + +Cross-compiled runs +------------------- + +objtool klp is built only where ARCH_HAS_KLP is set, which today means x86 -- +so an arm64 machine cannot run any of this natively. It can run all of it +cross, because objtool is a host tool that only reads and rewrites ELF, and +the tests only compile fixtures and inspect the objects. Nothing has to +execute target code. + +:: + + ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool + ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool tests + +objtool itself stays a native binary: it is built with HOSTCC, not CC, so +setting a cross compiler cannot produce one the host is unable to run. ARCH +selects both the objtool target and the directory of tests to run. + +clang needs telling, since it defaults to the host triple however it is +invoked. LLVM=1 with CROSS_COMPILE does that for you -- the --target= it +derives is forwarded to the tests -- and the fixtures include no kernel +headers, so no sysroot is needed: + +:: + + ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- LLVM=1 \ + make -C tools/objtool tests + +Naming the compiler by hand works too, and is what to reach for when the +target triple is not the one CROSS_COMPILE implies: + +:: + + ARCH=x86_64 CC="clang --target=x86_64-linux-gnu" \ + OBJCOPY=llvm-objcopy make -C tools/objtool tests + +CROSS_COMPILE picks the binutils, and each can be overridden on its own. +readelf reads any target and rarely needs overriding; BFD's objcopy is usually +built for the host's alone, hence OBJCOPY=llvm-objcopy above, or install +binutils-multiarch. + +Either readelf will do. The assertions read readelf's output, and the two +spell some of it differently -- GNU prints "OS [0xff20]" for SHN_LIVEPATCH +where llvm-readelf prints "OS[0xff20]" -- so they accept both. + +Getting this wrong is easy and the harness refuses rather than producing a +misleading result. "CC=clang ARCH=x86_64" alone selects the x86 tests and +then builds arm64 objects; preflight compiles a probe object, hands it to +objtool, and stops the run if they disagree about the architecture, or if +ARCH does not match what the compiler emits. + + +What this does not cover +======================== + +These are unit tests for objtool's klp subcommands. They do not build a +kernel, do not run scripts/livepatch/klp-build, and do not load a +livepatch. Behaviour which only appears when the kernel applies a patch -- +the module loader refusing a relocation, late module patching ordering -- has +to be tested by booting, and is out of scope here. diff --git a/tools/objtool/Documentation/klp-write-tests.txt b/tools/objtool/Documentation/klp-write-tests.txt new file mode 100644 index 000000000000..eb2cacaf5720 --- /dev/null +++ b/tools/objtool/Documentation/klp-write-tests.txt @@ -0,0 +1,266 @@ +.. SPDX-License-Identifier: GPL-2.0 + +===================================== +Writing a test for objtool's klp code +===================================== + +Instructions for adding a test to tools/objtool/tests/. Read +klp-test-design.txt first if you need to know how the harness works; this +document is the procedure and the rules. + +Two kinds of request bring you here: + +* *"write a test for commit <sha>"* -- a fix went in without one. +* *"port the test at <location>, written against another harness"* -- a case + exists elsewhere and should live in tree. + +Both follow the same procedure. + + +The one rule that matters +========================= + +**A test is not finished until you have watched it fail.** + +Break the thing it guards -- revert the fix, or sabotage the exact line -- and +confirm the test fails. Then restore and confirm it passes. A test that has +never failed is not known to test anything, and this suite has produced +several that passed against deliberately broken code: + +* an alternatives fixture whose empty entry pointed at its own end label rather + than the neighbour's replacement, so the bug it guarded made no difference; +* a sympos fixture where symbol-table order and address order agreed, so + counting and reading the linked image gave the same answer; +* a string fixture using a named ``char[]``, which never reached the + contents-hashing path because that keys on ``SHF_STRINGS``; +* a static array the compiler proved constant, folded to zero, and emitted no + relocation for -- so the two builds were byte-identical. + +Every one looked correct. Say in the commit message how you verified, and if +you could not isolate the behaviour to a single line, **say that too** rather +than implying otherwise. + + +Procedure +========= + +1. **Read the fix.** What input reaches the broken line? What is observable + in the output object when it misbehaves -- a missing section, a relocation + naming the wrong symbol, an unchanged checksum, a rejected build? If + nothing is observable, stop and say so; see `When to give up`_. + +2. **Decide where it lives.** ``generic/`` unless the fixture needs + architecture-specific assembly or the behaviour is architecture-specific, + in which case ``x86/`` (or a new directory named for the architecture). + +3. **Write the fixture** in the same directory's ``fixtures/``. Reuse an + existing one if it already produces the shape; add a ``-D`` knob rather + than copying a fixture to change one line. + +4. **Write the test.** Assert the *premise* before the result -- see + `State the premise`_. + +5. **Verify by breaking the code.** Then restore. + +6. **Run the whole suite under both compilers**:: + + make -C tools/objtool tests + CC=clang make -C tools/objtool tests + +7. **Commit** the test and its fixture together, alone. One test per commit. + + +Writing the fixture +=================== + +Fixtures are freestanding C. No kernel headers -- write out the kernel +structure by hand if you need one, as the existing special-section fixtures do. + +Every fixture needs a ``.modinfo`` name, because klp diff reads the object's +module name from it:: + + static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +Use ``MODNAME`` if the test needs to vary it. Note that most fixtures hardcode +``vmlinux``: passing ``-DMODNAME`` to one that does silently does nothing and +the test quietly becomes a vmlinux test. + +The patched build is selected with ``-DPATCHED``. For a fixture with several +variants, gate each on both, so the original is always the baseline:: + + #if defined(PATCHED) && defined(WHICH_CALL) + r = callee_b(x); + #else + r = callee_a(x); + #endif + +and select one per build: ``build_pair foo.c -DWHICH_CALL``. Without the +``defined(PATCHED)`` the flag applies to *both* builds and nothing differs. + +Traps that have bitten before +----------------------------- + +* **The compiler optimises your fixture away.** A static never written is + proved constant, its reads folded, and no relocation emitted. Add a writer + the compiler cannot see through. +* **String literals versus named arrays.** The contents-hashing path keys on + ``SHF_STRINGS``, which the compiler sets on the mergeable section a *literal* + lands in, not on a ``char[]`` given a section of its own. +* **Per-function sections hide movement.** With the default + ``-ffunction-sections`` every function sits at offset 0 of its own section, + so nothing ever moves. A test about position needs + ``build_pair foo.c -fno-function-sections``. +* **Special sections need boundaries.** Either an entsize on the section or an + ``ANNOTATE_DATA_SPECIAL`` annotation, or klp diff reports "missing special + section entsize or annotations". Their targets need real (global) symbols, + or it reports "failed to convert reloc sym". +* **Prefer letting objtool generate what objtool generates.** + ``.static_call_sites``, ``.mcount_loc``, ``.ibt_endbr_seal`` and ORC come + from its check pass. Call ``run_objtool_check --mcount`` and let it build + them; a hand-written copy tests your reading of the format, not the format. + + Write one by hand only when the test needs a shape objtool will not produce, + and say so in the fixture. generic/fixtures/static_call.c does: objtool + emits the site but not the ``ANNOTATE_DATA_SPECIAL`` that describes its + boundaries -- those come from the kernel's macros -- so a fixture which has + to vary whether the annotation is there writes both itself. + + +Writing the test +================ + +Start from the shortest existing test, generic/test-basic.sh. + +State the premise +----------------- + +A test which asserts only on the output passes when the compiler never emitted +the construct in the first place, and reads as coverage it does not have. Say +what the input must contain:: + + assert_input_section __jump_table # the fixture must produce it -> fail + require_input_section .kcfi_traps # this compiler may not -> skip + +Prefer ``assert_*``. Reach for ``require_*`` only where absence genuinely +depends on compiler version or flags, and follow it with something +unconditional so the test can never be entirely vacuous. + +Where a test would otherwise duplicate a sibling, assert what makes it +different. ``test-jump-label-module-static-key`` checks that the key really is +reached through its section symbol -- without that it is a second copy of +``test-jump-label-module-key``. + +Assert both directions +---------------------- + +Check that the right thing happened *and* that the wrong thing did not. A klp +diff which clones everything is as wrong as one which clones nothing:: + + assert_patched changed + assert_not_patched untouched + +Skips +----- + +* ``gcc_only``/``clang_only`` -- a settled fact about the compiler. Declared, + so it reads as expected forever. +* ``probe_skip`` -- this toolchain did not produce the construct. Include what + to do about it if there is anything:: + + probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_CC and THIN_LD to one" + +* A bare ``skip`` is **counted as a failure**. Never use it. + +Standing in for a kernel configuration +-------------------------------------- + +``FIXTURE_CFLAGS`` is what a fixture is built with. Since objtool reads no +``.config``, changing these flags is how a test covers a configuration without +building a kernel. Two ways: + +* trailing arguments to ``build_pair``/``build_one``, which win, and cover + anything expressible as a negation:: + + build_pair foo.c -fno-function-sections + +* otherwise assign ``FIXTURE_CFLAGS`` before building. + +Either way **say in a comment which kernel configuration the change stands in +for**. A flag with no stated motive is indistinguishable from a mistake. + + +What the test's comment must say +================================ + +The comment at the top is the test's justification. It should let a reader +decide, without archaeology, whether a skip or a failure matters. Include: + +* **what breaks** in the running kernel if the behaviour regresses -- not the + mechanism, the consequence; +* **why it is not caught otherwise**, which is usually "nothing fails at build + time"; +* **the fix commit** it guards, if there is one; +* **anything load-bearing about the fixture** that is not obvious, especially + anything you got wrong first. + +That last point is the one people skip. If the fixture has to be built without +per-function sections, or the static must not be named ``__warned``, or the key +must be file-local -- write it down, or the next person will simplify it away. + + +Porting a test from another harness +=================================== + +Read the original's *case*, not its code. The other harness probably builds a +real kernel module; here you write freestanding C. A transliteration will +usually test something else. + +* Work out which objtool behaviour the case exercises, then produce that shape + the cheapest way here. +* Verify by breaking the code, exactly as for a new test -- a port is not + correct because the original was. +* If the original names a fix commit, cite it. +* Credit the source in the commit message with the trailers the original + carried, followed by your own. + +Sometimes the port shows the case is already covered, and sometimes it shows +the case cannot be reproduced here. Both are results; report them rather than +committing something that passes vacuously. + + +When to give up +=============== + +Some behaviour cannot be reached from a compiled fixture. Say so, with what +you tried, instead of committing a test that passes either way. Examples that +were genuinely abandoned: + +* a memory leak -- needs valgrind, not an assertion on ELF; +* ``mkstemp`` with long paths, and other I/O edge cases; +* a NULL dereference reachable only through a debug path; +* changes made redundant by a fallback: removing the code changes no output + because something else already handles the case; +* a fix whose code has since been rewritten, so there is nothing left to + revert. + +Also stop when the behaviour depends on something outside the fixture's +control -- an ELF library's handling of empty sections, or a compiler version's +naming of anonymous data. A test which passes for you and skips for everyone +else is worse than none. + + +Checklist +========= + +Before committing: + +* the test fails with the code broken, and passes with it fixed +* the whole suite passes under **both** gcc and clang +* the premise is asserted, not assumed +* both directions are asserted where that applies +* no bare ``skip`` +* the fixture is in the same directory as the test +* the comment names the consequence, the fix commit, and anything load-bearing +* the commit contains one test and its fixtures, and nothing else +* the commit message says how you verified it diff --git a/tools/objtool/Makefile b/tools/objtool/Makefile index 4cc2e756af84..db99c4759b2c 100644 --- a/tools/objtool/Makefile +++ b/tools/objtool/Makefile @@ -153,6 +153,19 @@ clean: $(LIBSUBCMD)-clean mrproper: clean $(call QUIET_CLEAN, objtool) $(RM) $(OBJTOOL) +# The kernel's own Makefile sets OBJCOPY, to llvm-objcopy under LLVM and to +# $(CROSS_COMPILE)objcopy otherwise; tools/scripts/Makefile.include names only +# LLVM_OBJCOPY and leaves OBJCOPY unset. Forwarding it unset would hand the +# tests an empty string, and they would fall back to GNU objcopy even when +# asked for an LLVM toolchain. +OBJCOPY ?= $(if $(LLVM),$(LLVM_OBJCOPY),$(CROSS_COMPILE)objcopy) + +tests: $(OBJTOOL) + $(Q)OBJTOOL=$(abspath $(OBJTOOL)) ARCH=$(ARCH) CROSS_COMPILE=$(CROSS_COMPILE) \ + CC='$(CC) $(CLANG_CROSS_FLAGS)' LD='$(LD)' \ + READELF='$(READELF)' OBJCOPY='$(OBJCOPY)' \ + KEEP=$(KEEP) $(srctree)/tools/objtool/tests/run-tests.sh + FORCE: -.PHONY: clean mrproper FORCE +.PHONY: clean mrproper tests FORCE diff --git a/tools/objtool/tests/generic/fixtures/abs_and_addressable.c b/tools/objtool/tests/generic/fixtures/abs_and_addressable.c new file mode 100644 index 000000000000..6392ff99af42 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/abs_and_addressable.c @@ -0,0 +1,44 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Two constructs which appear all over the kernel and must not upset klp + * checksum or klp diff. + * + * An absolute symbol (SHN_ABS) has no section, so anything walking sym->sec + * without checking dereferences NULL. The kernel makes them with linker + * scripts and with .set in asm; VDSO and the fixed-address per-cpu bases are + * the usual sources. + * + * __ADDRESSABLE() emits a pointer into .discard.addressable purely to keep a + * symbol referenced. It is discarded at link time and means nothing to a + * livepatch, but the pointer is a relocation like any other and has to survive + * being looked at. + * + * Neither is the subject of the patch; the point is that their presence does + * not disturb the function that is. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* SHN_ABS, referenced from code. */ +extern char abs_sym[]; +__asm__(".globl abs_sym\n" + ".set abs_sym, 0x1234\n"); + +int helper(int x); +int helper(int x) { return x + 1; } + +/* The shape of __ADDRESSABLE(helper). */ +__asm__(".pushsection .discard.addressable, \"aw\"\n" + ".balign 8\n" + ".quad helper\n" + ".popsection\n"); + +int target(int x) +{ +#ifdef PATCHED + return helper(x) + (int)(long)abs_sym + 1; +#else + return helper(x) + (int)(long)abs_sym; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/basic.c b/tools/objtool/tests/generic/fixtures/basic.c new file mode 100644 index 000000000000..811529e7bfb1 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/basic.c @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: GPL-2.0 +/* One changed function and one unchanged function. */ + +/* klp diff takes the object's module name from .modinfo */ +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int untouched(int x) +{ + return x * 3; +} + +int changed(int x) +{ +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/changed_data.c b/tools/objtool/tests/generic/fixtures/changed_data.c new file mode 100644 index 000000000000..b52461835444 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/changed_data.c @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Data whose value differs between the two builds. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +#ifdef PATCHED +int klp_test_data = 2; +#else +int klp_test_data = 1; +#endif + +int target(int x) +{ + return x + klp_test_data; +} diff --git a/tools/objtool/tests/generic/fixtures/checksum_data.c b/tools/objtool/tests/generic/fixtures/checksum_data.c new file mode 100644 index 000000000000..6310af5c02d3 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/checksum_data.c @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Data objects whose checksums must move for reasons the raw bytes do not + * show. + * + * checksum_update_object() hashes a data symbol's length and its bytes, and + * then walks its relocations: a reference into a string section contributes + * the string's *contents*, and any other reference contributes the target + * symbol's name and the adjusted addend. So three changes that leave the + * object's own bytes identical still have to change its checksum: + * + * Each variant is selected by a -D on the patched build only, so the original + * is always the baseline: + * + * WHICH_FUNC the function pointer points somewhere else + * WHICH_STR the string pointer points at a different literal + * STR_CONTENT the string it points at is edited in place + * WHICH_SLOT the same array, at a different index: addend only + * WHICH_PRIV likewise, but a static, reached through its section symbol + * + * The last is the interesting one. Nothing in the pointer changes -- same + * section, same offset -- so a checksum that hashed only the relocation and + * not what it referred to would call the object unchanged, and the patched + * kernel would keep the old string. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int callee_a(int x); +int callee_b(int x); +int callee_a(int x) { return x + 1; } +int callee_b(int x) { return x + 2; } + +/* + * String *literals*, not named arrays. The contents-hashing path keys on + * SHF_STRINGS, which the compiler sets on the mergeable .rodata.str1.1 a + * literal lands in and not on a named char[] given a section of its own. A + * fixture using the latter exercises the ordinary name-and-addend path and + * reports nothing when the text changes. + */ +#if defined(PATCHED) && defined(STR_CONTENT) +#define MESSAGE "edited" +#else +#define MESSAGE "original" +#endif + +/* A plain data object: only its own bytes decide the checksum. */ +#if defined(PATCHED) && defined(PLAIN_VALUE) +int plain = 43; +#else +int plain = 42; +#endif + +/* + * A .bss object, where length is the only thing there is to hash: the section + * has no data, so the bytes are skipped and only sym->len distinguishes this + * from an object of another size. An initialised array would not isolate it + * -- growing one changes the hashed bytes as well. + */ +#if defined(PATCHED) && defined(LONGER) +char sized[4]; +#else +char sized[2]; +#endif + +/* + * A reference into the middle of an array: same target symbol, different + * addend. Nothing else in the object changes, so this is the only way to see + * whether the addend is hashed at all. + */ +int slots[4]; + +/* + * A file-local array. A reference to a static lands on its section symbol + * plus an offset, so the hash has to resolve that back to the underlying + * object before it has a name to hash at all -- a different code path from the + * global above, and one that silently contributes nothing when it fails. + */ +static int priv_slots[4]; + +struct desc { + int (*fn)(int arg); + const char *str; + int *slot; + int *priv; +}; + +const struct desc descriptor = { +#if defined(PATCHED) && defined(WHICH_FUNC) + .fn = callee_b, +#else + .fn = callee_a, +#endif +#if defined(PATCHED) && defined(WHICH_STR) + .str = "a different literal", +#else + .str = MESSAGE, +#endif +#if defined(PATCHED) && defined(WHICH_SLOT) + .slot = &slots[2], +#else + .slot = &slots[1], +#endif +#if defined(PATCHED) && defined(WHICH_PRIV) + .priv = &priv_slots[3], +#else + .priv = &priv_slots[1], +#endif +}; + +int target(int x) +{ + return descriptor.fn(x) + plain + sized[0] + (int)descriptor.str[0] + + *descriptor.slot + *descriptor.priv; +} diff --git a/tools/objtool/tests/generic/fixtures/checksum_insn.c b/tools/objtool/tests/generic/fixtures/checksum_insn.c new file mode 100644 index 000000000000..10f70a74a976 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/checksum_insn.c @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Instruction operands whose change must move a function's checksum even + * though the instruction bytes themselves do not. + * + * checksum_update_insn() hashes the raw bytes and then, when the instruction + * carries a relocation, what that relocation refers to: a string section + * contributes the string's contents, anything else the target symbol's name + * and the adjusted addend. A reference to a static arrives as a section + * symbol and has to be resolved back to the object first. + * + * The bytes are identical in every case below -- a rel32 operand is zero in + * the object and supplied by the relocation -- so a checksum that stopped at + * the bytes would call all of these unchanged. + * + * Each variant applies to the patched build only: + * + * WHICH_CALL calls a different function + * STR_CONTENT passes a literal whose text was edited + * WHICH_SLOT reads a different index of a global array: addend only + * WHICH_PRIV the same, for a static, reached through its section symbol + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int callee_a(int x); +int callee_b(int x); +int sink(const char *s); + +int slots[4]; + +/* + * A file-local array, plus a writer the compiler cannot see through. Without + * one it can prove the array is never written, folds every read to zero, and + * emits no relocation at all -- so the reference this is here to exercise does + * not exist. + */ +static int priv_slots[4]; + +void set_priv(int i, int v); +void set_priv(int i, int v) +{ + priv_slots[i] = v; +} + +#if defined(PATCHED) && defined(STR_CONTENT) +#define MESSAGE "edited" +#else +#define MESSAGE "original" +#endif + +int target(int x) +{ + int r; + +#if defined(PATCHED) && defined(WHICH_CALL) + r = callee_b(x); +#else + r = callee_a(x); +#endif + + r += sink(MESSAGE); + +#if defined(PATCHED) && defined(WHICH_SLOT) + r += slots[2]; +#else + r += slots[1]; +#endif + +#if defined(PATCHED) && defined(WHICH_PRIV) + r += priv_slots[3]; +#else + r += priv_slots[1]; +#endif + + return r; +} diff --git a/tools/objtool/tests/generic/fixtures/checksum_position.c b/tools/objtool/tests/generic/fixtures/checksum_position.c new file mode 100644 index 000000000000..4320bc220592 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/checksum_position.c @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A function whose position in the section changes between the two builds, + * without the function itself changing. + * + * target() calls callee() twice, and a call within the same section needs no + * relocation: the displacement is in the instruction. It is that displacement + * which moves, and hashing those bytes makes the checksum move with it. Both + * must therefore share a section, which is why the test passes + * -fno-function-sections. + * + * What has to change is the distance between the two, and PATCHED changes it + * by aligning them rather than by inserting a function between them. Where a + * compiler puts an added function is its own business: gcc emits these in + * source order, so a function written between callee() and target() separates + * them, but clang emits target() immediately before callee() whatever the + * source says, and an added function lands ahead of both. That moves target() + * without moving it relative to callee(), the displacement comes out identical + * in both builds, and the test passes without having asked anything. + * + * Alignment moves the functions apart on both, and moves neither function's + * own instructions -- which is exactly the distinction under test. + */ + +#ifdef PATCHED +#define MOVED __attribute__((aligned(64))) +#else +#define MOVED +#endif + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +__attribute__((noinline)) MOVED static int callee(int x) +{ + return x * 5 + 1; +} + +__attribute__((noinline)) MOVED int target(int x) +{ + return callee(x) + callee(x + 1); +} diff --git a/tools/objtool/tests/generic/fixtures/checksum_skip.c b/tools/objtool/tests/generic/fixtures/checksum_skip.c new file mode 100644 index 000000000000..973bdc10295d --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/checksum_skip.c @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Symbols calculate_checksums() must not give an entry of their own. + * + * Three kinds are skipped, for two different reasons: + * + * zero-length there is nothing to hash, and an entry keyed on the + * symbol's address would collide with whatever really lives + * there. + * alias a second name for an address already checksummed. + * cold part hashed as part of its parent, which func_for_each_insn() + * walks into, so a separate entry would double-count it. + * + * An entry per address is the invariant: .discard.sym_checksum is looked up by + * the address a relocation points at, so two entries for one address make the + * lookup ambiguous and one of the two checksums unreachable. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* Zero-length: an object symbol of size 0, in a section of its own. */ +extern char empty_marker[]; +__asm__(".pushsection .data.empty_marker,\"aw\",@progbits\n" + ".globl empty_marker\n" + ".type empty_marker, @object\n" + "empty_marker:\n" + ".size empty_marker, 0\n" + ".popsection\n"); + +int real_function(int x); +int real_function(int x) +{ +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} + +/* Alias: a second name for real_function's address. */ +int alias_function(int x) __attribute__((alias("real_function"))); + +int target(int x) +{ + return real_function(x) + alias_function(x) + (int)(long)empty_marker; +} diff --git a/tools/objtool/tests/generic/fixtures/cold_function.c b/tools/objtool/tests/generic/fixtures/cold_function.c new file mode 100644 index 000000000000..f6d410983ce2 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/cold_function.c @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Function the compiler may split into a hot part and a foo.cold part. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +static void __attribute__((cold, noinline)) slow_path(int x) +{ + __asm__ volatile("" :: "r"(x)); +} + +int target(int x) +{ + if (__builtin_expect(x < 0, 0)) + slow_path(x); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/cross_module.c b/tools/objtool/tests/generic/fixtures/cross_module.c new file mode 100644 index 000000000000..c170bde0666f --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/cross_module.c @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A function which calls out to another object. MODNAME selects which object + * this one is, so a test can make the caller a module and the callee's owner + * something else. + */ + +#ifndef MODNAME +#define MODNAME "vmlinux" +#endif + +/* klp diff takes the object's module name from .modinfo */ +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME; + +extern int other_mod_func(int x); + +int target(int x) +{ +#ifdef PATCHED + return other_mod_func(x) + 2; +#else + return other_mod_func(x) + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/data_alignment.c b/tools/objtool/tests/generic/fixtures/data_alignment.c new file mode 100644 index 000000000000..900253dfb2cb --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/data_alignment.c @@ -0,0 +1,29 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Data with an alignment stricter than its size. + * + * A cloned data section has to keep its sh_addralign. The kernel has plenty + * of data whose alignment is a correctness property rather than an + * optimisation -- per-CPU variables, anything touched by an aligned SSE move, + * cacheline-aligned locks -- and a clone that lands under-aligned faults or + * silently shares a cacheline it was written to avoid. + * + * The object is new in the patched build, so klp diff has to clone it rather + * than reference the kernel's copy. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +#ifdef PATCHED +int aligned_data[2] __attribute__((aligned(64))) = { 1, 2 }; +#endif + +int target(int x) +{ +#ifdef PATCHED + return x + aligned_data[0]; +#else + return x; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/function_removal.c b/tools/objtool/tests/generic/fixtures/function_removal.c new file mode 100644 index 000000000000..d65ff604c2c5 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/function_removal.c @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A function the patch deletes, along with its only caller's use of it. The + * original has a symbol which the patched object simply does not, so there is + * nothing to correlate it against. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +#ifndef PATCHED +int going_away(int x) +{ + return x + 7; +} +#endif + +int caller(int x) +{ +#ifdef PATCHED + return x + 1; +#else + return going_away(x); +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/init_reference.c b/tools/objtool/tests/generic/fixtures/init_reference.c new file mode 100644 index 000000000000..9e59bdf708d3 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/init_reference.c @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Patched function referencing data in an .init section. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* volatile so the read cannot be folded into a constant, leaving no reference */ +static volatile int init_only __attribute__((section(".init.data"), used)) = 5; + +int target(int x) +{ +#ifdef PATCHED + return x + init_only + 1; +#else + return x + init_only; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/jump_label.c b/tools/objtool/tests/generic/fixtures/jump_label.c new file mode 100644 index 000000000000..ecd3c06912af --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/jump_label.c @@ -0,0 +1,76 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Static branch in a patched function. The jump table entry is written out by + * hand, mirroring JUMP_TABLE_ENTRY(), so the fixture builds without kernel + * headers. The key is an STT_OBJECT; anything else is ignored by + * validate_special_section_klp_reloc(). + * + * MODNAME selects whether the key is taken to belong to vmlinux or a module. + * + * NEW_KEY puts the whole static branch behind PATCHED, so the patch introduces + * one where the original had none -- a different question from patching code + * that already has a key, because the __jump_table entry itself is new. + * + * KEY_NAME renames the key. Two names are special to + * validate_special_section_klp_reloc(): a __tracepoint_* key and the + * __UNIQUE_ID_ddebug_* one pr_debug() generates are both unsupported in a + * module, but are disabled with a warning rather than rejected, because the + * kernel is full of them and refusing outright would make ordinary functions + * unpatchable. + * + * STATIC_KEY makes the key file-local. That changes the shape of the + * relocation rather than the meaning of the code: a reference to a static lands + * on the section symbol plus an addend, so the key has to be resolved from the + * section before it can be recognised as a key at all. + */ + +#ifndef MODNAME +#define MODNAME "vmlinux" +#endif + +#ifndef KEY_NAME +#define KEY_NAME klp_test_key +#endif + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME; + +#ifdef STATIC_KEY +static long KEY_NAME; +#else +long KEY_NAME; +#endif + +int target(int x) +{ + int r = x; + +#if defined(NEW_KEY) && !defined(PATCHED) + /* The original has no static branch at all. */ + return r + 1; +#else + asm goto( + "1: nop\n\t" + ".pushsection __jump_table, \"aw\"\n\t" + ".balign 8\n\t" + "912:\n\t" + ".pushsection .discard.annotate_data, \"M\", @progbits, 8\n\t" + ".long 912b - ., 1\n\t" + ".popsection\n\t" + ".long 1b - ., %l[l_yes] - .\n\t" + ".quad %c0 - .\n\t" + ".popsection\n\t" + : : "i" (&KEY_NAME) : : l_yes); + + r += 1; + goto out; +l_yes: + r += 2; +out: +#endif +#ifdef PATCHED + return r + 100; +#else + return r; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/klp_funcs.c b/tools/objtool/tests/generic/fixtures/klp_funcs.c new file mode 100644 index 000000000000..3f0d3e206cef --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/klp_funcs.c @@ -0,0 +1,31 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Two changed functions and one untouched, so the patch's function list has a + * length worth checking and something that must not appear in it. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int first(int x) +{ +#ifdef PATCHED + return x + 11; +#else + return x + 1; +#endif +} + +int second(int x) +{ +#ifdef PATCHED + return x + 22; +#else + return x + 2; +#endif +} + +int third(int x) +{ + return x + 3; +} diff --git a/tools/objtool/tests/generic/fixtures/local_to_global.c b/tools/objtool/tests/generic/fixtures/local_to_global.c new file mode 100644 index 000000000000..3c9eb9200ce5 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/local_to_global.c @@ -0,0 +1,34 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A function which the patch changes from static to non-static, and a variable + * that goes the other way. The names are unchanged; only the binding moves. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* noinline, or the static one is folded into its caller and has no symbol */ +#ifdef PATCHED +__attribute__((noinline)) int flipped_up(int x) /* was static */ +#else +__attribute__((noinline)) static int flipped_up(int x) +#endif +{ + return x + 1; +} + +#ifdef PATCHED +static volatile int flipped_down = 5; /* was global */ +#else +volatile int flipped_down = 5; +#endif + +int caller(int x) +{ + flipped_down += x; +#ifdef PATCHED + return flipped_up(x) + flipped_down + 2; +#else + return flipped_up(x) + flipped_down + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/new_data.c b/tools/objtool/tests/generic/fixtures/new_data.c new file mode 100644 index 000000000000..ac364622f12a --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/new_data.c @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Data introduced by the patch. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +#ifdef PATCHED +/* + * Not an arithmetic progression: { 1, 2, 3, 4 } indexed by x & 3 is something + * a compiler can compute instead of load, and then target() has no reference + * to the array and there is nothing for klp diff to carry. + */ +static const int klp_new_data[4] __attribute__((used)) = { 7, 3, 11, 5 }; +#endif + +int target(int x) +{ +#ifdef PATCHED + return x + klp_new_data[x & 3]; +#else + return x; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/new_export_ref.c b/tools/objtool/tests/generic/fixtures/new_export_ref.c new file mode 100644 index 000000000000..73210aacb4ae --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/new_export_ref.c @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A reference which only exists in the patched build. The symbol has no twin + * in the original object, so what klp diff may do with it depends entirely on + * whether Module.symvers says it is exported, and by what. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +extern int newly_referenced(int x); + +/* + * A reference both builds have. When Module.symvers says a module exports + * this one, the original already depends on that module, which is what makes + * a new reference to it safe -- the loader will not let the patched module + * load without it. EXISTING_DEP leaves it out, for the case where there is + * no such dependency to inherit. + */ +extern int existing_dep(int x); + +int target(int x) +{ +#ifdef EXISTING_DEP + int base = existing_dep(x); +#else + int base = x; +#endif + +#ifdef PATCHED + return newly_referenced(base); +#else + return base + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/new_function.c b/tools/objtool/tests/generic/fixtures/new_function.c new file mode 100644 index 000000000000..e7886eaaff3e --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/new_function.c @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Function introduced by the patch. noinline keeps it from being folded. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +#ifdef PATCHED +static __attribute__((noinline)) int klp_new_helper(int x) +{ + return x * 7; +} +#endif + +int target(int x) +{ +#ifdef PATCHED + return klp_new_helper(x); +#else + return x; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/no_modinfo.c b/tools/objtool/tests/generic/fixtures/no_modinfo.c new file mode 100644 index 000000000000..e46374702b1a --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/no_modinfo.c @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Deliberately has no .modinfo section. */ + +int target(int x) +{ +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/special_section.c b/tools/objtool/tests/generic/fixtures/special_section.c new file mode 100644 index 000000000000..d28c5541e337 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/special_section.c @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Special section entry with no ANNOTATE_DATA_SPECIAL annotation and a local + * label at offset 0, the shape Clang produces for .kcfi_traps. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int target(int x) +{ + asm volatile( + "1:\n\t" + ".pushsection .kcfi_traps, \"a\"\n\t" + ".balign 4\n\t" + "trap_marker:\n\t" + ".long 1b - .\n\t" + ".popsection\n\t"); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/special_section_shared.c b/tools/objtool/tests/generic/fixtures/special_section_shared.c new file mode 100644 index 000000000000..f54e24f862c6 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/special_section_shared.c @@ -0,0 +1,31 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Two functions contribute to one special section; only one is patched. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int other(int x) +{ + asm volatile( + "2:\n\t" + ".pushsection .kcfi_traps, \"a\"\n\t" + ".balign 4\n\t" + ".long 2b - .\n\t" + ".popsection\n\t"); + return x * 5; +} + +int target(int x) +{ + asm volatile( + "1:\n\t" + ".pushsection .kcfi_traps, \"a\"\n\t" + ".balign 4\n\t" + ".long 1b - .\n\t" + ".popsection\n\t"); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/static_call.c b/tools/objtool/tests/generic/fixtures/static_call.c new file mode 100644 index 000000000000..4a0c4c25321e --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/static_call.c @@ -0,0 +1,59 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Static call site in a patched function, laid out by hand as for + * jump_label.c. MODNAME selects whether the key belongs to vmlinux or a + * module. + * + * objtool's check pass would emit the site, and klp-write-tests.txt says to + * let it. Not here: it does not emit the ANNOTATE_DATA_SPECIAL describing + * the entry boundaries -- in the kernel that comes from the static_call + * macros -- and NO_ANNOTATE below has to be able to take it away. A fixture + * which varies the annotation has to write the entry that goes with it. + * + * NO_ANNOTATE drops the ANNOTATE_DATA_SPECIAL block from the patched build, + * leaving .static_call_sites with no annotation to describe its entry + * boundaries. The section carries no entsize either, so klp diff has to fall + * back on the annotations it can still see -- and when the patched object is + * the only one that lost them, the two sides disagree about how the section is + * divided up. + * + * NEW_CALL puts the call site behind PATCHED, so the patch introduces one + * where the original had none. The .static_call_sites entry is then new, with + * nothing in the original to correlate it against. + */ + +#ifndef MODNAME +#define MODNAME "vmlinux" +#endif + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME; + +long __SCK__klp_test_call; + +int target(int x) +{ +#if defined(NEW_CALL) && !defined(PATCHED) + /* The original has no static call at all. */ + return x + 1; +#else + __asm__ volatile( + "1: nop\n\t" + ".pushsection .static_call_sites, \"aw\"\n\t" + ".balign 8\n\t" + "912:\n\t" +#if !(defined(PATCHED) && defined(NO_ANNOTATE)) + ".pushsection .discard.annotate_data, \"M\", @progbits, 8\n\t" + ".long 912b - ., 1\n\t" + ".popsection\n\t" +#endif + ".long 1b - ., %c0 - .\n\t" + ".popsection\n\t" + :: "i" (&__SCK__klp_test_call)); +#endif +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/static_local.c b/tools/objtool/tests/generic/fixtures/static_local.c new file mode 100644 index 000000000000..f2f025d00a39 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/static_local.c @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Static local in a patched function. */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int target(int x) +{ + static int counter; + + counter += 1; +#ifdef PATCHED + return x + counter + 1; +#else + return x + counter; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c b/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c new file mode 100644 index 000000000000..cb4cdd7a496e --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c @@ -0,0 +1,41 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Static locals of three kinds, in one patched function. + * + * Most static locals must be correlated, so the patched code keeps using the + * running kernel's copy. Two kinds must not: + * + * - anything in .data..once, the flag behind WARN_ONCE and friends. Sharing + * it would mean a patch inherits "already warned" from before the patch. + * - the well-known names the kernel generates for such things (__warned, + * __key, __func__, ...), which are per-instance by nature. gcc names them + * <var>.<id> and Clang <func>.<var>, so both spellings have to be caught. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int target(int x) +{ + /* + * A .data..once variable whose name is *not* on the list below, so + * only the section can disqualify it. Naming it __warned would let + * the name rule catch it and the section rule go untested. + */ + static int once_flag __attribute__((section(".data..once"))); + /* a never-correlate name, in an ordinary section */ + static int __key; + /* and one that must be correlated */ + static int ordinary; + + if (!once_flag) + once_flag = 1; + __key += x; + ordinary += x; + +#ifdef PATCHED + return __key + ordinary + once_flag + 2; +#else + return __key + ordinary + once_flag + 1; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/switch_rodata.c b/tools/objtool/tests/generic/fixtures/switch_rodata.c new file mode 100644 index 000000000000..817ddac92d81 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/switch_rodata.c @@ -0,0 +1,31 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A switch dense enough that Clang builds a jump table for it, in a section of + * its own: .rodata..Lswitch.table.<function>. + * + * The table belongs to the function and has to travel with it. It is named + * after the function but is not part of it, so klp diff has to associate the + * two rather than treating the table as unrelated data. + * + * The patch adds a case, which changes the table's contents and length. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; +const char *status_to_string(unsigned int c) +{ + switch (c) { + case 0: return "idle"; + case 1: return "running"; + case 2: return "stopped"; + case 3: return "error"; + case 4: return "paused"; + case 5: return "waiting"; + case 6: return "starting"; + case 7: return "stopping"; +#ifdef PATCHED + case 8: return "completed"; +#endif + } + return "unknown"; +} diff --git a/tools/objtool/tests/generic/fixtures/symid_discarded.c b/tools/objtool/tests/generic/fixtures/symid_discarded.c new file mode 100644 index 000000000000..573cc2d4474b --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/symid_discarded.c @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Compiled twice and partially linked so the result has duplicate locals, + * which is what symid_needed() requires. dup_normal is in a live section, + * dup_discarded in one the vmlinux link throws away. DISCARDED_SEC selects + * which discarded section, since there is more than one and each was its own + * bug. + */ + +#ifndef DISCARDED_SEC +#define DISCARDED_SEC ".exitcall.exit" +#endif + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +static int dup_normal = 1; + +static void *dup_discarded + __attribute__((section(DISCARDED_SEC), used)) = &dup_normal; + +int FUNC_NAME(void) +{ + return dup_normal + (dup_discarded != (void *)0); +} diff --git a/tools/objtool/tests/generic/fixtures/sympos_dup.c b/tools/objtool/tests/generic/fixtures/sympos_dup.c new file mode 100644 index 000000000000..7eded9b12cfc --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/sympos_dup.c @@ -0,0 +1,32 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A static whose name recurs in every translation unit that includes it. + * Compiled once for a single-copy object and twice, partially linked, for one + * with duplicates -- which is the only case where sympos is non-zero. + * + * FUNC_NAME keeps the referencing functions distinct so both get patched. + * Only the first copy carries .modinfo; two would be a second thing to + * disambiguate and is not what this fixture is about. + */ + +#ifndef FUNC_NAME +#define FUNC_NAME use_a +#endif + +#ifndef NO_MODINFO +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; +#endif + +/* volatile so it survives as an STT_OBJECT rather than being folded away */ +static volatile int dup_counter = 1; + +int FUNC_NAME(int x) +{ + dup_counter += x; +#ifdef PATCHED + return dup_counter + 1; +#else + return dup_counter; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c b/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c new file mode 100644 index 000000000000..d5e70994c582 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Two translation units with a same-named static, placed so that the linker + * puts them in the opposite order to the one they appear in the symbol table. + * + * VARSEC selects the section the static lands in. Linking with + * --sort-section=name then orders them alphabetically rather than by object + * order, so the first symbol in the symbol table ends up at the *higher* + * address. That is the whole point: counting symbol table order and reading + * the linked image's addresses now give different answers, which is what makes + * it possible to tell which one klp diff used. + * + * Only use_a is patched, so exactly one sympos is emitted and there is nothing + * to attribute. + */ + +#ifndef FUNC_NAME +#define FUNC_NAME use_a +#endif +#ifndef VARSEC +#define VARSEC ".data.mmm" +#endif + +#ifndef NO_MODINFO +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; +#endif + +/* volatile so it survives as an STT_OBJECT rather than being folded away */ +static volatile int dup_counter __attribute__((section(VARSEC))) = 1; + +int FUNC_NAME(int x) +{ + dup_counter += x; +#ifdef PATCHED + return dup_counter + 1; +#else + return dup_counter; +#endif +} diff --git a/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c b/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c new file mode 100644 index 000000000000..b88830f41a92 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c @@ -0,0 +1,57 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Three translation units linked with ThinLTO, two of which have a file-local + * helper of the same name. + * + * TU_C calls into both of the others, so ThinLTO imports entry_a and entry_b + * and with them the static helper each one calls. A file-local symbol which + * has to become visible is renamed helper.llvm.<hash>, and the hash is content + * derived -- so the two helpers get different hashes from each other, and + * TU_A's gets a different one again after the patch changes it. Only TU_A's + * changes: were both bodies to change, both would be cloned whichever way + * they were paired, and the pairing would not be observable. + * + * That leaves klp diff with two symbols in the original and two in the patched + * object, all four named differently, which have to be paired up correctly. + * Demangling alone gives "helper" for all of them; something else has to + * decide which is which. + * + * Only TU_A's helper changes. That is what makes a wrong pairing observable: + * paired correctly, one helper is changed and the other is not, so exactly one + * is cloned. Paired the wrong way round, both look changed -- or the wrong + * one does. If both bodies changed the outcome would be the same either way + * and the test would prove nothing. + * + * BASE differs between the two so their bodies are not identical to begin + * with. + */ + +#if defined(TU_C) +extern int entry_a(int x); +extern int entry_b(int x); +int glue(int x) { return entry_a(x) + entry_b(x + 1); } +#else +#ifdef TU_B +#define ENTRY entry_b +#define BASE 5 +#else +#define ENTRY entry_a +#define BASE 10 +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; +#endif +static __attribute__((noinline)) int helper(int x, int len) +{ + int sum = 0, i; + + for (i = 0; i < len; i++) +#if defined(PATCHED) && !defined(TU_B) + sum += i * 2 + BASE; /* only TU_A's helper changes */ +#else + sum += i + BASE; +#endif + return sum + x; +} + +int ENTRY(int x) { return helper(x, 4); } +#endif diff --git a/tools/objtool/tests/generic/fixtures/thinlto_local.c b/tools/objtool/tests/generic/fixtures/thinlto_local.c new file mode 100644 index 000000000000..fe9f9e6bf44f --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/thinlto_local.c @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Two translation units (TU_B selects the second) linked with ThinLTO. + * Importing bump() promotes the file-local counter, renaming it + * counter.llvm.<hash>. The hash is content derived, so it differs between the + * original and patched builds. + */ + +#ifdef TU_B + +extern int bump(void); + +int other_entry(void) +{ + return bump() + bump(); +} + +#else + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +static int counter; + +int bump(void) +{ + return ++counter; +} + +int target(void) +{ +#ifdef PATCHED + return counter + 1; +#else + return counter; +#endif +} + +#endif diff --git a/tools/objtool/tests/generic/fixtures/ubsan_noise.c b/tools/objtool/tests/generic/fixtures/ubsan_noise.c new file mode 100644 index 000000000000..bf5999254163 --- /dev/null +++ b/tools/objtool/tests/generic/fixtures/ubsan_noise.c @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A translation unit built with UBSAN, where only one of two functions is + * patched. + * + * Every instrumented operation gets a per-callsite metadata object in an + * anonymous data section -- .data..Lubsan_data and .data..Lubsan_type from + * GCC, .data..L__unnamed_ from Clang -- and a call to a __ubsan_handle_* + * routine. The names are compiler-generated and carry no meaning across a + * rebuild, so klp diff has to treat those sections as uncorrelated rather than + * pairing them up by name. + * + * untouched() is byte-identical in both builds and exists to catch the false + * positive: if the metadata were correlated by name, its shifts would look + * changed and it would be dragged into the patch. + * + * The shifts are what draw the instrumentation. A bounds check would do as + * well but neither compiler emits one for an index it can prove in range. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int shift_by(int v, int n); + +int untouched(int v, int n) +{ + int s = 0; + + s += v << (n & 31); + s += v << ((n + 1) & 31); + s += shift_by(v, n); + + return s; +} + +int touched(int v, int n) +{ + int s = 0; + + s += v << (n & 31); +#ifdef PATCHED + s += v << ((n + 3) & 31); +#else + s += v << ((n + 2) & 31); +#endif + + return s; +} diff --git a/tools/objtool/tests/generic/test-abs-and-addressable.sh b/tools/objtool/tests/generic/test-abs-and-addressable.sh new file mode 100755 index 000000000000..6adb23ed4b88 --- /dev/null +++ b/tools/objtool/tests/generic/test-abs-and-addressable.sh @@ -0,0 +1,50 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# An absolute symbol and an __ADDRESSABLE() pointer must not disturb the +# function being patched. +# +# A SHN_ABS symbol has no section, so any walk of sym->sec which does not check +# dereferences NULL -- and the kernel has plenty, from linker scripts and from +# .set in assembly. __ADDRESSABLE() emits a pointer into .discard.addressable +# to keep a symbol referenced; it means nothing to a livepatch and is discarded +# at link time, but it is a relocation like any other and gets looked at. +# +# Neither is what the patch changes. The failure this guards against is not a +# wrong answer but a crash or an error on input the kernel produces routinely, +# which would make any function near one unpatchable. +# +# Not isolated to a single guard: the absolute symbol here has zero length, so +# it is excluded before the section check is reached and removing that check +# alone changes nothing observable. This stands as a check on the behaviour +# rather than on the line which produces it. +# +# Covers the same ground as corpus/x86_64/checksum-abs-sym-skip and +# addressable-symbols in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair abs_and_addressable.c + +# The premise: the fixture really did produce both. +in_symbols orig.o | grep -q 'ABS.*abs_sym' || + probe_skip "assembler did not make abs_sym absolute here" +assert_input_section .discard.addressable + +# Checksumming has to survive them, and still see the function that changed. +run_checksum +assert_checksum_differs target +assert_checksum_matches helper + +# So does the diff. +run_diff +assert_patched target +assert_not_patched helper + +# An absolute symbol has no address to record a checksum against, so it gets +# no entry -- the reference to it is what mattered, not the symbol itself. +in_relocs orig.o | awk '/rela\.discard\.sym_checksum/,/^$/' | grep -qw abs_sym && + fail "absolute symbol got a checksum entry" + +pass "absolute and __ADDRESSABLE symbols do not disturb the patched function" diff --git a/tools/objtool/tests/generic/test-basic.sh b/tools/objtool/tests/generic/test-basic.sh new file mode 100755 index 000000000000..562edfbc8646 --- /dev/null +++ b/tools/objtool/tests/generic/test-basic.sh @@ -0,0 +1,17 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Only functions whose code changed get cloned into the patch. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair basic.c +run_diff + +assert_patched changed +assert_not_patched untouched +assert_section ".init.klp_funcs" +assert_section ".init.klp_objects" + +pass "changed function cloned, unchanged function left alone" diff --git a/tools/objtool/tests/generic/test-changed-data.sh b/tools/objtool/tests/generic/test-changed-data.sh new file mode 100755 index 000000000000..c5c7381bb409 --- /dev/null +++ b/tools/objtool/tests/generic/test-changed-data.sh @@ -0,0 +1,18 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Livepatching replaces functions, not data. A changed data symbol must be +# rejected. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair changed_data.c +run_diff 255 + +diff_log | grep -q 'changed data: klp_test_data' || + fail "expected rejection, got: $(diff_log | tail -1)" +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "changed data symbol rejected" diff --git a/tools/objtool/tests/generic/test-checksum-data.sh b/tools/objtool/tests/generic/test-checksum-data.sh new file mode 100755 index 000000000000..e915026b79a7 --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-data.sh @@ -0,0 +1,61 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# What a data object's checksum has to cover. +# +# checksum_update_object() hashes the symbol's length, its bytes (when the +# section has any -- .bss does not), and then +# every relocation it carries -- as the target's name plus the adjusted addend, +# except for a reference into a string section, which contributes the string's +# contents instead. +# +# Each of those is load-bearing, and the failure is always the same shape: a +# checksum that ignores one of them calls a changed object unchanged, klp diff +# leaves it out of the patch, and the patched code goes on reading the +# kernel's old copy. Nothing says so at build time. +# +# The string case is the one that cannot be caught by hashing bytes alone. The +# pointer is identical -- same section, same offset -- and only the text it +# refers to moved. +# +# Covers the same ground as corpus/x86_64/checksum-data-basic, +# checksum-data-func-ptr, checksum-data-string-ptr and checksum-string-reloc in +# Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup + +# check <flag> <symbol> <what changed> +# +# Build the pair with one difference and require that symbol's checksum to move. +check() +{ + build_pair checksum_data.c "-D$1" + run_checksum + + assert_checksum_differs "$2" +} + +# The object's own bytes. +check PLAIN_VALUE plain +# Its length, for a .bss object whose bytes are not hashed at all. +check LONGER sized +# A relocation's target: same bytes in the object, different symbol named. +check WHICH_FUNC descriptor +check WHICH_STR descriptor +# The contents of a string the object points at, with the pointer untouched. +check STR_CONTENT descriptor +# A relocation's addend: same target symbol, different offset into it. +check WHICH_SLOT descriptor +# The same, for a static reached through its section symbol: the reference has +# to be resolved back to the object before there is a name or offset to hash. +check WHICH_PRIV descriptor + +# Having shown five things that must change it, show one that must not: an +# unrelated edit elsewhere in the file leaves this object alone. +build_pair checksum_data.c -DPLAIN_VALUE +run_checksum +assert_checksum_matches descriptor + +pass "data checksums cover length, bytes, reloc targets and string contents" diff --git a/tools/objtool/tests/generic/test-checksum-debug.sh b/tools/objtool/tests/generic/test-checksum-debug.sh new file mode 100755 index 000000000000..78856b191636 --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-debug.sh @@ -0,0 +1,49 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# "klp checksum --debug-checksum" prints a per-instruction checksum stream, +# and klp-build -f (--show-first-changed) parses it to report where a function +# first differs between the original and patched builds. +# +# It is a debugging aid, so nothing fails when it breaks: klp-build greps the +# stream, and an unmatched grep just yields no output, which reads as "no +# instruction changed". That is exactly how the format drifted out from under +# it once already. Pin the shape klp-build depends on: +# +# DEBUG: <object>: checksum: <func>(): <sym>+0x<offset> <16 hex digits> +# +# and that --dry-run leaves the object alone, since klp-build runs this against +# objects it is going to checksum again for real. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair basic.c + +before="$(md5sum < "$workdir/orig.o")" + +"$OBJTOOL" klp checksum --dry-run --debug-checksum=changed \ + "$workdir/orig.o" > "$workdir/debug.log" 2>&1 || + fail "klp checksum --debug-checksum failed" + +# --dry-run has to mean it: klp-build checksums these objects again afterwards, +# and "already has .discard.sym_checksum, skipping" would lose the real run. +[ "$(md5sum < "$workdir/orig.o")" = "$before" ] || + fail "--dry-run modified the object" +has_input_section orig.o .discard.sym_checksum && + fail "--dry-run created .discard.sym_checksum" + +grep -qE '^DEBUG: .*: checksum: changed\(\): [^ ]+\+0x[0-9a-f]+ [0-9a-f]{16}$' \ + "$workdir/debug.log" || + fail "unexpected --debug-checksum format: $(head -1 "$workdir/debug.log")" + +# This is the pattern klp-build greps with. Keep it working verbatim. +grep -qE "^DEBUG: .*checksum: changed\(\): " "$workdir/debug.log" || + fail "klp-build's --show-first-changed pattern no longer matches" + +# Only the requested function, or klp-build attributes instructions to the +# wrong one. +grep -qE 'checksum: untouched\(\)' "$workdir/debug.log" && + fail "--debug-checksum=changed also dumped untouched()" + +pass "--debug-checksum format is the one klp-build -f parses" diff --git a/tools/objtool/tests/generic/test-checksum-insn.sh b/tools/objtool/tests/generic/test-checksum-insn.sh new file mode 100755 index 000000000000..e1c04a1f518a --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-insn.sh @@ -0,0 +1,49 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# What a function's checksum has to cover beyond its instruction bytes. +# +# checksum_update_insn() hashes the raw bytes, and then what any relocation on +# the instruction refers to: a string section contributes the string's +# contents, anything else the target symbol's name and the adjusted addend, +# with a reference to a static resolved back through its section symbol first. +# +# None of these show up in the bytes. A rel32 operand is zero in the object +# and supplied by the relocation, so every change below leaves the encoded +# instruction byte-identical. A checksum stopping at the bytes reports the +# function unchanged, klp diff omits it, and the patch silently does not +# contain the fix. +# +# test-checksum-position is the other half of this: it covers what must *not* +# change the checksum when a function merely moves. +# +# Covers the same ground as corpus/x86_64/checksum-reloc-sym, +# checksum-pc-relative-addend, checksum-string-reloc and +# checksum-sec-sym-resolve in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup + +# check <flag> <what it changes> +check() +{ + build_pair checksum_insn.c "-D$1" + run_checksum + + # The premise for all of them: the operand is a relocation, not bytes. + assert_checksum_differs target +} + +check WHICH_CALL # relocation target name +check STR_CONTENT # contents of a string the code passes +check WHICH_SLOT # addend, same target symbol +check WHICH_PRIV # addend via a static's section symbol + +# The converse: rebuilding identical source leaves it alone, so the above is +# not just "any rebuild moves the checksum". +build_pair checksum_insn.c +run_checksum +assert_checksum_matches target + +pass "instruction checksums cover reloc targets, addends and string contents" diff --git a/tools/objtool/tests/generic/test-checksum-position.sh b/tools/objtool/tests/generic/test-checksum-position.sh new file mode 100755 index 000000000000..5ae759e719eb --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-position.sh @@ -0,0 +1,51 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A function's checksum must not depend on where the function sits. +# +# A jump or call without a relocation encodes its target as an offset from the +# instruction. Hashing those bytes makes the checksum change whenever anything +# ahead of the function changes size -- so an unrelated edit elsewhere in the +# file reports this function as changed too, and the patch grows to include it +# and everything it references. Nothing fails; the livepatch is just larger and +# riskier than the patch it came from. +# +# Here the "patch" moves target() away from callee() and changes nothing else: +# both are aligned to 64 in the patched build, which shifts them apart without +# touching a byte of either. See the fixture for why it is done that way. + +. "$(dirname "$0")/../lib.sh" + +setup + +# -fno-function-sections, or each function is at offset 0 of its own section +# and target() never moves. +build_pair checksum_position.c -fno-function-sections + +assert_input_symbol target + +# The fixture is only meaningful if the displacement target's calls encode +# actually changed, and that is the distance to callee() -- not target's own +# offset. A compiler which shifted the two by the same amount would move +# target and leave the distance alone, and then the bytes are identical and +# the checksum matches for the uninteresting reason. Ask about the distance. +sym_off() # $1 object, $2 symbol +{ + in_symbols "$1" | awk -v n="$2" '$NF == n { print $2; exit }' +} + +orig_t="$(sym_off orig.o target)"; orig_c="$(sym_off orig.o callee)" +new_t="$(sym_off patched.o target)"; new_c="$(sym_off patched.o callee)" + +[ -n "$orig_t" ] && [ -n "$orig_c" ] && [ -n "$new_t" ] && [ -n "$new_c" ] || + fail "target or callee missing from one of the objects" + +orig_gap=$(( 16#$orig_t - 16#$orig_c )) +new_gap=$(( 16#$new_t - 16#$new_c )) +[ "$orig_gap" != "$new_gap" ] || + probe_skip "this compiler kept target() and callee() the same distance" \ + "apart; the call displacement did not change" + +assert_checksum_matches target + +pass "checksum unchanged when the function only moves" diff --git a/tools/objtool/tests/generic/test-checksum-skip.sh b/tools/objtool/tests/generic/test-checksum-skip.sh new file mode 100755 index 000000000000..f245535a0f36 --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-skip.sh @@ -0,0 +1,81 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Symbols which must not get a checksum entry of their own. +# +# .discard.sym_checksum is an array of { address, checksum } looked up by the +# address a relocation points at, so the invariant is one entry per address. +# calculate_checksums() skips three kinds of symbol to keep it: +# +# zero-length nothing to hash, and its address belongs to whatever really +# lives there +# alias a second name for an address already covered +# cold part hashed into its parent, which func_for_each_insn() walks +# into, so its own entry would double-count +# +# A duplicate entry is not a build failure. It makes the lookup ambiguous, and +# whichever checksum loses is simply never consulted again -- so a function +# whose code changed can be read as unchanged and dropped from the patch. +# +# Of the three, only the alias skip is isolated here: removing it makes this +# test fail. A zero-length symbol is excluded by more than one of the guards +# at once -- its section has no data either -- so no single change makes that +# assertion fail, and it stands as a check on the behaviour rather than on the +# line which produces it. Nothing here reaches the cold-part skip, which +# wants a compiler that splits functions; test-cold-function covers that +# symbol surviving into the patch, not its checksum. +# +# Covers the same ground as corpus/x86_64/checksum-zero-len-sym, +# checksum-alias-skip and checksum-cold-skip in Joe Lawrence's klp-build unit +# test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair checksum_skip.c + +assert_input_symbol empty_marker +assert_input_symbol alias_function +run_checksum + +# entries_for <object> [-t] +# +# The symbol names .discard.sym_checksum has an entry for, one per line. With +# -t, what each entry points at instead: the name and its addend, which is the +# address the kernel looks the entry up by. A name alone is not that address, +# since a relocation against a section symbol names the section and puts the +# offset in the addend, and several entries can then share a name honestly. +entries_for() +{ + in_relocs "$1" | awk -v target="${2:-}" '/rela\.discard\.sym_checksum/,/^$/ { + if ($1 !~ /^[0-9a-f]{8,}/) + next + if (target == "-t") + print $5, $6, $7 + else + print $5 + }' +} + +entries="$(entries_for orig.o)" + +# The control: something real did get an entry, so an empty listing cannot +# make the rest of this pass by default. +echo "$entries" | grep -qx target || + fail "no checksum entry for target" + +echo "$entries" | grep -qx empty_marker && + fail "zero-length symbol got a checksum entry" + +# One of the two names for that address is kept and the other skipped; which +# one falls out of symbol table order and is not the point. Two would be. +n="$(echo "$entries" | grep -cxE 'real_function|alias_function')" +[ "$n" = 1 ] || + fail "expected 1 checksum entry across real_function and its alias, found $n" + +# One entry per address, which is what the skipping is for. +dupes="$(entries_for orig.o -t | sort | uniq -d)" +[ -z "$dupes" ] || + fail "two checksum entries for one address: $dupes" + +pass "zero-length symbols and aliases get no checksum entry of their own" diff --git a/tools/objtool/tests/generic/test-checksum-value.sh b/tools/objtool/tests/generic/test-checksum-value.sh new file mode 100755 index 000000000000..feae6a12e98d --- /dev/null +++ b/tools/objtool/tests/generic/test-checksum-value.sh @@ -0,0 +1,37 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# The per-function checksums klp checksum records are what klp diff uses to +# decide which functions changed. A checksum covering too little misses a real +# change and the patch silently omits the function; one covering too much, or +# unstable across identical input, clones functions nobody patched and drags +# their dependencies in with them. +# +# test-basic covers which functions got cloned, which is downstream of this and +# passes for either kind of wrong checksum as long as the two errors do not +# happen to cancel. This checks the checksums themselves. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair basic.c + +assert_input_symbol changed +assert_input_symbol untouched + +run_checksum + +# The edited function's checksum has to move, the untouched one's must not. +assert_checksum_differs changed +assert_checksum_matches untouched + +# And it has to be a function of the code, not of the build: checksumming the +# same input twice has to give the same answer, or every rebuild reports +# spurious changes. +first="$(checksum_of orig.o changed)" +build_pair basic.c +run_checksum +[ "$(checksum_of orig.o changed)" = "$first" ] || + fail "checksum for 'changed' differs between builds of identical source" + +pass "checksums track the changed function and are stable across rebuilds" diff --git a/tools/objtool/tests/generic/test-cold-function.sh b/tools/objtool/tests/generic/test-cold-function.sh new file mode 100755 index 000000000000..a02652fe5341 --- /dev/null +++ b/tools/objtool/tests/generic/test-cold-function.sh @@ -0,0 +1,39 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Both halves of a split function belong to the patch; carrying only the hot +# part leaves the cold path branching into unpatched code. + +. "$(dirname "$0")/../lib.sh" + +# Clang does not split functions into a cold part at all, so there is nothing +# for this test to look at there. A given gcc may or may not split, which is a +# version property rather than a compiler choice -- that stays a probe below. +gcc_only "clang does not split functions into a cold part" + +setup + +split_flag=-freorder-blocks-and-partition +cc_supports "$split_flag" || split_flag= + +build_pair cold_function.c $split_flag + +# Find what the compiler called the cold half -- target.cold, target.cold.0, +# depending on version -- and name it exactly from here on. +cold_sym="$(in_symbols orig.o | awk '$NF ~ /^target\.cold/ { print $NF; exit }')" +[ -n "$cold_sym" ] || + probe_skip "compiler did not split the function into a cold part" + +run_diff + +assert_patched target +# Match the name field exactly, and require it to be defined. Had the cold +# half been left behind, the branch to it would appear as an undefined +# .klp.sym.vmlinux.target.cold,0 -- a different name, which happens to contain +# this one. Asking about a column instead of the name would not tell them +# apart: readelf prints SHN_LIVEPATCH as "OS [0xff20]" and llvm-readelf as +# "OS[0xff20]", so the fields either side of the name shift between the two. +out_symbols | awk -v n="$cold_sym" '$NF == n && $(NF - 1) != "UND"' | grep -q . || + fail "cold half ($cold_sym) was not carried into the patch" + +pass "cold half carried into the patch with its parent" diff --git a/tools/objtool/tests/generic/test-data-alignment.sh b/tools/objtool/tests/generic/test-data-alignment.sh new file mode 100755 index 000000000000..8e389544a3b1 --- /dev/null +++ b/tools/objtool/tests/generic/test-data-alignment.sh @@ -0,0 +1,40 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A cloned data section keeps its alignment. +# +# Plenty of kernel data is aligned for correctness rather than speed: per-CPU +# variables, anything touched by an aligned vector move, structures padded to +# own a cacheline. A clone that lands under-aligned either faults on first use +# or silently shares a line it was laid out to avoid, and neither shows up +# until the patch is loaded on hardware that cares. +# +# Fixed by 2f2600decb30 ("objtool/klp: Fix alignment of cloned data +# sections"). +# +# Covers the same ground as corpus/x86_64/cloned-data-alignment in Joe +# Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair data_alignment.c + +# The premise: the compiler really did over-align it, and the object is new in +# the patch so it has to be cloned rather than referenced. +want="$(in_sections patched.o | sed 's/^ *\[[ 0-9]*\] *//' | + awk '$1 == ".data.aligned_data" { print $NF }')" +[ "$want" = 64 ] || + probe_skip "compiler gave .data.aligned_data alignment '$want', not 64" +has_input_section orig.o .data.aligned_data && + fail "fixture put aligned_data in the original; nothing to clone" + +run_diff +assert_section .data.aligned_data + +got="$(out_sections | sed 's/^ *\[[ 0-9]*\] *//' | + awk '$1 == ".data.aligned_data" { print $NF }')" +[ "$got" = "$want" ] || + fail "cloned .data.aligned_data has alignment $got, expected $want" + +pass "cloned data section keeps its alignment" diff --git a/tools/objtool/tests/generic/test-export-symbol-for-modules.sh b/tools/objtool/tests/generic/test-export-symbol-for-modules.sh new file mode 100755 index 000000000000..7e7bdde6a7ac --- /dev/null +++ b/tools/objtool/tests/generic/test-export-symbol-for-modules.sh @@ -0,0 +1,39 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# EXPORT_SYMBOL_FOR_MODULES() puts a vmlinux symbol in a "module:<names>" +# namespace, and the module loader grants access by matching the importing +# module's name against that list. A livepatch module is never on the list, so +# referencing such a symbol with a normal relocation fails modpost, and if that +# is silenced, fails to load with "Unknown symbol". It needs a klp relocation, +# the same as an unexported symbol. +# +# Ordinary namespaces are not affected: copy_import_ns() propagates the patched +# object's import tags to the patch module, so a normal relocation works. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair cross_module.c + +sym=other_mod_func + +# Plain vmlinux export: a normal relocation is what we want. +export_syms "$sym" +run_diff +assert_no_klp_sym "$sym" + +# Ordinary namespace: still a normal relocation. +export_syms +add_exports_ns vmlinux MY_NS "$sym" +run_diff +assert_no_klp_sym "$sym" + +# module: namespace: has to become a klp relocation. +export_syms +add_exports_ns vmlinux module:kvm "$sym" +run_diff +assert_klp_sym "$sym" vmlinux +assert_section __klp_relocs.vmlinux + +pass "EXPORT_SYMBOL_FOR_MODULES symbol referenced with a klp relocation" diff --git a/tools/objtool/tests/generic/test-function-removal.sh b/tools/objtool/tests/generic/test-function-removal.sh new file mode 100755 index 000000000000..df76e81e3936 --- /dev/null +++ b/tools/objtool/tests/generic/test-function-removal.sh @@ -0,0 +1,34 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A patch which deletes a function leaves a symbol in the original with no +# counterpart in the patched object. klp diff cannot correlate it, and must +# say so and carry on: livepatching cannot remove code from a running kernel, +# so what matters is that the surviving caller is patched and the deleted +# function is not dragged into the patch module. +# +# Cloning it would be worse than useless -- dead code in the patch, plus +# whatever it references, resolved against a kernel where it may not exist. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair function_removal.c + +# One-sided by construction: present in the original, gone from the patched. +has_input_symbol orig.o going_away || + fail "fixture has no going_away in the original" +has_input_symbol patched.o going_away && + fail "fixture still has going_away in the patched object" + +run_diff + +assert_diff_log 'no correlation: going_away' + +# The caller changed, so it is patched ... +assert_patched caller +# ... and the deleted function comes along in no form at all. +assert_not_patched going_away +assert_no_symbol going_away + +pass "deleted function reported as uncorrelated and left out of the patch" diff --git a/tools/objtool/tests/generic/test-init-reference.sh b/tools/objtool/tests/generic/test-init-reference.sh new file mode 100755 index 000000000000..921cf493cd79 --- /dev/null +++ b/tools/objtool/tests/generic/test-init-reference.sh @@ -0,0 +1,29 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Init code and data are freed after boot, so a klp relocation against them can +# never resolve. Such a patch must be rejected. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair init_reference.c + +# The rejection can only happen if the patched build really does reference the +# init symbol. The fixture makes init_only volatile so the read cannot be +# folded away, and this checks that it worked: without a relocation klp diff +# would succeed, and the failure below would read as a missing check rather +# than as a fixture which stopped posing the question. +# Either spelling will do. A file-local variable is reached through its +# section symbol -- .init.data -- and a global one by name; which of the two +# the compiler picks is its business, and the reference is what matters. +in_relocs "$patched_obj" | + awk '$5 == ".init.data" || $5 == "init_only"' | grep -q . || + fail "patched object has no reference into .init.data; the fixture tests nothing" + +run_diff 255 + +diff_log | grep -q "can't patch or reference init code/data" || + fail "expected rejection, got: $(diff_log | tail -1)" + +pass "reference to init data rejected" diff --git a/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh b/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh new file mode 100755 index 000000000000..fa2f913d860b --- /dev/null +++ b/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh @@ -0,0 +1,51 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Two kinds of module-owned static branch key are disabled with a warning +# instead of rejected. +# +# A module-local key is normally fatal: late module patching lets the livepatch +# load before the module it depends on, so the unresolved __jump_table entry is +# dereferenced by jump_label_add_module(). test-jump-label-module-key covers +# that rejection. +# +# Tracepoints and pr_debug() generate such keys everywhere, though, and +# refusing them outright would make any function containing a trace_*() call or +# a pr_debug() unpatchable. So klp diff drops the entry, says so, and carries +# on: the patched code keeps working with that one tracepoint or debug print +# permanently off. +# +# Both halves matter. A build that fails is a function nobody can patch; an +# entry left in place is the memory corruption the rejection exists to prevent. +# +# The two exemptions are isolated: remove either and this fails. That the +# entry is then dropped is asserted but not isolated -- making the caller keep +# it anyway produces no output difference here, so that assertion stands as a +# check on the behaviour rather than on the line which produces it. +# +# Covers the same ground as corpus/x86_64/static-call-module-tracepoint and +# pr-debug-unsupported in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup + +# check <key name> <expected warning> +check() +{ + build_pair jump_label.c -DKEY_NAME="$1" -DMODNAME='"klp_testmod"' + require_input_section __jump_table + + # Accepted, not rejected: this is the whole point. + run_diff + assert_diff_log "$2" + + # And the entry is gone, not merely complained about. + assert_patched target + assert_no_section __jump_table +} + +check __tracepoint_klp_test 'disabling unsupported tracepoint klp_test' +check __UNIQUE_ID_ddebug_klp_test 'disabling unsupported pr_debug' + +pass "tracepoint and pr_debug keys disabled with a warning, not rejected" diff --git a/tools/objtool/tests/generic/test-jump-label-key.sh b/tools/objtool/tests/generic/test-jump-label-key.sh new file mode 100755 index 000000000000..142f94bd38ec --- /dev/null +++ b/tools/objtool/tests/generic/test-jump-label-key.sh @@ -0,0 +1,44 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A cloned __jump_table entry must keep a relocation in its key slot, whether +# the key needs a klp relocation or not. + +. "$(dirname "$0")/../lib.sh" + +key=klp_test_key + +setup +build_pair jump_label.c + +has_input_section orig.o __jump_table || + probe_skip "fixture produced no __jump_table on this arch" + +key_slot_relocs() +{ + out_relocs | awk '/rela__jump_table/,/^$/' | grep -c "^0*8[[:space:]]" +} + +# Unexported: klp relocation, key slot holds a tombstone. +export_syms +run_diff + +[ "$(key_slot_relocs)" = 1 ] || + fail "unexported key: key slot has no relocation" +out_relocs | awk '/rela__jump_table/,/^$/' | grep -q "\.klp\.tombstone\.$key" || + fail "unexported key: expected a .klp.tombstone.$key relocation" +out_symbols | grep -q "\.klp\.sym\..*\.$key," || + fail "unexported key: no .klp.sym reference for the real relocation" + +# Exported: ordinary relocation, no klp machinery. +export_syms "$key" +run_diff + +[ "$(key_slot_relocs)" = 1 ] || + fail "exported key: key slot has no relocation" +out_relocs | awk '/rela__jump_table/,/^$/' | grep -q "[[:space:]]$key[[:space:]]*+" || + fail "exported key: expected a direct relocation to $key" +out_symbols | grep -q '\.klp\.tombstone\.' && + fail "exported key: tombstone emitted for an exported symbol" + +pass "key slot populated for exported and unexported vmlinux keys" diff --git a/tools/objtool/tests/generic/test-jump-label-module-key.sh b/tools/objtool/tests/generic/test-jump-label-module-key.sh new file mode 100755 index 000000000000..7c3f82cdd3c1 --- /dev/null +++ b/tools/objtool/tests/generic/test-jump-label-module-key.sh @@ -0,0 +1,22 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A static branch key owned by a module cannot be reached with a klp +# relocation and must be rejected. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair jump_label.c -DMODNAME='"klp_testmod"' + +has_input_section orig.o __jump_table || + probe_skip "fixture produced no __jump_table on this arch" + +run_diff 255 + +diff_log | grep -q 'unsupported static branch key klp_test_key' || + fail "expected rejection, got: $(diff_log | tail -1)" +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "module-owned static branch key rejected" diff --git a/tools/objtool/tests/generic/test-jump-label-module-static-key.sh b/tools/objtool/tests/generic/test-jump-label-module-static-key.sh new file mode 100755 index 000000000000..7bdcbaf2fa60 --- /dev/null +++ b/tools/objtool/tests/generic/test-jump-label-module-static-key.sh @@ -0,0 +1,45 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A static branch key owned by a module is rejected whether the key is global +# or file-local. +# +# The rejection matters because late module patching allows the livepatch +# module to load before the module it depends on: the __jump_table klp reloc is +# then unresolved, and jump_label_add_module() dereferences an uninitialized +# pointer. Catching it at build time is the only defence. +# +# test-jump-label-module-key covers the global key. A file-local one reaches +# the same check by a different route: the compiler emits the reference against +# the section symbol plus an addend, so validate_special_section_klp_reloc() +# has to resolve it to the underlying object before it can see a key at all. +# Until it did, a static key was passed over as "not STT_OBJECT" and the +# unsupported reference was emitted with nothing said. +# +# Fixed by f9fb44b0ecef ("objtool/klp: Fix detection of corrupt static +# branch/call entries"). + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair jump_label.c -DSTATIC_KEY -DMODNAME='"klp_testmod"' + +require_input_section __jump_table + +# The premise: the key is reached through its section symbol, not by name. +# Without that this is just a second copy of test-jump-label-module-key. +input_jump_relocs="$(in_relocs orig.o | awk '/rela__jump_table/,/^$/')" + +echo "$input_jump_relocs" | grep -q klp_test_key || + fail "fixture produced no __jump_table reference to the key" +echo "$input_jump_relocs" | grep -qE '\.(bss|data)\.klp_test_key' || + probe_skip "compiler referenced the static key by name, not through its section" + +run_diff 255 + +diff_log | grep -q 'unsupported static branch key klp_test_key' || + fail "expected rejection, got: $(diff_log | tail -1)" +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "module-owned file-local static branch key rejected" diff --git a/tools/objtool/tests/generic/test-jump-label-new-key.sh b/tools/objtool/tests/generic/test-jump-label-new-key.sh new file mode 100755 index 000000000000..3bc6005fc776 --- /dev/null +++ b/tools/objtool/tests/generic/test-jump-label-new-key.sh @@ -0,0 +1,51 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A patch may introduce a static branch where the original function had none. +# +# That is not the same case as patching a function which already has one. The +# __jump_table entry is itself new, so there is no counterpart in the original +# to correlate it against: klp diff has to carry the entry and the key into the +# patch from scratch, and the key has to be reached the way any other reference +# to a vmlinux symbol is. +# +# Get it wrong and the entry is dropped, leaving a static branch the kernel +# never patches -- the code takes the wrong arm forever, silently. +# +# Where the key lives still decides whether that is allowed, exactly as it does +# for a key the original already had: a module-owned one cannot be reached, so +# introducing one has to stop the build rather than emit an entry nothing will +# resolve. +# +# Covers the same ground as corpus/x86_64/static-branch-vmlinux-new and +# static-branch-module-new in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup klp_test_key +build_pair jump_label.c -DNEW_KEY + +# The premise: the original really has no jump table, and the patched one does. +has_input_section orig.o __jump_table && + fail "fixture put a __jump_table in the original; nothing new to add" +has_input_section patched.o __jump_table || + probe_skip "compiler produced no __jump_table on this arch" + +run_diff + +assert_patched target +assert_section __jump_table +assert_reloc_sym __jump_table target + +# The same new branch, with the key owned by a module. Drop the vmlinux export +# first: while it is exported the key is reachable and being new changes +# nothing, which is what the first version of this got wrong. +export_syms +rm -f "$workdir/out.o" +build_pair jump_label.c -DNEW_KEY -DMODNAME='"klp_testmod"' +run_diff 255 +assert_diff_log 'unsupported static branch key klp_test_key' +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "static branch introduced by the patch carried in, or rejected for a module key" diff --git a/tools/objtool/tests/generic/test-klp-funcs-content.sh b/tools/objtool/tests/generic/test-klp-funcs-content.sh new file mode 100755 index 000000000000..32fbd5f6a6fa --- /dev/null +++ b/tools/objtool/tests/generic/test-klp-funcs-content.sh @@ -0,0 +1,45 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# .init.klp_funcs is the list the kernel walks to decide what to patch, and +# .init.klp_objects points at it. Existing tests assert only that the sections +# exist, which they do whether the list names the right functions, the wrong +# ones, or none at all -- and a patch module with an empty function list loads +# perfectly happily and patches nothing. +# +# Each entry pairs a name string in .rodata.klp.str1.1 with a relocation to the +# new function, so both halves are checkable. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair klp_funcs.c +run_diff + +assert_section .init.klp_funcs +assert_section .init.klp_objects + +# Two functions changed, so two entries, each contributing a name relocation +# and a function relocation. +assert_reloc_count .init.klp_funcs 4 + +# The functions that changed are named ... +assert_reloc_sym .init.klp_funcs first +assert_reloc_sym .init.klp_funcs second +# ... and the one that did not is absent, from the list and from the patch. +assert_no_reloc_sym .init.klp_funcs third +assert_not_patched third + +# The names the kernel matches on are real strings, not just relocations. +# readelf prints one per line as "[ offset] <string>", so compare the whole +# name: a word-boundary match would also accept ".text.first", since a dot is +# not a word character. +for name in first second; do + out_strings .rodata.klp.str1.1 | awk -v n="$name" '$NF == n' | grep -q . || + fail "no '$name' string in .rodata.klp.str1.1" +done + +# The object list has to reach the function list, or nothing is walked. +assert_reloc_sym .init.klp_objects .init.klp_funcs + +pass "klp_funcs lists exactly the changed functions, by name and relocation" diff --git a/tools/objtool/tests/generic/test-local-to-global-flip.sh b/tools/objtool/tests/generic/test-local-to-global-flip.sh new file mode 100755 index 000000000000..45e92797190b --- /dev/null +++ b/tools/objtool/tests/generic/test-local-to-global-flip.sh @@ -0,0 +1,63 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A patch can change a symbol's linkage without renaming it: dropping "static" +# from a helper so something else can call it, or adding it to one that is no +# longer shared. Correlation keys off more than the name, so a symbol whose +# binding moved can fail to pair with itself. +# +# Failing to correlate is not a build failure. The symbol looks new, and a +# "new" data symbol is either rejected or cloned as a second copy -- at which +# point the patched code updates its own private variable and the rest of the +# kernel keeps reading the original. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair local_to_global.c + +# Confirm the fixture really moved the bindings, in both directions. +# +# Match the binding and the name as fields, not as substrings of the line. +# -ffunction-sections and -fdata-sections give these symbols sections of their +# own, and the section symbols -- .text.flipped_up, .data.flipped_down -- are +# always LOCAL, so "does a LOCAL line mention flipped_up" is answered by the +# wrong symbol. GNU readelf 2.35 happens to leave section symbol names blank, +# but llvm-readelf prints them, and a premise that holds on one readelf and +# not the other is no premise at all. +has_binding() # $1 object, $2 binding, $3 symbol +{ + in_symbols "$1" | awk -v b="$2" -v n="$3" '$5 == b && $NF == n' | grep -q . +} + +has_binding orig.o LOCAL flipped_up || + fail "flipped_up is not local in the original" +has_binding patched.o GLOBAL flipped_up || + fail "flipped_up is not global in the patched object" +has_binding orig.o GLOBAL flipped_down || + fail "flipped_down is not global in the original" +has_binding patched.o LOCAL flipped_down || + fail "flipped_down is not local in the patched object" + +run_diff + +assert_diff_log 'changed function: caller' + +# Correlated means each pairs with its own counterpart in the original, so the +# patch refers back to the kernel's copy ... +assert_klp_sym flipped_up vmlinux +assert_klp_sym flipped_down vmlinux + +# ... rather than carrying its own. A second copy of flipped_down is the bad +# outcome: patched code would update its private one while the rest of the +# kernel keeps reading the original. +assert_not_patched flipped_up +assert_no_section .data.flipped_down +assert_no_section .bss.flipped_down + +diff_log | grep -q 'no correlation' && + fail "linkage change reported as an uncorrelated symbol" +diff_log | grep -q 'changed data' && + fail "linkage change reported as changed data" + +pass "symbols correlated across a change of linkage" diff --git a/tools/objtool/tests/generic/test-local-vs-export.sh b/tools/objtool/tests/generic/test-local-vs-export.sh new file mode 100755 index 000000000000..5b91101dadd6 --- /dev/null +++ b/tools/objtool/tests/generic/test-local-vs-export.sh @@ -0,0 +1,32 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# find_export() matched on symbol name alone, so a static function or variable +# sharing a name with an export was mistaken for a reference to that export. +# For a vmlinux export that means no klp relocation at all: the normal +# relocation left behind is resolved by the module loader to the vmlinux +# symbol, and the patched code quietly reads and writes the wrong object. +# +# Exports are always global, so a local symbol is never one. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_local.c + +# The static the fixture uses. Compilers mangle statics variously -- gcc says +# counter.0, clang says target.counter -- so find what this one produced rather +# than assuming a shape. +local_sym="$(in_symbols orig.o | + awk '$4 == "OBJECT" && $5 == "LOCAL" && $8 ~ /counter/ { print $8; exit }')" +[ -n "$local_sym" ] || + fail "fixture produced no local 'counter' symbol" + +# Contrive the collision: something else exports that same name. +export_syms "$local_sym" counter +run_diff + +# Still treated as the local it is, not as the export. +assert_klp_sym "$local_sym" vmlinux + +pass "local symbol not mistaken for an export of the same name" diff --git a/tools/objtool/tests/generic/test-missing-checksum.sh b/tools/objtool/tests/generic/test-missing-checksum.sh new file mode 100755 index 000000000000..89a05b1869ed --- /dev/null +++ b/tools/objtool/tests/generic/test-missing-checksum.sh @@ -0,0 +1,18 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Without .discard.sym_checksum there is nothing to compare; concluding that +# nothing changed would be worse than failing. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair basic.c + +( cd "$workdir" && "$OBJTOOL" klp diff orig.o patched.o out.o ) \ + > "$workdir/diff.log" 2>&1 && fail "klp diff accepted an unchecksummed object" + +grep -q 'sym_checksum' "$workdir/diff.log" || + fail "expected a complaint about the checksum section, got: $(tail -1 "$workdir/diff.log")" + +pass "unchecksummed input rejected" diff --git a/tools/objtool/tests/generic/test-missing-modinfo.sh b/tools/objtool/tests/generic/test-missing-modinfo.sh new file mode 100755 index 000000000000..6f242f7eb756 --- /dev/null +++ b/tools/objtool/tests/generic/test-missing-modinfo.sh @@ -0,0 +1,16 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# The module name in .modinfo ends up in the livepatch's klp_object, so an +# object without one cannot be diffed. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair no_modinfo.c +run_diff 255 + +diff_log | grep -q 'modinfo' || + fail "expected a complaint about .modinfo, got: $(diff_log | tail -1)" + +pass "object without .modinfo rejected" diff --git a/tools/objtool/tests/generic/test-modname-normalize.sh b/tools/objtool/tests/generic/test-modname-normalize.sh new file mode 100755 index 000000000000..a8059bdbd667 --- /dev/null +++ b/tools/objtool/tests/generic/test-modname-normalize.sh @@ -0,0 +1,26 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Module.symvers records the build-tree path of the object that exports a +# symbol, not the name the module has at runtime: "arch/x86/kvm/kvm-intel", +# where the kernel knows the module as "kvm_intel". +# +# The klp symbol name embeds the owning object, and livepatch matches it +# against loaded modules by name. Left unnormalized it names a module that +# does not exist, and the relocation is never resolved -- at load time, with no +# build-time complaint. + +. "$(dirname "$0")/../lib.sh" + +setup +build_module_pair cross_module.c klp_testmod + +# A path with directory components, a dash, and no extension. +export_syms +add_exports "arch/x86/kvm/kvm-intel" other_mod_func +run_diff + +# Directories stripped, dash to underscore. +assert_klp_sym other_mod_func kvm_intel + +pass "Module.symvers paths normalized to runtime module names" diff --git a/tools/objtool/tests/generic/test-module-object.sh b/tools/objtool/tests/generic/test-module-object.sh new file mode 100755 index 000000000000..95c8402a523a --- /dev/null +++ b/tools/objtool/tests/generic/test-module-object.sh @@ -0,0 +1,31 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A klp relocation section is named for the object being patched, not for the +# object which happens to own the symbol being referenced. Deriving it from +# the symbol means a cross-module reference lands in a section for an object +# the patch may not even touch, so the relocation is never applied and the call +# goes somewhere arbitrary. + +. "$(dirname "$0")/../lib.sh" + +setup +build_module_pair cross_module.c klp_testmod + +# The fixture has to have built as a module for any of this to mean anything. +in_sections orig.o | grep -q '\.modinfo' || + fail "fixture has no .modinfo" + +# other_mod_func belongs to a different module than the one being patched. +add_exports other_mod other_mod_func +run_diff + +# Named for the patched object ... +assert_section __klp_relocs.klp_testmod +# ... not for the object owning the symbol. +assert_no_section __klp_relocs.other_mod + +run_post_link +assert_klp_rela klp_testmod .text.target + +pass "klp relocation section named for the patched object" diff --git a/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh b/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh new file mode 100755 index 000000000000..635a75f6b8ba --- /dev/null +++ b/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh @@ -0,0 +1,40 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Patching a module, where the patched code references a vmlinux symbol which +# needs a klp relocation. +# +# The kernel does not allow a module-targeted klp relocation to reference a +# vmlinux symbol, and a symbol exported with EXPORT_SYMBOL_FOR_MODULES gets a +# klp relocation. Put together, filing that relocation under the patched +# module produces a patch the kernel refuses to apply to its target. +# +# So it goes under vmlinux instead, and is applied when the patch module loads +# rather than when the patched module does. That is the opposite of the rule +# for a reference to a module's symbol, which test-module-object covers; this +# is the other branch of the same decision. + +. "$(dirname "$0")/../lib.sh" + +setup +# The object being patched is a module ... +build_module_pair cross_module.c klp_testmod + +# ... and the symbol it references belongs to vmlinux, exported in a way that +# still requires a klp relocation. +export_syms +add_exports_ns vmlinux module:kvm other_mod_func +run_diff + +# Filed against vmlinux, applied when the patch loads. +assert_section __klp_relocs.vmlinux +assert_klp_sym other_mod_func vmlinux + +# Not against the patched module: that is the relocation the kernel rejects. +assert_no_section __klp_relocs.klp_testmod + +run_post_link +assert_klp_rela vmlinux .text.target +assert_no_section ".klp.rela.klp_testmod..text.target" + +pass "klp relocation to a vmlinux symbol filed under vmlinux, not the patched module" diff --git a/tools/objtool/tests/generic/test-new-data.sh b/tools/objtool/tests/generic/test-new-data.sh new file mode 100755 index 000000000000..acb68ff19c99 --- /dev/null +++ b/tools/objtool/tests/generic/test-new-data.sh @@ -0,0 +1,27 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Data added by the patch has no counterpart in the running kernel and must be +# carried into the livepatch. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair new_data.c + +# State the premise on both sides. The array is new in the patched build and +# absent from the original; if the compiler folded it into the code instead of +# emitting it, the assertion below would fail without saying why. +has_input_symbol "$orig_obj" klp_new_data && + fail "fixture put klp_new_data in the original; nothing new to carry" +has_input_symbol "$patched_obj" klp_new_data || + fail "compiler did not emit klp_new_data; the fixture tests nothing" +in_relocs "$patched_obj" | grep -q 'klp_new_data' || + fail "target() does not reference klp_new_data; the fixture tests nothing" + +run_diff + +assert_patched target +assert_symbol klp_new_data + +pass "new data carried into the patch" diff --git a/tools/objtool/tests/generic/test-new-export-ref.sh b/tools/objtool/tests/generic/test-new-export-ref.sh new file mode 100755 index 000000000000..f0be2cf87fe3 --- /dev/null +++ b/tools/objtool/tests/generic/test-new-export-ref.sh @@ -0,0 +1,46 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A reference the patch adds has no counterpart in the original object. klp +# diff used to reject any such reference needing a klp relocation, which ruled +# out patches that call something they did not call before -- a common enough +# thing for a fix to do. +# +# Module.symvers is what makes it safe: it says the symbol exists and who owns +# it. But that is only sufficient for a vmlinux export. A new reference to a +# module's export is a dependency the patch module does not declare, and the +# relocation would resolve only if that module happened to be loaded, so it +# stays an error. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair new_export_ref.c + +# Exported by vmlinux, in a module: namespace so it needs a klp relocation +# rather than an ordinary one. Allowed. +export_syms +add_exports_ns vmlinux module:kvm newly_referenced +run_diff +assert_klp_sym newly_referenced vmlinux + +# Exported by a module the patched object does not depend on. Rejected, and +# for that reason rather than some other. +export_syms +add_exports other_mod newly_referenced +run_diff 255 +assert_diff_log 'undeclared module dependency' + +# ... unless the original already referenced something that module exports. +# The loader will not let the patched object load without other_mod, so the +# klp relocation has something to resolve against, and klp diff allows it. +# This is the other half of the rule, and it fails in the opposite direction: +# refusing here would reject a patch which is safe to apply. +rm -f "$workdir/out.o" +build_pair new_export_ref.c -DEXISTING_DEP +export_syms +add_exports other_mod newly_referenced existing_dep +run_diff +assert_klp_sym newly_referenced other_mod + +pass "new reference allowed for vmlinux and for a module already depended on" diff --git a/tools/objtool/tests/generic/test-new-function.sh b/tools/objtool/tests/generic/test-new-function.sh new file mode 100755 index 000000000000..b0b7f6443110 --- /dev/null +++ b/tools/objtool/tests/generic/test-new-function.sh @@ -0,0 +1,16 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A function added by the patch has no original to correlate against and must +# still be carried into the livepatch. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair new_function.c +run_diff + +assert_patched target +assert_symbol klp_new_helper + +pass "new function carried into the patch with its caller" diff --git a/tools/objtool/tests/generic/test-post-link.sh b/tools/objtool/tests/generic/test-post-link.sh new file mode 100755 index 000000000000..a6d40ae1b18d --- /dev/null +++ b/tools/objtool/tests/generic/test-post-link.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# klp post-link converts the intermediate __klp_relocs.* sections into the +# .klp.rela.* form the kernel applies at patch load. Getting this wrong is +# invisible at build time: the module links and loads, and the relocations are +# simply never applied. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_local.c + +# An unexported symbol is what produces a klp relocation in the first place. +# The static local is renamed by the compiler -- counter.0 with gcc -- so the +# fixture's premise is that some local of that shape exists, not that a symbol +# called "counter" does. Find it first; everything below names it. +local_sym="$(in_symbols orig.o | + awk '$4 == "OBJECT" && $5 == "LOCAL" && $8 ~ /counter/ { print $8; exit }')" +[ -n "$local_sym" ] || fail "fixture produced no local 'counter' symbol" + +run_diff +assert_section __klp_relocs.vmlinux + +# Nothing has converted them yet. +assert_no_section ".klp.rela.vmlinux..text.target" + +# The original relocation is neutralised by pointing it at a tombstone, which +# is what stops the module loader resolving it behind livepatch's back. +# +# Compilers mangle a static local differently -- gcc says counter.0, clang +# target.counter -- so find what this one produced rather than assuming. +assert_tombstone "$local_sym" + +run_post_link + +# One .klp.rela section per base section, carrying SHF_RELA_LIVEPATCH, against +# a symbol in SHN_LIVEPATCH for the kernel to resolve. +assert_klp_rela vmlinux .text.target +assert_livepatch_sym "$local_sym" + +pass "klp relocations converted to .klp.rela with SHN_LIVEPATCH symbols" diff --git a/tools/objtool/tests/generic/test-special-section-shared.sh b/tools/objtool/tests/generic/test-special-section-shared.sh new file mode 100755 index 000000000000..05c6110b6aef --- /dev/null +++ b/tools/objtool/tests/generic/test-special-section-shared.sh @@ -0,0 +1,26 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Only the patched function's special section entry may be extracted. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair special_section_shared.c + +has_input_section orig.o .kcfi_traps || + probe_skip "fixture produced no .kcfi_traps on this arch" + +run_diff + +assert_patched target +assert_not_patched other +assert_section ".kcfi_traps" + +entries="$(out_relocs | awk '/rela\.kcfi_traps/,/^$/' | grep -c 'target')" +[ "$entries" = 1 ] || fail "expected one .kcfi_traps entry, found $entries" + +out_relocs | awk '/rela\.kcfi_traps/,/^$/' | grep -q 'other' && + fail "the untouched function's entry was dragged in" + +pass "only the patched function's entry extracted" diff --git a/tools/objtool/tests/generic/test-special-section.sh b/tools/objtool/tests/generic/test-special-section.sh new file mode 100755 index 000000000000..b6a9139c0062 --- /dev/null +++ b/tools/objtool/tests/generic/test-special-section.sh @@ -0,0 +1,20 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A .kcfi_traps entry belonging to a patched function must be extracted even +# without ANNOTATE_DATA_SPECIAL and with a local label already at offset 0. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair special_section.c + +in_symbols orig.o | grep -q 'trap_marker' || + probe_skip "fixture produced no .kcfi_traps on this arch" + +run_diff + +assert_patched target +assert_section ".kcfi_traps" + +pass ".kcfi_traps extracted despite a local label at offset 0" diff --git a/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh b/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh new file mode 100755 index 000000000000..818b047ed4b6 --- /dev/null +++ b/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A patch may remove the last ANNOTATE_DATA_SPECIAL in a translation unit while +# leaving the special section it described in place. +# +# klp diff needs entry boundaries for a special section: either an entsize, or +# annotations naming where each entry starts. .static_call_sites has no +# entsize, so the annotations are all there is -- and when the patched object +# is the only side that lost them, the two sides no longer agree on how the +# section divides up. +# +# The section must still be handled. Dropping it would leave the patched +# function's static call unregistered; misreading its boundaries would attach +# the entry to the wrong code. Either way nothing is reported at build time. +# +# Fixed by commit 3de711fba73a ("objtool/klp: Fix create_fake_symbols() +# skipping entsize-based sections"). +# +# Covers the same ground as corpus/x86_64/static-call-annotate-stripped in Joe +# Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_call.c -DNO_ANNOTATE + +# The premise: the original describes its entry, the patched one no longer +# does, and both still have the section itself. +has_input_section orig.o .discard.annotate_data || + fail "fixture produced no annotation in the original" +has_input_section patched.o .discard.annotate_data && + fail "patched object still has the annotation; nothing was stripped" +assert_input_section .static_call_sites + +run_diff + +assert_patched target +assert_section .static_call_sites +assert_reloc_sym .static_call_sites target + +pass "static call site kept when the patch strips its data annotation" diff --git a/tools/objtool/tests/generic/test-static-call-module-key.sh b/tools/objtool/tests/generic/test-static-call-module-key.sh new file mode 100755 index 000000000000..260c4af72e05 --- /dev/null +++ b/tools/objtool/tests/generic/test-static-call-module-key.sh @@ -0,0 +1,36 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# As for static branches, a static call key owned by a module must be rejected +# while a vmlinux-owned one is accepted. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_call.c + +has_input_section orig.o .static_call_sites || + probe_skip "fixture produced no .static_call_sites on this arch" + +run_diff +assert_patched target + +# The accepted half has to show the entry was carried, not just that the +# function was: dropping the section silently would leave the patched call +# unregistered, and "target was cloned" cannot tell the two apart. +assert_section .static_call_sites +assert_reloc_sym .static_call_sites target + +rm -f "$workdir/out.o" +build_pair static_call.c -DMODNAME='"klp_testmod"' +run_diff 255 + +diff_log | grep -q 'unsupported static call key __SCK__klp_test_call' || + fail "expected rejection, got: $(diff_log | tail -1)" + +# A rejection has to leave nothing behind. out.o was removed above, so +# anything here was written by the run which was supposed to refuse. +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "module-owned static call key rejected, vmlinux-owned accepted" diff --git a/tools/objtool/tests/generic/test-static-call-new.sh b/tools/objtool/tests/generic/test-static-call-new.sh new file mode 100755 index 000000000000..f7d2b39c0fec --- /dev/null +++ b/tools/objtool/tests/generic/test-static-call-new.sh @@ -0,0 +1,45 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A patch may introduce a static call where the original function had none. +# +# The .static_call_sites entry is then new, with nothing in the original to +# correlate it against, so klp diff has to carry it into the patch from +# scratch. Where the key lives still decides whether that is allowed: a +# vmlinux key is reachable, and a module-owned one is not, for the same reason +# an existing module key is not -- late module patching lets the livepatch load +# first, and the unresolved entry is dereferenced when the module arrives. +# +# Both halves are here because they fail in opposite directions. Dropping the +# new entry leaves a static call the kernel never patches; accepting a new +# module-owned one is the corruption the check exists to prevent. +# +# Covers the same ground as corpus/x86_64/static-call-vmlinux-new and +# static-call-module-new in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_call.c -DNEW_CALL + +# The premise: the original really has no static call, the patched one does. +has_input_section orig.o .static_call_sites && + fail "fixture put a .static_call_sites in the original; nothing new to add" +has_input_section patched.o .static_call_sites || + probe_skip "compiler produced no .static_call_sites on this arch" + +run_diff +assert_patched target +assert_section .static_call_sites +assert_reloc_sym .static_call_sites target + +# The same new call, with the key owned by a module: not reachable, so the +# build has to stop rather than emit a relocation nothing will resolve. +rm -f "$workdir/out.o" +build_pair static_call.c -DNEW_CALL -DMODNAME='"klp_testmod"' +run_diff 255 +assert_diff_log 'unsupported static call key __SCK__klp_test_call' +[ -e "$workdir/out.o" ] && + fail "output object produced for a rejected input" + +pass "static call introduced by the patch carried in, or rejected for a module key" diff --git a/tools/objtool/tests/generic/test-static-local-uncorrelated.sh b/tools/objtool/tests/generic/test-static-local-uncorrelated.sh new file mode 100755 index 000000000000..d7711587d2d4 --- /dev/null +++ b/tools/objtool/tests/generic/test-static-local-uncorrelated.sh @@ -0,0 +1,41 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Some static locals must not be correlated with their counterparts in the +# running kernel; the patched code has to use a fresh copy instead. +# +# .data..once holds the "have we warned yet" flags behind WARN_ONCE. Correlate +# one and the patched function inherits the flag from before the patch, so the +# warning the patch was written to produce never fires. The same goes for the +# names the kernel generates for per-instance state -- __warned, __key, +# __func__ and friends. +# +# Both directions matter, so an ordinary static local is here too: a rule that +# refuses to correlate anything would pass a test that only checks the +# refusals. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_local_uncorrelated.c +run_diff + +# Compilers mangle static locals differently -- gcc gives __key.1, Clang +# target.__key -- so match on the base name. + +# Correlated: referenced through a klp symbol, pointing at the kernel's copy. +out_symbols | grep -q '\.klp\.sym\..*ordinary' || + fail "ordinary static local was not correlated" + +# Not correlated: no klp symbol, and a copy cloned into the patch instead. +out_symbols | grep -q '\.klp\.sym\..*__key' && + fail "__key was correlated; it must use a fresh copy" +# .sbss/.sdata on the architectures with a small-data area. +out_sections | grep -qE '\.s?(bss|data)[^ ]*__key' || + fail "__key was neither correlated nor cloned" + +out_symbols | grep -q '\.klp\.sym\..*once_flag' && + fail ".data..once variable was correlated; it must use a fresh copy" +assert_section '.data..once' + +pass "per-instance static locals cloned, ordinary ones correlated" diff --git a/tools/objtool/tests/generic/test-static-local.sh b/tools/objtool/tests/generic/test-static-local.sh new file mode 100755 index 000000000000..d57efa64dfd9 --- /dev/null +++ b/tools/objtool/tests/generic/test-static-local.sh @@ -0,0 +1,24 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A static local must be correlated with the original, not duplicated: a second +# copy would discard the state the running kernel accumulated. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair static_local.c + +in_symbols orig.o | grep -q 'counter' || + probe_skip "compiler emitted no distinct static local symbol" + +run_diff +assert_patched target + +out_symbols | grep -q '\.klp\.sym\..*\.counter' || + fail "static local not referenced through a klp relocation" + +out_symbols | grep 'counter' | grep -qvE 'UND|\.klp\.(sym|tombstone)' && + fail "static local was given a fresh definition" + +pass "static local correlated rather than duplicated" diff --git a/tools/objtool/tests/generic/test-switch-rodata.sh b/tools/objtool/tests/generic/test-switch-rodata.sh new file mode 100755 index 000000000000..fb27e96c65f6 --- /dev/null +++ b/tools/objtool/tests/generic/test-switch-rodata.sh @@ -0,0 +1,53 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A Clang switch jump table travels with the function it belongs to. +# +# For a dense enough switch Clang emits the targets as a table in +# .rodata..Lswitch.table.<function>, named after the function but not part of +# it. klp diff has to associate the two: the patched function indexes into +# that table, so a clone which does not bring it along jumps through whatever +# the kernel's copy holds -- which, when the patch changed the switch, is the +# wrong set of targets. +# +# That is an indirect jump to a stale address, not a missing symbol, so nothing +# reports it at build or load time. +# +# objtool has no switch-specific code: the table is carried by the general +# mechanism for data a cloned function references. So this is a regression +# test on that mechanism reaching a shape it is easy to get wrong, not a guard +# on a particular line -- making the table uncorrelated, the nearest sabotage, +# does not change the outcome. +# +# Covers the same ground as corpus/x86_64-llvm-switch-rodata/ +# clang-switch-rodata-assoc in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +clang_only "only Clang emits switch jump tables in their own section" + +setup +build_pair switch_rodata.c + +# The premise: this Clang really did build a table rather than a chain of +# comparisons, and the added case really did change it. +tbl=.rodata..Lswitch.table.status_to_string +has_input_section orig.o "$tbl" || + probe_skip "this clang built no jump table for the switch" +# readelf prefixes each line with "[nn]", which splits into one or two fields +# depending on the index, so strip it before counting columns. +tbl_size() +{ + in_sections "$1" | sed 's/^ *\[[ 0-9]*\] *//' | + awk -v s="$tbl" '$1 == s { print $5 }' +} +[ "$(tbl_size orig.o)" != "$(tbl_size patched.o)" ] || + fail "fixture's added case did not change the jump table" + +run_diff + +assert_patched status_to_string +assert_section "$tbl" +assert_reloc_sym .text.status_to_string "$tbl" + +pass "Clang switch jump table carried with the function it belongs to" diff --git a/tools/objtool/tests/generic/test-symid-discarded.sh b/tools/objtool/tests/generic/test-symid-discarded.sh new file mode 100755 index 000000000000..388a24984ec0 --- /dev/null +++ b/tools/objtool/tests/generic/test-symid-discarded.sh @@ -0,0 +1,44 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# .klp.symid must not reference symbols in sections the vmlinux link discards. +# Each such section has been its own bug, found only when someone built a +# config where a duplicate happened to land there, so cover the whole list +# rather than whichever one was reported last. + +. "$(dirname "$0")/../lib.sh" + +setup + +# Allocated sections which vmlinux.lds.h discards unconditionally. A symid +# referencing one of these fails the vmlinux link outright: +# +# `__exitcall_hid_exit' referenced in section `.klp.symid' of vmlinux.o: +# defined in discarded section `.exitcall.exit' of vmlinux.o +for sec in .exitcall.exit .no_trim_symbol; do + build_one symid_discarded.c a.o \ + -DFUNC_NAME=use_a -DDISCARDED_SEC="\"$sec\"" + build_one symid_discarded.c b.o \ + -DFUNC_NAME=use_b -DDISCARDED_SEC="\"$sec\"" + + # --klp-symids only runs on a file named vmlinux.o + rm -f "$workdir/vmlinux.o" + partial_link "$workdir/vmlinux.o" "$workdir/a.o" "$workdir/b.o" || + probe_skip "partial link unavailable" + + "$OBJTOOL" --klp-symids --link "$workdir/vmlinux.o" || + fail "objtool --klp-symids failed" + + symids="$(in_relocs vmlinux.o | + awk '/rela.klp.symid/,/^$/')" + + # Without this the test would also pass if symid generation stopped + # entirely. + echo "$symids" | grep -q 'dup_normal' || + fail "$sec: no symid for the duplicate in a live section" + + echo "$symids" | grep -q 'dup_discarded' && + fail "symid emitted for a symbol in discarded section $sec" +done + +pass "no symids for symbols in discarded sections" diff --git a/tools/objtool/tests/generic/test-sympos-vmlinux.sh b/tools/objtool/tests/generic/test-sympos-vmlinux.sh new file mode 100755 index 000000000000..b2b44001446f --- /dev/null +++ b/tools/objtool/tests/generic/test-sympos-vmlinux.sh @@ -0,0 +1,57 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# sympos for vmlinux, which is resolved differently from sympos for a module. +# +# A module's .ko preserves symbol table order, so klp diff can count -- that is +# what test-sympos covers. vmlinux cannot be counted: the final link reorders +# sub-sections, so the order in vmlinux.o is not the order the running kernel +# has. klp diff bridges that with .klp.symid, a table of { id, address } +# emitted into vmlinux.o whose addresses the linker resolves, read back out of +# the linked vmlinux. +# +# Getting it wrong points the relocation at a different symbol of the same +# name. Nothing fails to build or load; the patched code uses the wrong +# object. +# +# The fixture is arranged so the two answers differ: the static that comes +# first in the symbol table is placed at the *higher* address, so counting +# gives 1 and reading the linked image gives 2. Without that, both paths agree +# and the test cannot tell them apart. + +. "$(dirname "$0")/../lib.sh" + +setup + +# use_a's static sorts last by section name, use_b's first. Only use_a is +# patched, so exactly one sympos comes out. +build_one sympos_vmlinux.c orig_a.o -DFUNC_NAME=use_a -DVARSEC='".data.zzz"' +build_one sympos_vmlinux.c patched_a.o -DFUNC_NAME=use_a -DVARSEC='".data.zzz"' -DPATCHED +build_one sympos_vmlinux.c b.o -DFUNC_NAME=use_b -DVARSEC='".data.aaa"' -DNO_MODINFO + +make_vmlinux_pair "$workdir/orig_a.o" "$workdir/b.o" \ + -- "$workdir/patched_a.o" "$workdir/b.o" + +[ "$(count_input_symbols vmlinux.o dup_counter)" = 2 ] || + fail "fixture did not produce two dup_counter symbols" +has_input_section vmlinux.o .klp.symid || + fail "objtool --klp-symids emitted no .klp.symid table" +has_input_section vmlinux .klp.symid || + fail ".klp.symid did not survive the link" + +# The premise: symbol table order and address order must disagree, or the test +# proves nothing. +first_addr="$(in_symbols vmlinux | awk '$8 == "dup_counter" { print $2; exit }')" +low_addr="$(in_symbols vmlinux | awk '$8 == "dup_counter" { print $2 }' | sort | head -1)" +[ "$first_addr" != "$low_addr" ] || + probe_skip "linker did not reorder the two statics" + +assert_input_symbol dup_counter +run_diff + +# Address order says 2. Counting symbol table order would say 1. +assert_klp_sympos dup_counter 2 +out_symbols | grep -q 'dup_counter,1' && + fail "sympos 1 emitted: counted symbol table order instead of reading the linked vmlinux" + +pass "vmlinux sympos taken from the linked image, not from symbol table order" diff --git a/tools/objtool/tests/generic/test-sympos.sh b/tools/objtool/tests/generic/test-sympos.sh new file mode 100755 index 000000000000..b71d4930a22a --- /dev/null +++ b/tools/objtool/tests/generic/test-sympos.sh @@ -0,0 +1,51 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# sympos is what livepatch uses to tell duplicate symbol names apart in the +# patched object: which "dup_counter" of several the relocation means. Get it +# wrong and the patch resolves to the wrong object at load time, silently. +# +# klp_find_sympos() reports 0 when a name is unique and a 1-based position when +# it is not, so both need checking -- always reporting a position, or never, +# each looks right in one of the two cases. +# +# This is the module path, counting symbol table order. vmlinux is reordered +# by the final link and goes through .klp.symid instead; that needs a linked +# vmlinux next to vmlinux.o and is not covered here. + +. "$(dirname "$0")/../lib.sh" + +setup + +# One copy: the name is unique, so there is nothing to disambiguate. +build_one sympos_dup.c orig.o -DFUNC_NAME=use_a +build_one sympos_dup.c patched.o -DFUNC_NAME=use_a -DPATCHED +run_diff + +assert_klp_sympos dup_counter 0 + +# Two copies: positions, in symbol table order. +for p in "" "-DPATCHED"; do + # shellcheck disable=SC2086 + build_one sympos_dup.c "a$p.o" -DFUNC_NAME=use_a $p + # shellcheck disable=SC2086 + build_one sympos_dup.c "b$p.o" -DFUNC_NAME=use_b -DNO_MODINFO $p +done +partial_link "$workdir/orig.o" "$workdir/a.o" "$workdir/b.o" || + probe_skip "partial link unavailable" +partial_link "$workdir/patched.o" "$workdir/a-DPATCHED.o" "$workdir/b-DPATCHED.o" || + probe_skip "partial link unavailable" + +# Without duplicates in the input there is nothing for sympos to number. +[ "$(count_input_symbols orig.o dup_counter)" = 2 ] || + fail "fixture did not produce two dup_counter symbols" + +run_diff + +assert_klp_sympos dup_counter 1 +assert_klp_sympos dup_counter 2 +# ... and nothing still claiming the name is unique +out_symbols | grep -qE '\.klp\.sym\.[^.]+\.dup_counter,0([[:space:]]|$)' && + fail "sympos 0 emitted for a duplicated symbol" + +pass "sympos numbers duplicate symbols and stays 0 for unique ones" diff --git a/tools/objtool/tests/generic/test-symvers-parse-error.sh b/tools/objtool/tests/generic/test-symvers-parse-error.sh new file mode 100755 index 000000000000..d597f28e5617 --- /dev/null +++ b/tools/objtool/tests/generic/test-symvers-parse-error.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A malformed Module.symvers has to be reported against the line it is on. +# Module.symvers has tens of thousands of lines and is generated, so a wrong +# line number sends whoever has to fix it to the wrong place, and "line 1" is +# wrong in a way that looks plausible. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair basic.c + +# Three well-formed lines, then one with no tabs at all. +export_syms a b c +echo 'this line has no fields' >> "$workdir/Module.symvers" + +run_diff 255 + +assert_diff_log 'malformed Module.symvers' +assert_diff_log 'at line 4' + +pass "malformed Module.symvers reported against the offending line" diff --git a/tools/objtool/tests/generic/test-thinlto-ambiguity.sh b/tools/objtool/tests/generic/test-thinlto-ambiguity.sh new file mode 100755 index 000000000000..34f4f3a58e20 --- /dev/null +++ b/tools/objtool/tests/generic/test-thinlto-ambiguity.sh @@ -0,0 +1,77 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Two ThinLTO-promoted symbols sharing a demangled name must be paired up +# correctly. +# +# A file-local symbol which ThinLTO has to make visible is renamed +# helper.llvm.<hash>. With two such helpers in one link the original and the +# patched object hold two each, all four spelled differently, and demangling +# gives "helper" for all of them -- so the name is not enough to say which +# corresponds to which. +# +# Getting it wrong is silent and specific: the patch is built against the wrong +# body, so one call site gets the other helper's arithmetic. Nothing fails to +# build and nothing fails to load. +# +# test-thinlto-local covers the unambiguous case, one promoted symbol whose +# hash moved. This is the case where demangling alone is not an answer. +# +# The outcome is asserted, not the machinery: with the clang tested here the +# pairing succeeds even with the .llvm.<hash> suffix map disabled and with +# llvm_suffix() stubbed out, so no single-line sabotage distinguishes it. The +# tiered matcher this case was written for is not needed for this shape. +# +# Covers the same ground as corpus/x86_64-llvm-thinlto/ +# thin-lto-demangled-ambiguity and thin-lto-demangled-global-match in Joe +# Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +clang_only "ThinLTO requires clang" + +find_thinlto_toolchain || + probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_LD to one" + +build_thinlto() # $1 output object, $2 extra flags +{ + local t + for t in "" -DTU_B -DTU_C; do + $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections \ + $2 $t -c "$FIXTURES_DIR/thinlto_ambiguity.c" \ + -o "$workdir/tu$t.o" 2>/dev/null || return 1 + done + "$THIN_LD" -r "$workdir/tu.o" "$workdir/tu-DTU_B.o" "$workdir/tu-DTU_C.o" \ + -o "$1" 2>/dev/null || return 1 +} + +build_thinlto "$workdir/orig.o" "" || + probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)" +build_thinlto "$workdir/patched.o" -DPATCHED || + probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)" + +# The premise: two promoted helpers per object, and exactly one of them kept +# its hash -- the one the patch did not touch. Without that there is nothing +# to disambiguate. +orig_syms="$(in_symbols orig.o | grep -oE 'helper\.llvm\.[0-9]+' | sort -u)" +new_syms="$( in_symbols patched.o | grep -oE 'helper\.llvm\.[0-9]+' | sort -u)" +[ "$(echo "$orig_syms" | wc -l)" = 2 ] && [ "$(echo "$new_syms" | wc -l)" = 2 ] || + probe_skip "ThinLTO did not promote two distinct helpers here" + +kept="$(comm -12 <(echo "$orig_syms") <(echo "$new_syms"))" +moved="$(comm -13 <(echo "$orig_syms") <(echo "$new_syms"))" +[ "$(echo "$kept" | wc -w)" = 1 ] && [ "$(echo "$moved" | wc -w)" = 1 ] || + probe_skip "expected one helper to keep its hash and one to move" + +run_diff + +# Exactly one helper is cloned, and it is the one whose body changed. Cloning +# the other, or both, is what a wrong pairing looks like. +assert_not_patched "$kept" + +n="$(out_sections | grep -cE '[[:space:]]\.text\.helper\.llvm\.[0-9]+[[:space:]]')" +[ "$n" = 1 ] || + fail "expected 1 cloned helper, found $n" + +pass "ThinLTO helpers sharing a demangled name paired up correctly" diff --git a/tools/objtool/tests/generic/test-thinlto-local.sh b/tools/objtool/tests/generic/test-thinlto-local.sh new file mode 100755 index 000000000000..a266263676ed --- /dev/null +++ b/tools/objtool/tests/generic/test-thinlto-local.sh @@ -0,0 +1,48 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Correlating ThinLTO-promoted locals requires demangling the .llvm.<hash> +# suffix, and the resulting klp relocation must name the original symbol: that +# is the one in the running kernel's kallsyms. + +. "$(dirname "$0")/../lib.sh" + +setup +clang_only "ThinLTO requires clang" + +find_thinlto_toolchain || + probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_LD to one" + +build_thinlto() # $1 output object, $2 extra flags +{ + $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections $2 \ + -c "$FIXTURES_DIR/thinlto_local.c" -o "$workdir/tu_a.o" 2>/dev/null || return 1 + $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections $2 -DTU_B \ + -c "$FIXTURES_DIR/thinlto_local.c" -o "$workdir/tu_b.o" 2>/dev/null || return 1 + "$THIN_LD" -r "$workdir/tu_a.o" "$workdir/tu_b.o" -o "$1" 2>/dev/null || return 1 +} + +build_thinlto "$workdir/orig.o" "" || + probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)" +build_thinlto "$workdir/patched.o" -DPATCHED || + probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)" + +orig_sym="$(in_symbols orig.o | grep -o 'counter\.llvm\.[0-9]*' | head -1)" +new_sym="$( in_symbols patched.o | grep -o 'counter\.llvm\.[0-9]*' | head -1)" + +[ -n "$orig_sym" ] && [ -n "$new_sym" ] || + probe_skip "$THIN_CC did not promote the local symbol" + +# Equal hashes would make plain name matching work, testing nothing. +[ "$orig_sym" != "$new_sym" ] || + probe_skip "$THIN_CC gave the same ThinLTO hash for both builds" + +run_diff +assert_patched target + +out_symbols | grep -q "\.klp\.sym\.vmlinux\.$orig_sym," || + fail "expected a klp relocation naming $orig_sym" +out_symbols | grep -q "\.klp\.sym\.vmlinux\.$new_sym," && + fail "klp relocation names $new_sym, which the running kernel does not have" + +pass "ThinLTO-mangled local correlated across differing hashes ($THIN_CC, $THIN_LD)" diff --git a/tools/objtool/tests/generic/test-ubsan-noise.sh b/tools/objtool/tests/generic/test-ubsan-noise.sh new file mode 100755 index 000000000000..b415eb16bfd2 --- /dev/null +++ b/tools/objtool/tests/generic/test-ubsan-noise.sh @@ -0,0 +1,48 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# UBSAN instrumentation in an unchanged function must not make it look changed. +# +# Every instrumented operation gets a per-callsite metadata object in an +# anonymous data section -- .data..Lubsan_data and .data..Lubsan_type from GCC, +# .data..L__unnamed_ from Clang -- whose names are compiler-generated and mean +# nothing across a rebuild. is_uncorrelated_section() exists so klp diff does +# not try to pair them up. +# +# Without that, the metadata belonging to a function nobody touched compares as +# different and drags the function into the patch. A livepatch which replaces +# functions the patch never changed is not a build failure: it is a larger +# patch than intended, taking its dependencies with it, and every extra +# function is one more that can fail to correlate or to apply. +# +# Covers the same ground as corpus/x86_64-ubsan/{ubsan-shift-noise, +# ubsan-metadata-data-section,gcc-ubsan-anonymous-data,ubsan-handler-cloning} +# and corpus/x86_64-llvm-ubsan/{clang-ubsan-bounds-noise, +# clang-ubsan-handler-cloning} in Joe Lawrence's klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair ubsan_noise.c -fsanitize=shift + +# The premise: this compiler really did instrument, and left its metadata in an +# anonymous section. Without that the test is just test-basic again. +ubsan_sec="$(in_sections orig.o | + grep -oE '\.data\.\.L(ubsan_data|__unnamed_)[A-Za-z0-9_.]*' | head -1)" +[ -n "$ubsan_sec" ] || + probe_skip "compiler emitted no anonymous UBSAN data section" +assert_input_symbol untouched + +run_diff + +# The changed function is patched, and the untouched one is left alone despite +# carrying instrumentation of its own. +assert_patched touched +assert_not_patched untouched + +# The handler the patched code calls has to come with it, or the clone calls +# nothing when its check fires. +out_symbols | grep -q '__ubsan_handle_' || + fail "no __ubsan_handle_* reference in the patched output" + +pass "UBSAN metadata in an unchanged function does not drag it into the patch" diff --git a/tools/objtool/tests/lib.sh b/tools/objtool/tests/lib.sh new file mode 100644 index 000000000000..4da168d0ddca --- /dev/null +++ b/tools/objtool/tests/lib.sh @@ -0,0 +1,911 @@ +# SPDX-License-Identifier: GPL-2.0 +# +# Helpers for the objtool klp tests. A test builds a fixture twice, as the +# original and (with -DPATCHED) the patched object, runs both through +# "klp checksum" and diffs them, then asserts on the result. +# +# Assertions check properties rather than compare against recorded output: +# codegen varies between compilers and golden files would report churn instead +# of regressions. + +set -u + +TESTS_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +# Tests live in generic/ or in an architecture directory beside it, and each +# carries its own fixtures. +FIXTURES_DIR="$(cd "$(dirname "$0")/fixtures" 2>/dev/null && pwd)" + +# The kernel's convention: CROSS_COMPILE is the one knob, with per-tool +# overrides for what it does not cover. objtool itself is always a host binary +# -- it is built with HOSTCC and only reads ELF -- so an arm64 machine can run +# the x86 tests against x86 objects given a compiler that emits them. +# +# readelf reads any target, so it rarely needs overriding, and either GNU +# readelf or llvm-readelf will do: the assertions match on fields rather than +# on columns, and where the two spell something differently -- "OS [0xff20]" +# against "OS[0xff20]" for SHN_LIVEPATCH -- they accept both. BFD's objcopy is +# usually built for the host's target alone, and llvm-objcopy is the +# target-agnostic replacement. +CROSS_COMPILE="${CROSS_COMPILE:-}" +CC="${CC:-${CROSS_COMPILE}gcc}" +LD="${LD:-${CROSS_COMPILE}ld}" +READELF="${READELF:-${CROSS_COMPILE}readelf}" +OBJCOPY="${OBJCOPY:-${CROSS_COMPILE}objcopy}" + +OBJTOOL="${OBJTOOL:-$TESTS_DIR/../objtool}" + +# klp_preflight +# +# Check the environment once, before any test runs, and report what was found. +# +klp_preflight() +{ + local tmp tool cc_version arch host cc_arch + + bail() { echo "Bail out! $*" >&2; exit 1; } + + # A relative $OBJTOOL is relative to the objtool directory, not tests/. + [ -x "$OBJTOOL" ] || [ ! -x "$TESTS_DIR/../$OBJTOOL" ] || + OBJTOOL="$TESTS_DIR/../$OBJTOOL" + + [ -x "$OBJTOOL" ] || + bail "objtool not found at '$OBJTOOL' -- build it first" + + # run_diff() runs objtool from inside the test's working directory, so + # a relative path would resolve against that instead. + OBJTOOL="$(realpath "$OBJTOOL")" + + "$OBJTOOL" klp 2>&1 | grep -q checksum || + bail "objtool was built without klp support; install libxxhash (>= 0.8) and rebuild" + + command -v "${CC%% *}" >/dev/null || bail "compiler not found: $CC" + + for tool in "$READELF" "$OBJCOPY" "$LD"; do + command -v "${tool%% *}" >/dev/null || bail "$tool not found" + done + + tmp="$(mktemp -d)" || bail "mktemp failed" + echo 'int probe(void) { return 0; }' > "$tmp/probe.c" + $CC -c -o "$tmp/probe.o" "$tmp/probe.c" 2>/dev/null || + { rm -rf "$tmp"; bail "$CC cannot compile a trivial object"; } + + # $CC, $ARCH and objtool have to agree about the target, and cross runs + # are where they stop agreeing: plain "CC=clang ARCH=x86_64" on an arm64 + # box selects the x86 tests and then builds arm64 objects, because clang + # needs --target= to emit anything but the host's. + # + # Ask objtool rather than comparing machine names. It rejects an object + # it was not built for -- "unexpected ELF machine type" -- so one check + # covers every way the three can disagree, and says so once instead of + # failing every test for the same reason. + "$OBJTOOL" klp checksum "$tmp/probe.o" >/dev/null 2>&1 || + { rm -rf "$tmp" + bail "objtool rejects an object built by '$CC'; they target" \ + "different architectures (set CROSS_COMPILE, or" \ + "--target= for clang)"; } + + # BFD objcopy is usually built for the host's target alone, and + # checksum_of() needs it to read the object under test. + $OBJCOPY -O binary --only-section=.text "$tmp/probe.o" "$tmp/probe.bin" 2>/dev/null || + { rm -rf "$tmp" + bail "$OBJCOPY cannot read objects built by '$CC'; install" \ + "binutils-multiarch or set OBJCOPY=llvm-objcopy"; } + # $ARCH only chooses which directory of tests runs, so it can disagree + # with what $CC builds without objtool noticing -- and the result is the + # wrong set of tests, quietly. + case "$($READELF -hW "$tmp/probe.o" | sed -n 's/.*Machine: *//p')" in + *X86-64*|*Intel*80386*) cc_arch=x86 ;; + *AArch64*) cc_arch=arm64 ;; + *) cc_arch= ;; + esac + rm -rf "$tmp" + + # Normalize to the kernel's SRCARCH. + case "${ARCH:-$(uname -m)}" in + x86_64|i?86) arch=x86 ;; + aarch64*) arch=arm64 ;; + *) arch="${ARCH:-$(uname -m)}" ;; + esac + + case "$(uname -m)" in + x86_64|i?86) host=x86 ;; + aarch64*) host=arm64 ;; + *) host="$(uname -m)" ;; + esac + + [ -z "$cc_arch" ] || [ "$cc_arch" = "$arch" ] || + bail "ARCH says $arch but '$CC' builds $cc_arch objects;" \ + "the $arch tests would run against the wrong architecture" + + KLP_TEST_ARCH="$arch" + KLP_TEST_PREFLIGHT=done + export OBJTOOL CC KLP_TEST_ARCH KLP_TEST_PREFLIGHT + + cc_version="$($CC --version 2>/dev/null | head -1)" + cat <<EOF +# preflight +# objtool $OBJTOOL (klp: yes) +# compiler $cc_version +# arch $KLP_TEST_ARCH$([ "$arch" = "$host" ] || echo " (host $host, cross)") +# tmpdir ${TMPDIR:-/tmp} (each test builds in a fresh directory here) +EOF +} + +[ -n "${KLP_TEST_PREFLIGHT:-}" ] || klp_preflight + +# What every fixture is built with. These describe the kernel a fixture stands +# in for; -c is build_one's contract rather than a property of that kernel, so +# it lives at the compile where an override cannot drop it. +# +# -O2 the kernel's default +# -ffunction-sections -fdata-sections klp-build passes these itself, through +# KCFLAGS, whatever the configuration +# -fno-asynchronous-unwind-tables arch/x86/Makefile sets this always, so +# kernel objects carry no .eh_frame +# -fno-common the kernel's Makefile sets it, so an +# uninitialised global there lands in +# .bss rather than being SHN_COMMON, +# which has no section and so no +# checksum +# +# A test overrides it; see tools/objtool/Documentation/klp-write-tests.txt. +FIXTURE_CFLAGS="-O2 -ffunction-sections -fdata-sections -fno-common \ + -fno-asynchronous-unwind-tables" + +test_name="$(basename "$0" .sh)" +workdir= + +# The pair run_diff() and the checksum helpers work on. build_pair() names +# them again and make_vmlinux_pair() repoints orig_obj at the image it links, +# but a test which builds its objects itself with build_one() sets neither, so +# the default belongs here. +orig_obj=orig.o +patched_obj=patched.o + +pass() { KLP_TEST_REPORTED=1; echo "ok - $test_name${1:+: $1}"; exit 0; } +fail() { KLP_TEST_FAILED=1; echo "not ok - $test_name: $*"; exit 1; } + +# Two kinds of skip, and the runner tells them apart. +# +# declared_skip the test said in advance it does not apply here, e.g. +# gcc_only on a clang run. Expected indefinitely. +# probe_skip the construct did not turn up in the built object this +# time. Weaker: it may appear on another compiler version, +# and one which becomes permanent is a fixture that quietly +# stopped testing anything. +# +# A bare skip() is neither, and the runner counts it as a failure: a test which +# gives up for a reason it never declared is a hole, not an outcome. +declared_skip() +{ + KLP_TEST_REPORTED=1 + echo "ok - $test_name # SKIP (declared) $*" + exit 0 +} + +probe_skip() +{ + KLP_TEST_REPORTED=1 + echo "ok - $test_name # SKIP (probe) $*" + exit 0 +} +skip() { echo "ok - $test_name # SKIP $*"; exit 0; } + +# TAP directives. A test which is known to fail reports it rather than being +# commented out and forgotten, and one which starts passing again says so +# instead of quietly going green: the expectation has to be removed by hand, +# which is the point. +xfail() +{ + KLP_TEST_REPORTED=1 + echo "not ok - $test_name${1:+: $1} # TODO known failure" + exit 0 +} + +xpass() +{ + KLP_TEST_FAILED=1 + echo "ok - $test_name${1:+: $1} # TODO expected failure, but passed" + exit 1 +} + +# cleanup [exit] +# +# Called with "exit" from the trap, when the test is over and what it built may +# be worth keeping. Called bare by a test which has finished with one segment +# and is about to setup() another: that one is done with, whatever the outcome +# of the segments still to come, so it goes. +cleanup() +{ + [ -n "$workdir" ] || return 0 + + # run-tests.sh exports KLP_TEST_KEEP, having validated it; a test run on + # its own reads KEEP itself, so the same setting means the same thing + # either way. + case "${KLP_TEST_KEEP:-${KEEP:-failed}}" in + all) return 0 ;; + none) rm -rf "$workdir" ;; + failed|*) + [ "${1:-}" = exit ] || { + rm -rf "$workdir" + return 0 + } + # Keep what the runner is going to point at. It counts as a + # failure anything which did not report an expected outcome -- + # including a test which died before printing one, and an + # undeclared skip -- and none of those set KLP_TEST_FAILED, so + # the question to ask is whether a result was reported at all. + # An exit status cannot answer it: a test killed by a signal + # runs this trap with the status of whatever ran last. + [ -n "${KLP_TEST_REPORTED:-}" ] && [ -z "${KLP_TEST_FAILED:-}" ] && { + rm -rf "$workdir" + return 0 + } + # run on its own there is no runner to say where it was kept + [ -n "${KLP_TEST_WORKDIR:-}" ] || + echo "# kept $workdir" + ;; + esac +} + +# setup [exported symbol...] +setup() +{ + if [ -n "${KLP_TEST_WORKDIR:-}" ]; then + workdir="$KLP_TEST_WORKDIR" + mkdir -p "$workdir" || fail "cannot create $workdir" + else + workdir="$(mktemp -d)" || fail "mktemp failed" + fi + trap 'cleanup exit' EXIT + + export_syms "$@" +} + +# export_syms [symbol...] +# +# Rewrite Module.symvers so exactly these symbols are exported by vmlinux. +# Whether a symbol is listed decides between an ordinary relocation and a klp +# relocation, so tests flip it to cover both. +export_syms() +{ + : > "$workdir/Module.symvers" + add_exports vmlinux "$@" +} + +# add_exports <object> [symbol...] +# +# Append exports owned by one object, without clearing what is already there, +# so a test can describe a kernel where several objects export things. +# +# Which object owns a symbol is not cosmetic: a reference to a vmlinux symbol +# is applied when the patch module loads, and a reference to a module's symbol +# when that patched module loads, so klp diff files them in different sections. +add_exports() +{ + local owner="$1"; shift + + add_exports_ns "$owner" "" "$@" +} + +# add_exports_ns <object> <namespace> [symbol...] +# +# Exports in a symbol namespace, the last field of a Module.symvers line. +# +# A "module:<names>" namespace is EXPORT_SYMBOL_FOR_MODULES(), where the module +# loader grants access by matching the importing module's name against the +# list. A livepatch module is never on that list, so such a symbol has to be +# referenced the way an unexported one is. Ordinary namespaces are not +# special here. +add_exports_ns() +{ + local owner="$1" ns="$2"; shift 2 + + local sym + + for sym in "$@"; do + printf '0x00000000\t%s\t%s\tEXPORT_SYMBOL\t%s\n' \ + "$sym" "$owner" "$ns" >> "$workdir/Module.symvers" + done +} + +# gcc_only / clang_only <reason> +gcc_only() +{ + case "$($CC --version 2>/dev/null | head -1)" in + *[Gg][Cc][Cc]*) return 0 ;; + esac + declared_skip "gcc only${1:+: $1}" +} + +clang_only() +{ + case "$($CC --version 2>/dev/null | head -1)" in + *clang*) return 0 ;; + esac + declared_skip "clang only${1:+: $1}" +} + +# build_one <fixture.c> <output object> [cflags...] +build_one() +{ + local fixture out + fixture="$FIXTURES_DIR/$1" + out="$workdir/$2" + shift 2 + + [ -f "$fixture" ] || fail "missing fixture $fixture" + + # run_checksum only runs once per workdir. A fresh object has no + # checksums in it, so anything built now needs that to happen again. + rm -f "$workdir/.checksummed" + + $CC -c $FIXTURE_CFLAGS "$@" -o "$out" "$fixture" 2>"$workdir/cc.log" || + fail "$(basename "$fixture") does not build: $(tail -1 "$workdir/cc.log")" +} + +# build_pair <fixture.c> [cflags...] +build_pair() +{ + local fixture="$1"; shift + + # Name what this builds. A test may run several segments, and + # make_vmlinux_pair() repoints orig_obj at the image it links, so + # without this the next run_diff() would still be reading that. + orig_obj=orig.o + patched_obj=patched.o + + build_one "$fixture" orig.o "$@" + build_one "$fixture" patched.o "$@" -DPATCHED +} + +# run_objtool_check <objtool arguments...> +# +# Run objtool's ordinary check pass over the pair, as the kernel build does. +# +# Some of what klp diff consumes is produced by this pass rather than by the +# compiler: .static_call_sites, .mcount_loc, .ibt_endbr_seal, ORC. +# +# Only module objects see it before klp-build -- with CONFIG_KLP_BUILD the +# per-object pass is deferred, so built-in objects reach klp diff exactly as +# the compiler left them. +run_objtool_check() +{ + local obj + + # This rewrites both objects, so checksums taken before it describe + # something that no longer exists. As in build_one(), drop the marker + # so run_checksum() takes them again. + rm -f "$workdir/.checksummed" + + for obj in "$orig_obj" "$patched_obj"; do + "$OBJTOOL" "$@" "$workdir/$obj" || + fail "objtool $* failed on $obj" + done +} + +run_checksum() +{ + # Checksums live in the objects, and a test may ask for them more than + # once -- diffing the same pair again with a different Module.symvers, + # say. objtool does the right thing when asked twice, leaving the + # object alone, but it says so, and that warning would be most of what + # a passing run prints. Remember instead, and keep quiet. + [ -e "$workdir/.checksummed" ] && return 0 + + "$OBJTOOL" klp checksum "$workdir/$orig_obj" || + fail "klp checksum $orig_obj failed" + "$OBJTOOL" klp checksum "$workdir/$patched_obj" || + fail "klp checksum $patched_obj failed" + touch "$workdir/.checksummed" +} + +# run_diff [expected exit status] +run_diff() +{ + local expect="${1:-0}" rc=0 + + run_checksum + + # klp diff looks for Module.symvers relative to the working directory. + ( cd "$workdir" && "$OBJTOOL" klp diff "$orig_obj" "$patched_obj" out.o ) \ + > "$workdir/diff.log" 2>&1 || rc=$? + + [ "$rc" = "$expect" ] || + fail "klp diff exited $rc, expected $expect: $(tail -2 "$workdir/diff.log")" +} + +cc_supports() +{ + echo 'int f(void) { return 0; }' > "$workdir/flagtest.c" + $CC $1 -c "$workdir/flagtest.c" -o "$workdir/flagtest.o" 2>/dev/null +} + +# partial_link <output> <object...> +# +# "ld -r" through the compiler driver so the link targets the same +# architecture as the objects. +partial_link() +{ + local out="$1"; shift + + rm -f "$workdir/.checksummed" + + $CC -r -nostdlib -o "$out" "$@" 2>/dev/null || + $CC -r -nostdlib -fuse-ld=lld -o "$out" "$@" 2>/dev/null +} + +# link_vmlinux <output> <object...> +# +# Link objects into an executable, the way the kernel's final link produces +# vmlinux from vmlinux.o. Entry point 0 and no libc: nothing runs it, it only +# has to be a linked image with resolved addresses. +# +# The sub-sections have to come out in name order rather than object order, +# the way the kernel's linker script gathers .text.unlikely and .data.. apart +# from the rest. That reordering is the entire reason .klp.symid exists: a +# link which preserves order cannot tell a correct sympos from one that merely +# counted, and the caller checks the two orders really did diverge. +# +# A linker script rather than --sort-section=name, because lld accepts that +# option and ignores it -- so on a host where only lld can link the target, the +# test would quietly stop testing the thing it is named for. +# +# Three attempts because a cross run has neither $LD nor the compiler's default +# linker able to touch the target: on an arm64 host linking x86 objects, only +# lld will do it. +link_vmlinux() +{ + local out="$1" lds="$workdir/sort.lds"; shift + + echo 'SECTIONS { .data : { *(SORT_BY_NAME(.data.*)) } }' > "$lds" + + $LD -e 0 -T "$lds" -o "$out" "$@" 2>/dev/null || + $CC -nostdlib -Wl,-e,0 -Wl,-T,"$lds" \ + -o "$out" "$@" 2>/dev/null || + $CC -nostdlib -fuse-ld=lld -Wl,-e,0 -Wl,-T,"$lds" \ + -o "$out" "$@" 2>/dev/null +} + +# make_vmlinux_pair <orig object...> -- <patched object...> +# +# Build the vmlinux.o / vmlinux pair klp diff needs to resolve sympos the way +# it does for built-in code, and point the diff at it. +# +# For a module, sympos is a count in symbol table order, which klp diff can do +# from the object alone. vmlinux is different: the final link reorders +# sub-sections, so the position comes from the linked image, bridged by +# .klp.symid. klp diff only looks for that when the object it was handed is +# called vmlinux.o and a vmlinux sits beside it -- so both the name and the +# linked image matter. +make_vmlinux_pair() +{ + local orig=() patched=() seen= arg + + for arg in "$@"; do + if [ "$arg" = -- ]; then seen=y; continue; fi + if [ -n "$seen" ]; then patched+=( "$arg" ); else orig+=( "$arg" ); fi + done + + # Both sides have to have been named. Without this, forgetting the -- + # leaves one list empty, the link of nothing fails, and the test skips + # saying the toolchain cannot link -- which is a test bug wearing the + # costume of an environment one. + [ "${#orig[@]}" -gt 0 ] && [ "${#patched[@]}" -gt 0 ] || + fail "make_vmlinux_pair needs objects either side of --" + + partial_link "$workdir/vmlinux.o" "${orig[@]}" || + probe_skip "partial link unavailable" + partial_link "$workdir/patched.o" "${patched[@]}" || + probe_skip "partial link unavailable" + + "$OBJTOOL" --klp-symids --link "$workdir/vmlinux.o" || + fail "objtool --klp-symids failed" + + link_vmlinux "$workdir/vmlinux" "$workdir/vmlinux.o" || + probe_skip "cannot link a vmlinux here" + + orig_obj=vmlinux.o +} + +# build_module_pair <fixture.c> <module name> [cflags...] +# +# Build the pair as objects belonging to a module rather than to vmlinux. klp +# diff reads the object's module name from .modinfo, and that decides which +# object a relocation is attributed to and whether a reference counts as +# cross-module, so a good deal of the code has a module path the vmlinux +# fixtures never reach. +# +# The fixture defines its .modinfo name from MODNAME. Passing that through +# -D needs two levels of quoting, which is easy to get wrong at the call site. +build_module_pair() +{ + local fixture="$1" modname="$2"; shift 2 + + build_pair "$fixture" -DMODNAME="\"$modname\"" "$@" +} + +# find_thinlto_toolchain +# +# Set $THIN_LD to an lld from the same LLVM release as $CC (or THIN_CC). A +# mismatched pair fails with "Invalid summary version", which reads like a +# broken test rather than a broken environment. +# +# ThinLTO is clang-only; callers must use clang_only before calling this. +# Only $CC (or an explicit THIN_CC override) is consulted -- the harness does +# not search for a second compiler beside a gcc $CC. +find_thinlto_toolchain() +{ + local cc ver ld + + for cc in "${THIN_CC:-}" "$CC"; do + [ -n "$cc" ] || continue + command -v "${cc%% *}" >/dev/null 2>&1 || return 1 + + ver=$($cc -dumpversion 2>/dev/null | cut -d. -f1) + + for ld in "${THIN_LD:-}" "ld.lld-$ver" ld.lld; do + [ -n "$ld" ] || continue + command -v "$ld" >/dev/null 2>&1 || continue + + echo 'int probe(void) { return 0; }' > "$workdir/probe.c" + $cc -flto=thin -O2 -c "$workdir/probe.c" \ + -o "$workdir/probe.o" 2>/dev/null || continue + "$ld" -r "$workdir/probe.o" -o "$workdir/probe.elf" \ + 2>/dev/null || continue + + THIN_CC="$cc" + THIN_LD="$ld" + return 0 + done + done + + return 1 +} + +out_sections() { $READELF -S -W "$workdir/out.o" 2>/dev/null; } +out_relocs() { $READELF -r -W "$workdir/out.o" 2>/dev/null; } +out_symbols() { $READELF -s -W "$workdir/out.o" 2>/dev/null; } +diff_log() { cat "$workdir/diff.log"; } + +# out_strings <section> +# +# The strings in one section of the output, for the names livepatch matches on. +out_strings() { $READELF -p "$1" "$workdir/out.o" 2>/dev/null; } + +# Checks on the input objects, to run before klp diff. The two forms differ in +# what an absent construct means: +# +# require_* the compiler cannot produce it here -> skip +# assert_* the fixture is supposed to produce it -> fail + +in_sections() { $READELF -S -W "$workdir/$1" 2>/dev/null; } +in_symbols() { $READELF -s -W "$workdir/$1" 2>/dev/null; } +in_relocs() { $READELF -r -W "$workdir/$1" 2>/dev/null; } + +# count_input_symbols <object> <name> +# +# How many object symbols of exactly that name the input has. Deliberately not +# a grep: readelf lists section symbols too, and a newer binutils prints their +# name -- ".data.<name>" -- where an older one leaves the column blank. A dot +# is not a word character, so "grep -w <name>" counts that line as well, and +# the same object gives a different answer depending on which readelf reads it. +count_input_symbols() +{ + in_symbols "$1" | awk -v n="$2" '$4 == "OBJECT" && $8 == n' | wc -l +} + +# re_quote <string> +# +# A string as a literal basic regular expression. Nearly every name these +# assertions match on contains a dot -- .text.target, .klp.rela.vmlinux -- and +# an unescaped dot matches any character, so an assertion for one section can be +# satisfied by a different one whose name merely lines up. +re_quote() { printf '%s' "$1" | sed 's|[].[^$*\\/]|\\&|g'; } + +has_input_section() { in_sections "$1" | grep -q "[[:space:]]$(re_quote "$2")[[:space:]]"; } +has_input_symbol() { in_symbols "$1" | awk -v n="$2" '$NF == n' | grep -q .; } + +assert_input_section() +{ + local obj + + for obj in "$orig_obj" "$patched_obj"; do + has_input_section "$obj" "$1" || + fail "fixture produced no section '$1' in $obj" + done +} + +assert_input_symbol() +{ + local obj + + for obj in "$orig_obj" "$patched_obj"; do + has_input_symbol "$obj" "$1" || + fail "fixture produced no symbol '$1' in $obj" + done +} + +require_input_section() +{ + local obj + + for obj in "$orig_obj" "$patched_obj"; do + has_input_section "$obj" "$1" || + probe_skip "compiler produced no section '$1' here" + done +} + +assert_section() +{ + out_sections | grep -q "[[:space:]]$(re_quote "$1")[[:space:]]" || + fail "expected section '$1' in output" +} + +assert_no_section() +{ + out_sections | grep -q "[[:space:]]$(re_quote "$1")[[:space:]]" && + fail "unexpected section '$1' in output" + return 0 +} + +assert_patched() +{ + assert_section ".text.$1" +} + +assert_not_patched() +{ + out_sections | grep -q "[[:space:]]$(re_quote ".text.$1")[[:space:]]" && + fail "function '$1' should not have been cloned" + return 0 +} + +# section_relocs <section> +# +# The relocations against one section. readelf prints every relocation section +# in turn, so a test asking about ".smp_locks" has to cut its block out of the +# listing first. +section_relocs() +{ + local sec="${1//./\\.}" + + out_relocs | awk "/rela$sec'/,/^\$/" +} + +assert_reloc_sym() +{ + section_relocs "$1" | awk -v n="$2" '$5 == n' | grep -q . || + fail "expected a relocation to '$2' in '$1'" +} + +assert_no_reloc_sym() +{ + section_relocs "$1" | awk -v n="$2" '$5 == n' | grep -q . && + fail "unexpected relocation to '$2' in '$1'" + return 0 +} + +# assert_reloc_count <section> <count> +# +# Counts relocation entries, not header or blank lines: whether a special +# section entry was extracted once, twice or not at all is usually the whole +# question. +# +# A count of zero is ambiguous on its own -- a section with no relocations and +# no section at all both read as zero -- so require the section to exist. A +# test expecting nothing there wants assert_no_section. +assert_reloc_count() +{ + local n + + assert_section "$1" + + n="$(section_relocs "$1" | grep -cE '^[0-9a-f]{8,}')" + [ "$n" = "$2" ] || + fail "expected $2 relocations in '$1', found $n" +} + +# assert_klp_sym <symbol> [object] +# +# A klp symbol is named .klp.sym.<object>.<symbol>,<sympos>. The object +# defaults to any, since most tests care that the reference was converted at +# all rather than which object it resolved against. +assert_klp_sym() +{ + out_symbols | grep -q "\.klp\.sym\.${2:-[^.]*}\.$(re_quote "$1")," || + fail "expected klp symbol for '$1'" +} + +# assert_klp_sympos <symbol> <sympos> +# +# The number after the comma in .klp.sym.<object>.<symbol>,<sympos> says which +# of several same-named symbols livepatch should resolve to, counting from 1; +# 0 means the name is unique and no disambiguation is needed. Resolving to the +# wrong one is not a load failure, it is a patch quietly wired to the wrong +# object. +assert_klp_sympos() +{ + out_symbols | grep -qE "\.klp\.sym\.[^.]+\.$(re_quote "$1"),$2([[:space:]]|\$)" || + fail "expected klp symbol for '$1' with sympos $2, found:$( + out_symbols | grep -o "\.klp\.sym\.[^.]*\.$(re_quote "$1"),[0-9]*" | + sort -u | tr '\n' ' ')" +} + +assert_no_klp_sym() +{ + out_symbols | grep -q "\.klp\.sym\.${2:-[^.]*}\.$(re_quote "$1")," && + fail "unexpected klp symbol for '$1'" + return 0 +} + +assert_tombstone() +{ + out_symbols | grep -qE "\.klp\.tombstone\.$(re_quote "$1")([[:space:]]|\$)" || + fail "expected a tombstone for '$1'" +} + +assert_symbol() +{ + out_symbols | awk -v n="$1" '$NF == n' | grep -q . || + fail "expected symbol '$1' in output" +} + +assert_no_symbol() +{ + out_symbols | awk -v n="$1" '$NF == n' | grep -q . && + fail "unexpected symbol '$1' in output" + return 0 +} + +# assert_diff_log <regex> +# +# klp diff's combined output, for tests asserting on a diagnostic. Error +# messages are part of the interface when the whole point is that a construct +# gets rejected, and a rejection for the wrong reason is not a pass. +assert_diff_log() +{ + diff_log | grep -qE -- "$1" || + fail "expected '$1' in klp diff output: $(tail -2 "$workdir/diff.log")" +} + +# checksum_of <object> <symbol> +# +# The checksum "klp checksum" recorded for one symbol, as a hex string. +# +# .discard.sym_checksum is an array of { u64 addr; u64 checksum; }, where addr +# is the target of a relocation naming the symbol. Nothing in the section +# itself says which symbol an entry belongs to, so the relocation is what +# locates the entry; the checksum is the eight bytes after it. +# Callers use this in a command substitution, where fail() would only exit the +# subshell and the test would carry on with an empty checksum. So this returns +# non-zero and prints nothing, and the assertions below check for that. For +# the same reason it does not run run_checksum() itself: that one does call +# fail(), and from in here the message would be captured as the checksum +# rather than ending the test. The caller runs it first. +checksum_of() +{ + local obj="$workdir/$1" sym="$2" off + + off="$($READELF -rW "$obj" 2>/dev/null | + awk -v s="$sym" '/rela\.discard\.sym_checksum/,/^$/ { + if ($5 == s) { print $1; exit } + }')" + + [ -n "$off" ] || return 1 + + $OBJCOPY -O binary --only-section=.discard.sym_checksum \ + "$obj" "$workdir/checksums.bin" 2>/dev/null || return 1 + + dd if="$workdir/checksums.bin" bs=1 skip=$((16#$off + 8)) count=8 \ + status=none | od -An -tx1 | tr -d ' \n' +} + +# assert_checksum_differs <symbol> / assert_checksum_matches <symbol> +# +# Compare what klp checksum recorded for a symbol in the original against the +# patched object. This is what decides whether klp diff treats a function as +# changed, so a test asserting only that the right functions were cloned cannot +# tell a correct checksum from one which happens to differ. +checksum_pair() +{ + run_checksum + + orig_checksum="$(checksum_of "$orig_obj" "$1")" + patched_checksum="$(checksum_of "$patched_obj" "$1")" + + [ -n "$orig_checksum" ] || + fail "no checksum recorded for '$1' in $orig_obj" + [ -n "$patched_checksum" ] || + fail "no checksum recorded for '$1' in $patched_obj" +} + +assert_checksum_differs() +{ + checksum_pair "$1" + + [ "$orig_checksum" != "$patched_checksum" ] || + fail "checksum for '$1' unchanged at $orig_checksum, expected it to differ" +} + +assert_checksum_matches() +{ + checksum_pair "$1" + + [ "$orig_checksum" = "$patched_checksum" ] || + fail "checksum for '$1' changed from $orig_checksum to" \ + "$patched_checksum, expected no change" +} + +# run_post_link [expected exit status] +# +# klp post-link runs last in a livepatch build, converting the intermediate +# __klp_relocs.* sections into the .klp.rela.* form the kernel consumes. It +# needs nothing but an object containing those sections, which is what klp diff +# produces, so it runs on out.o here rather than on a built module. Rewrites +# out.o in place, so the out_* helpers show the result afterwards. +run_post_link() +{ + local expect="${1:-0}" rc=0 + + "$OBJTOOL" klp post-link "$workdir/out.o" \ + > "$workdir/post-link.log" 2>&1 || rc=$? + + [ "$rc" = "$expect" ] || + fail "klp post-link exited $rc, expected $expect:" \ + "$(tail -2 "$workdir/post-link.log")" +} + +# The flags readelf prints for a section, or nothing when it has none. The +# leading "[nn]" index is stripped first so the columns can be counted. +section_flags() +{ + out_sections | sed 's/^ *\[[ 0-9]*\] *//' | + awk -v s="$1" '$1 == s && $7 ~ /^[A-Za-z]+$/ { print $7 }' +} + +# assert_section_flag <section> <letter> +# +# SHF_RELA_LIVEPATCH is OS-specific, so readelf renders it as "o". A klp rela +# section which lost it is an ordinary rela section, which the linker may apply +# and the livepatch code will not. +assert_section_flag() +{ + local flags; flags="$(section_flags "$1")" + + [ -n "$flags" ] || + fail "section '$1' has no flags, expected '$2'" + case "$flags" in + *"$2"*) ;; + *) fail "section '$1' has flags '$flags', expected '$2'" ;; + esac +} + +# assert_klp_rela <object> <section> +# +# post-link names the converted sections .klp.rela.<object>.<section>, one per +# base section. Also checks SHF_RELA_LIVEPATCH, since the name alone is not +# what makes the kernel process it. +assert_klp_rela() +{ + local name=".klp.rela.$1.$2" + + out_sections | grep -q "[[:space:]]$(re_quote "$name")[[:space:]]" || + fail "expected section '$name' in output" + + assert_section_flag "$name" o +} + +# assert_livepatch_sym <symbol> +# +# Symbols a klp relocation resolves against live in SHN_LIVEPATCH, which +# readelf prints as "OS [0xff20]" -- llvm-readelf without the space, so match +# either. The kernel resolves these itself at patch load; anything else is a +# symbol the module loader will try, and fail, to resolve normally. +assert_livepatch_sym() +{ + out_symbols | grep -E 'OS ?\[0xff20\]' | + grep -qE "\.klp\.sym\.[^.]+\.$(re_quote "$1")," || + fail "expected a klp symbol for '$1' in SHN_LIVEPATCH" +} diff --git a/tools/objtool/tests/run-tests.sh b/tools/objtool/tests/run-tests.sh new file mode 100755 index 000000000000..c0762f532d2a --- /dev/null +++ b/tools/objtool/tests/run-tests.sh @@ -0,0 +1,239 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Run the objtool klp tests. Each test-*.sh prints one TAP result line. +# +# Tests live in generic/ and in a directory per architecture. A run executes +# generic/ plus the one matching this architecture, so a test which cannot +# apply here is not run rather than reporting a skip; what was left out is +# reported once, as a comment, so differing coverage is still visible. +# +# A run covers one compiler and one architecture; CI runs the combinations. +# The harness checks the environment once up front and fails the run if the +# suite cannot execute, rather than letting every test skip and exit 0. + +set -u + +# Determinism: the order test-*.sh expands in, and how grep's character ranges +# and sort's collation behave inside the tests, are all locale-dependent. A +# suite whose results depend on the invoking shell's locale is a suite whose +# failures cannot be reproduced. +export LC_ALL=C + +usage() +{ + cat <<EOF +usage: $(basename "$0") [-k|--keep] [test...] + +Run the objtool klp tests for this architecture: everything in generic/, plus +everything in the directory named for it. With no arguments, runs all of them. +A test may be named with or without its "test-" prefix and ".sh" suffix, and is +looked for in both directories. + +Options: + -k, --keep same as KEEP=all (see below) + +Environment: + OBJTOOL objtool binary to test (default ../objtool) + CC compiler used to build fixtures (default gcc) + ARCH architecture the tests are for (default: uname -m) + KEEP failed keep only failing tests (default) + all keep every test's working directory + none remove all working directories + +A test which needs something of its own says so in its skip message. +EOF + exit "${1:-0}" +} + +cd "$(dirname "$0")" || exit 1 + +keep_from_args= +while [ $# -gt 0 ]; do + case "$1" in + -h|--help) usage ;; + -k|--keep) keep_from_args=all; shift ;; + --) shift; break ;; + -*) echo "unknown option: $1" >&2; usage 1 ;; + *) break ;; + esac +done + +KLP_TEST_KEEP="${KEEP:-failed}" +[ -n "$keep_from_args" ] && KLP_TEST_KEEP="$keep_from_args" +case "$KLP_TEST_KEEP" in +all|none|failed) ;; +*) + echo "invalid KEEP=$KLP_TEST_KEEP (want failed, all, or none)" >&2 + exit 1 + ;; +esac +export KLP_TEST_KEEP + +echo "TAP version 13" + +# Sourcing the harness runs its preflight, which decides which architecture +# this run is for -- so the test list cannot be built before it has, and the +# tests inherit the answers rather than working them out again. +. ./lib.sh + +dirs=( generic ) +[ -d "$KLP_TEST_ARCH" ] && dirs+=( "$KLP_TEST_ARCH" ) + +if [ $# -gt 0 ]; then + tests=() + for arg in "$@"; do + name="test-${arg#test-}"; name="${name%.sh}.sh" + found= + for d in "${dirs[@]}"; do + [ -f "$d/$name" ] || continue + [ -x "$d/$name" ] || + { echo "not executable: $d/$name" >&2; exit 1; } + tests+=( "$d/$name" ); found=y + done + [ -n "$found" ] || + { echo "no such test for $KLP_TEST_ARCH: $arg" >&2; exit 1; } + done +else + tests=() + for d in "${dirs[@]}"; do + for t in "$d"/test-*.sh; do + [ -f "$t" ] && tests+=( "$t" ) + done + done + [ "${#tests[@]}" -gt 0 ] || + { echo "1..0 # SKIP no tests found"; exit 0; } + + # Tests for another architecture are absent from this run entirely. Say + # how many, so a run which covers less than the tree holds does not look + # like one that covers all of it. + for d in */; do + d="${d%/}" + case "$d" in generic|"$KLP_TEST_ARCH") continue ;; esac + n=$(ls "$d"/test-*.sh 2>/dev/null | wc -l) + [ "$n" -gt 0 ] || continue + echo "# not run: $n test$( [ "$n" = 1 ] || echo s ) in $d/" \ + "(this run is $KLP_TEST_ARCH)" + done +fi + +# One directory for the whole run, one per test inside it, mirroring the +# source layout. A run then leaves a single thing behind instead of 39 +# scattered among everything else using mktemp. +rundir="$(mktemp -d "${TMPDIR:-/tmp}/klp-tests.XXXXXXXX")" || + { echo "Bail out! cannot create a working directory" >&2; exit 1; } + +# The run's own two levels: each test's directory, and the one per source +# directory holding them. Take them away if the tests left them empty, and +# say so if they did not. Either rmdir may fail -- the glob stays unexpanded +# when nothing was created -- so ask the directory itself rather than trusting +# the status. Never rm -rf: what to keep is the tests' decision, made in +# cleanup() as each one exits, and this must not overrule it. +reap_rundir() +{ + rmdir "$rundir"/*/ 2>/dev/null + rmdir "$rundir" 2>/dev/null + [ -d "$rundir" ] +} + +# An interrupted run has the same directory to answer for, and the tests it +# never reached will not clean up on their way out. The one it was running +# has, and under the default it kept what it had built, so say where. +interrupted() +{ + reap_rundir && echo "# interrupted; what was built is in $rundir" + exit 130 +} +trap interrupted INT TERM HUP + +echo "1..${#tests[@]}" + +pass=0 fail=0 static_skip=0 probe_skip=0 xfail=0 xpass=0 +failed_dirs=() + +for t in "${tests[@]}"; do + out="$(KLP_TEST_WORKDIR="$rundir/${t%.sh}" ./"$t" 2>&1)" + rc=$? + + # A test prints one result line, but it is not necessarily the only + # thing it prints: objtool warns on stderr, and the runner captures + # that. Classify the result line itself rather than the whole of the + # output, or a stray line ahead of it makes every pattern below miss and + # the exit status decide -- which would count an expected failure, which + # exits 0, as a pass. + result="$(printf '%s\n' "$out" | grep -E '^(ok|not ok)' | tail -1)" + rest="$(printf '%s\n' "$out" | grep -Ev '^(ok|not ok)')" + + # Classify from the result line, not the exit status: a skip and a pass + # both exit 0, and telling them apart is the point of counting. + # + # The two skip kinds differ in what they promise. A static skip was + # declared before the test ran ("clang does not do this"), so it is + # expected indefinitely. A probe skip means the construct did not turn + # up this time, which is weaker and worth watching: one that becomes + # permanent is a fixture that quietly stopped testing anything. + case "$result" in + *"# SKIP (declared)"*) static_skip=$((static_skip + 1)) ;; + *"# SKIP (probe)"*) probe_skip=$((probe_skip + 1)) ;; + *"# SKIP"*) + # An undeclared skip: the test gave up for a reason it never + # said it might. That is a hole, not an expected outcome. + # + # Replace the line rather than adding one. Every test owes the + # plan exactly one result, and a consumer counting them is + # entitled to say so when the totals disagree. + rest="$rest${rest:+$'\n'}was: $result" + result="not ok - $(basename "$t" .sh): undeclared skip" + result="$result (use gcc_only/clang_only or require_input_*)" + fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;; + "not ok"*"# TODO"*) xfail=$((xfail + 1)) ;; + "ok"*"# TODO"*) xpass=$((xpass + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;; + "not ok"*) fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;; + "ok"*) pass=$((pass + 1)) ;; + *) + # No result line at all: the test died before reporting. + rest="$rest${rest:+$'\n'}exited $rc without a result line" + result="not ok - $(basename "$t" .sh): no TAP result" + fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;; + esac + + echo "$result" + [ -n "$rest" ] && printf '%s\n' "$rest" | sed 's/^[^#]/# &/' + +done + +echo "# pass:$pass fail:$fail static-skip:$static_skip" \ + "probe-skip:$probe_skip xfail:$xfail xpass:$xpass" + +case "$KLP_TEST_KEEP" in +all) + echo "# keep=all: workdirs kept in $rundir" + echo "# inspect: diff.log, readelf -S out.o under each test-* subdirectory" + echo "# cleanup: rm -rf $rundir" + ;; +failed) + if [ "${#failed_dirs[@]}" -gt 0 ]; then + echo "# keep=failed: ${#failed_dirs[@]} failing test(s) kept under $rundir:" + for d in "${failed_dirs[@]}"; do + echo "# ${d#"$rundir"/}/" + done + echo "# inspect: diff.log readelf -S out.o" + echo "# one test: $PWD/run-tests.sh <name>" + echo "# cleanup: rm -rf $rundir" + elif reap_rundir; then + echo "# $rundir was not empty;" \ + "a test did not clean up after itself" + fi + ;; +none) + if reap_rundir; then + echo "# $rundir was not empty;" \ + "a test did not clean up after itself" + elif [ "$fail" != 0 ] || [ "$xpass" != 0 ]; then + echo "# keep=none: artifacts were removed" \ + "(re-run with KEEP=failed or KEEP=all)" + fi + ;; +esac + +[ "$fail" = 0 ] && [ "$xpass" = 0 ] diff --git a/tools/objtool/tests/x86/fixtures/alt_annotate.c b/tools/objtool/tests/x86/fixtures/alt_annotate.c new file mode 100644 index 000000000000..af44d320dcb0 --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/alt_annotate.c @@ -0,0 +1,57 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An x86 alternative whose replacement instruction carries a text annotation. + * + * The kernel does this wherever an ALTERNATIVE contains something objtool has + * to be told about -- a retpoline-safe indirect branch, an intentionally + * missing ENDBR -- so the .discard.annotate_insn entry references an address + * inside .altinstr_replacement rather than inside a function. + * + * Two things make that awkward for klp diff, and both are why this fixture + * exists. Replacement code has no real symbol: objtool invents a NOTYPE fake + * symbol for it, so an annotation pointing there does not reference a FUNC. + * And .discard.annotate_insn has to be cloned after .altinstructions, or the + * replacement it names has no clone to point at yet. + * + * struct alt_instr is written out by hand as in empty_alternative.c: s32 + * instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen, u8 replacementlen, + * with an entsize so klp diff can find the entry boundaries. + * .discard.annotate_insn entries are s32 offset, s32 type; type 2 is + * ANNOTYPE_RETPOLINE_SAFE. + * + * The replacement label is global so the relocations name it rather than + * .altinstr_replacement plus an addend, which klp diff cannot convert. It is + * still NOTYPE, which is the shape that matters here. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +int target(int x) +{ + asm volatile( + "661: nop\n\t" + ".pushsection .altinstr_replacement, \"ax\"\n\t" + ".globl target_repl\n\t" + "target_repl:\n\t" + " nop\n\t" + /* The annotation lands inside the replacement. */ + ".pushsection .discard.annotate_insn, \"M\", @progbits, 8\n\t" + ".long target_repl - .\n\t" + ".long 2\n\t" + ".popsection\n\t" + "target_repl_end:\n\t" + ".popsection\n\t" + ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t" + ".long 661b - .\n\t" + ".long target_repl - .\n\t" + ".long 0\n\t" + ".byte 1\n\t" + ".byte target_repl_end - target_repl\n\t" + ".popsection\n\t"); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/x86/fixtures/checksum_alt.c b/tools/objtool/tests/x86/fixtures/checksum_alt.c new file mode 100644 index 000000000000..ed342d94eb9e --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/checksum_alt.c @@ -0,0 +1,66 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An x86 alternative whose replacement code is part of the patched function's + * checksum. + * + * checksum_update_insn() walks insn->alts after hashing the instruction + * itself, hashing the alternative's type and, when the replacement forms a + * group, its feature number and every instruction in it. So editing only the + * replacement -- code the CPU may or may not ever run -- has to move the + * function's checksum. + * + * It is reached through objtool's own alternative handling, so the object has + * to go through the check pass first: insn->alts is built there, not by the + * compiler. + * + * struct alt_instr is written out by hand as in empty_alternative.c: s32 + * instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen, u8 replacementlen. + * + * Variants, applied to the patched build only: + * + * ALT_REPL the replacement instruction changes; the original does not + * ALT_FEATURE the feature number changes; no code changes at all + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* + * Both spellings are two bytes, because a replacement may not be longer than + * the instruction it replaces: "xchg %ax, %ax" is 66 90 and two nops are + * 90 90. The original below is padded to match. + */ +#if defined(PATCHED) && defined(ALT_REPL) +#define REPL_INSN " nop\n\t nop\n\t" +#else +#define REPL_INSN " xchg %ax, %ax\n\t" +#endif + +#if defined(PATCHED) && defined(ALT_FEATURE) +#define FEATURE "7" +#else +#define FEATURE "3" +#endif + +int target(int x) +{ + asm volatile( + "661: nop\n\t" + " nop\n\t" + "662:\n\t" + ".pushsection .altinstr_replacement, \"ax\"\n\t" + ".globl target_repl\n\t" + "target_repl:\n\t" + REPL_INSN + "target_repl_end:\n\t" + ".popsection\n\t" + ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t" + ".long 661b - .\n\t" + ".long target_repl - .\n\t" + ".long " FEATURE "\n\t" + ".byte 662b - 661b\n\t" + ".byte target_repl_end - target_repl\n\t" + ".popsection\n\t"); + + return x + 1; +} diff --git a/tools/objtool/tests/x86/fixtures/empty_alternative.c b/tools/objtool/tests/x86/fixtures/empty_alternative.c new file mode 100644 index 000000000000..9336d74bfa92 --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/empty_alternative.c @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An x86 alternative with an empty replacement, as the second entry of + * ALTERNATIVE_2("orig", "repl", ft1, "", ft2) produces. Its replacement + * offset still gets a relocation, but the label it points at is the end of the + * previous replacement, which is also where the *next* one begins -- here, + * neighbor()'s. The value is meaningless; it is only ever used with a length + * of zero. + * + * struct alt_instr is written out by hand so the fixture builds without kernel + * headers: s32 instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen, + * u8 replacementlen. The section carries an entsize because klp diff needs + * either that or an ANNOTATE_DATA_SPECIAL annotation to find entry boundaries. + * + * The replacement labels are global so the relocations name them rather than + * .altinstr_replacement plus an addend. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +extern int neighbor_only(int x); + +int target(int x) +{ + asm volatile( + "661: nop\n\t" + ".pushsection .altinstr_replacement, \"ax\"\n\t" + ".globl target_repl\n\t" + "target_repl:\n\t" + " nop\n\t" + "target_repl_end:\n\t" + ".popsection\n\t" + ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t" + /* a real replacement */ + ".long 661b - .\n\t" + ".long target_repl - .\n\t" + ".long 0\n\t" + ".byte 1\n\t" + ".byte target_repl_end - target_repl\n\t" + /* an empty one, pointing at neighbor()'s replacement */ + ".long 661b - .\n\t" + ".long neighbor_repl - .\n\t" + ".long 0\n\t" + ".byte 1\n\t" + ".byte 0\n\t" + ".popsection\n\t"); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} + +/* + * Unrelated, unpatched, and referencing a symbol nothing else does, so that + * dragging its replacement in is visible. + */ +int neighbor(int x) +{ + asm volatile( + "771: nop\n\t" + ".pushsection .altinstr_replacement, \"ax\"\n\t" + ".globl neighbor_repl\n\t" + "neighbor_repl:\n\t" + " call neighbor_only\n\t" + "neighbor_repl_end:\n\t" + ".popsection\n\t" + ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t" + ".long 771b - .\n\t" + ".long neighbor_repl - .\n\t" + ".long 0\n\t" + ".byte 1\n\t" + ".byte neighbor_repl_end - neighbor_repl\n\t" + ".popsection\n\t"); + return x; +} diff --git a/tools/objtool/tests/x86/fixtures/kcfi.c b/tools/objtool/tests/x86/fixtures/kcfi.c new file mode 100644 index 000000000000..b62fc60634cf --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/kcfi.c @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An indirect call, which under kCFI is preceded by a type check and a trap. + * + * Clang emits a __cfi_<func> prefix symbol carrying the type hash ahead of + * every address-taken function, and records the trap site in .kcfi_traps. + * Both belong to the function and both have to come with it into a patch. + * + * Needs -fsanitize=kcfi, which only Clang has. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +static int impl_a(int x) +{ + return x + 1; +} + +static int impl_b(int x) +{ + return x * 2; +} + +__attribute__((noinline)) int (*pick(int x))(int) +{ + return (x & 1) ? impl_a : impl_b; +} + +int target(int x) +{ + int (*fn)(int arg) = pick(x); + +#ifdef PATCHED + return fn(x) + 2; +#else + return fn(x) + 1; +#endif +} diff --git a/tools/objtool/tests/x86/fixtures/special_sections.c b/tools/objtool/tests/x86/fixtures/special_sections.c new file mode 100644 index 000000000000..d42798848fda --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/special_sections.c @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A special section entry belonging to a patched function. SPECIAL_SEC picks + * which section, since klp diff treats eight of them alike and each needs + * extracting for the patched function and no other. + * + * The entry is written out by hand so the fixture builds without kernel + * headers. Only the leading relocation matters to klp diff; the rest is + * padded to the section's real entry size, because the entries have to be the + * right length for the boundaries between them to fall in the right places. + * + * SPECIAL_RELOCS covers __ex_table, whose entries relocate both the faulting + * instruction and its fixup; objtool rejects one with only the first. + */ + +#ifndef SPECIAL_SEC +#define SPECIAL_SEC "__bug_table" +#endif +#ifndef SPECIAL_ENTSIZE +#define SPECIAL_ENTSIZE 12 +#endif +#ifndef SPECIAL_RELOCS +#define SPECIAL_RELOCS 1 +#endif + +#define STR_(x) #x +#define STR(x) STR_(x) + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux"; + +/* + * other() gets an entry of its own, so that "the untouched function's entry + * was not dragged in" is a question the section can actually answer. With + * only target() contributing, there is nothing for klp diff to leave behind + * and the negative assertion holds however the extraction behaves. + */ +int other(int x) +{ + asm volatile( + "3: nop\n\t" + "4:\n\t" + ".pushsection " SPECIAL_SEC ", \"aM\", @progbits, " + STR(SPECIAL_ENTSIZE) "\n\t" + ".long 3b - .\n\t" +#if SPECIAL_RELOCS > 1 + ".long 4b - .\n\t" + ".fill " STR(SPECIAL_ENTSIZE) " - 8, 1, 0\n\t" +#else + ".fill " STR(SPECIAL_ENTSIZE) " - 4, 1, 0\n\t" +#endif + ".popsection\n\t"); + + return x + 9; +} + +int target(int x) +{ + asm volatile( + "1: nop\n\t" + "2:\n\t" + ".pushsection " SPECIAL_SEC ", \"aM\", @progbits, " + STR(SPECIAL_ENTSIZE) "\n\t" + ".long 1b - .\n\t" +#if SPECIAL_RELOCS > 1 + ".long 2b - .\n\t" + ".fill " STR(SPECIAL_ENTSIZE) " - 8, 1, 0\n\t" +#else + ".fill " STR(SPECIAL_ENTSIZE) " - 4, 1, 0\n\t" +#endif + ".popsection\n\t"); +#ifdef PATCHED + return x + 2; +#else + return x + 1; +#endif +} diff --git a/tools/objtool/tests/x86/fixtures/static_call_no_key.c b/tools/objtool/tests/x86/fixtures/static_call_no_key.c new file mode 100644 index 000000000000..748ba7c0f86d --- /dev/null +++ b/tools/objtool/tests/x86/fixtures/static_call_no_key.c @@ -0,0 +1,32 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A static call to a trampoline whose key symbol this object cannot see. + * + * That is the normal situation for a module: __SCK__* keys are not exported, + * and read-only access is granted at load time instead. objtool's static call + * handling has to accept it for any module, including a livepatch module built + * by hand rather than by klp-build. + * + * LIVEPATCH adds the .modinfo tag which makes objtool treat this as a + * livepatch module. + */ + +static const char __modinfo[] + __attribute__((section(".modinfo"), used, aligned(1))) = +#ifdef LIVEPATCH + "\0livepatch=Y" +#endif + "\0name=klp_testmod"; + +/* + * The trampoline is undefined here, exactly as it is for a module calling a + * static call defined in vmlinux. No __SCK__klp_test_call accompanies it. + */ +extern void __SCT__klp_test_call(void); + +int target(int x) +{ + __asm__ volatile("call __SCT__klp_test_call\n\t" ::: "memory"); + + return x + 1; +} diff --git a/tools/objtool/tests/x86/test-alt-annotation.sh b/tools/objtool/tests/x86/test-alt-annotation.sh new file mode 100755 index 000000000000..96760e6df9e5 --- /dev/null +++ b/tools/objtool/tests/x86/test-alt-annotation.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# A text annotation on an instruction inside an alternative's replacement must +# be carried into the patch. +# +# The kernel annotates replacement code wherever objtool has to be told +# something about it -- a retpoline-safe indirect branch, a deliberately absent +# ENDBR. Two things made klp diff drop those annotations: +# +# - replacement code has no real symbol, so objtool invents a NOTYPE fake +# one, and the extraction only kept references to FUNC symbols; +# - .discard.annotate_insn was processed before .altinstructions, so the +# replacement it referenced had no clone to point at yet. +# +# Nothing fails at build time when the annotation goes missing. It surfaces +# later as objtool warning about, or rejecting, the patched code it was there +# to explain. +# +# Fixed by 62a7a01fde87 ("objtool/klp: Fix extraction of text annotations for +# alternatives"). + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair alt_annotate.c + +assert_input_section .altinstructions +assert_input_section .discard.annotate_insn + +run_diff + +# The annotation has to survive, and to still name the replacement. Checking +# only the section would pass on an entry whose relocation was dropped. +assert_section .discard.annotate_insn +assert_reloc_sym .discard.annotate_insn target_repl + +pass "text annotation on an alternative replacement carried into the patch" diff --git a/tools/objtool/tests/x86/test-checksum-alt.sh b/tools/objtool/tests/x86/test-checksum-alt.sh new file mode 100755 index 000000000000..74ebfca2b3ff --- /dev/null +++ b/tools/objtool/tests/x86/test-checksum-alt.sh @@ -0,0 +1,45 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# An alternative's replacement code counts towards the checksum of the function +# it belongs to. +# +# checksum_update_insn() walks insn->alts after hashing the instruction itself: +# the alternative's type, and where the replacement forms a group, its feature +# number and every instruction in it. So a patch which edits only the +# replacement -- code that runs on some CPUs and not others -- still has to +# move the function's checksum. +# +# If it does not, klp diff decides the function is unchanged and leaves it out. +# The patch then ships the old replacement, and the bug is fixed only on +# machines whose CPU takes the other arm. Which machines those are depends on +# the feature bit, so the failure looks like a machine-specific bug rather than +# a missing patch. +# +# insn->alts exists only after objtool's check pass, so the pair goes through +# that first -- the compiler emits none of this structure itself. +# +# Covers the same ground as corpus/x86_64/checksum-alt-group, +# checksum-alt-no-group and checksum-alt-recursion-guard in Joe Lawrence's +# klp-build unit test corpus. + +. "$(dirname "$0")/../lib.sh" + +setup + +check() +{ + build_pair checksum_alt.c "-D$1" + assert_input_section .altinstructions + run_objtool_check --mcount + run_checksum + + assert_checksum_differs target +} + +# The replacement instruction itself. +check ALT_REPL +# The feature number, with no instruction anywhere changed. +check ALT_FEATURE + +pass "alternative replacement code counts towards the checksum" diff --git a/tools/objtool/tests/x86/test-empty-alternative.sh b/tools/objtool/tests/x86/test-empty-alternative.sh new file mode 100755 index 000000000000..9d40c3a405af --- /dev/null +++ b/tools/objtool/tests/x86/test-empty-alternative.sh @@ -0,0 +1,31 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# An x86 alternative with an empty replacement still gets a relocation for its +# replacement offset, but the label it points at is the end of the previous +# replacement -- which is also the start of the next one. The value is +# meaningless, and get_alt_entry() already ignores it. +# +# Cloning it drags in an unrelated neighboring replacement and everything that +# replacement references. In the reported case an empty alternative in +# meminfo_proc_show() pulled in one from proc_kcore_init(), emitting a klp +# relocation against init text which is long freed by the time the patch is +# applied. + +. "$(dirname "$0")/../lib.sh" + +setup +build_pair empty_alternative.c + +assert_input_section .altinstructions +assert_input_section .altinstr_replacement + +run_diff + +# target's own replacement comes along ... +assert_symbol target_repl +# ... neighbor's does not, nor what it references. +assert_no_symbol neighbor_repl +assert_no_symbol neighbor_only + +pass "empty alternative's replacement offset ignored when cloning" diff --git a/tools/objtool/tests/x86/test-kcfi.sh b/tools/objtool/tests/x86/test-kcfi.sh new file mode 100755 index 000000000000..b583ef608747 --- /dev/null +++ b/tools/objtool/tests/x86/test-kcfi.sh @@ -0,0 +1,39 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# Under kCFI an indirect call checks a type hash before jumping, and traps on a +# mismatch. Two things belong to the calling function and must come with it +# into a patch: +# +# - the __cfi_<func> prefix symbol holding the hash. Lose it and the patched +# function has no type identity, so indirect calls to it trap. +# - its .kcfi_traps entry. Lose that and the trap is not recognised as a +# CFI failure, so what should be a clean report becomes an oops. +# +# Neither shows up at build time. + +. "$(dirname "$0")/../lib.sh" + +clang_only "kCFI is a Clang feature" + +setup + +# Declared above that this is Clang's; a given Clang may still be too old. +cc_supports -fsanitize=kcfi || + probe_skip "this clang does not support -fsanitize=kcfi" + +build_pair kcfi.c -fsanitize=kcfi + +assert_input_section .kcfi_traps +assert_input_symbol __cfi_target + +run_diff + +assert_patched target + +# The prefix symbol comes with its function ... +assert_symbol __cfi_target +# ... and so does the trap entry. +assert_section .kcfi_traps + +pass "kCFI prefix symbol and trap entry carried with the patched function" diff --git a/tools/objtool/tests/x86/test-manual-klp-static-call.sh b/tools/objtool/tests/x86/test-manual-klp-static-call.sh new file mode 100755 index 000000000000..6c4d275e3549 --- /dev/null +++ b/tools/objtool/tests/x86/test-manual-klp-static-call.sh @@ -0,0 +1,40 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# objtool's static call handling must accept a livepatch module which cannot +# see a static call's key symbol. +# +# __SCK__* keys are not exported; modules get read-only access at load time +# instead. Livepatch modules built by klp-build do have full access to their +# keys, and a check was added on the strength of that -- but a livepatch module +# can also be written by hand, and samples/livepatch is full of them. One of +# those needs a key it cannot see as soon as it does anything that expands to a +# static call, which with CONFIG_MEM_ALLOC_PROFILING_DEBUG includes allocating +# memory: +# +# samples/livepatch/livepatch-shadow-fix1.o: error: objtool: static_call: +# can't find static_call_key symbol: __SCK__WARN_trap +# +# The module built without the livepatch tag is the control: it takes the same +# path and has always been accepted, so a test which only built the livepatch +# one could not tell this fix from the check being removed altogether. +# +# Fixed by f495054bd12e ("objtool/klp: Fix unexported static call key access +# for manually built livepatch modules"). + +. "$(dirname "$0")/../lib.sh" + +setup + +# Not a klp subcommand: this is objtool's ordinary check pass, which is what +# runs over a hand-built livepatch module during a normal kernel build. +for tag in "" -DLIVEPATCH; do + build_one static_call_no_key.c mod.o $tag + + "$OBJTOOL" --module --static-call "$workdir/mod.o" \ + > "$workdir/objtool.log" 2>&1 || + fail "objtool rejected a ${tag:+livepatch }module which cannot" \ + "see its static call key: $(tail -1 "$workdir/objtool.log")" +done + +pass "livepatch module accepted without access to its static call key" diff --git a/tools/objtool/tests/x86/test-special-sections.sh b/tools/objtool/tests/x86/test-special-sections.sh new file mode 100755 index 000000000000..8dfaa4fc9a36 --- /dev/null +++ b/tools/objtool/tests/x86/test-special-sections.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# +# klp diff extracts entries from eight special sections. Between them the +# existing tests reach .kcfi_traps, __jump_table, .static_call_sites and +# .altinstructions; __bug_table, __ex_table and __mcount_loc are covered by +# nothing, though the same extraction code serves all of them. +# +# Losing an entry is quiet in every case and wrong in a different way for each: +# a WARN() in patched code that no longer reports where it came from, an +# exception fixup that is simply not there when the faulting instruction traps, +# a function ftrace can no longer see. + +. "$(dirname "$0")/../lib.sh" + + +# section, entry size, relocations per entry +for spec in "__bug_table 12 1" "__ex_table 12 2" "__mcount_loc 8 1"; do + set -- $spec + sec=$1 + + # A fresh workdir per section: run_diff caches its checksums. + setup + build_pair special_sections.c \ + -DSPECIAL_SEC="\"$1\"" -DSPECIAL_ENTSIZE="$2" -DSPECIAL_RELOCS="$3" + + assert_input_section "$sec" + run_diff + + # Extracted, and pointing at the function that was patched. + assert_section "$sec" + assert_reloc_sym "$sec" target + assert_patched target + + # Nothing belonging to the function that was not. + assert_not_patched other + assert_no_reloc_sym "$sec" other + + cleanup +done + +pass "__bug_table, __ex_table and __mcount_loc entries extracted" diff --git a/tools/perf/trace/beauty/include/uapi/linux/sched.h b/tools/perf/trace/beauty/include/uapi/linux/sched.h index 33a4624285cd..19ffeba89428 100644 --- a/tools/perf/trace/beauty/include/uapi/linux/sched.h +++ b/tools/perf/trace/beauty/include/uapi/linux/sched.h @@ -53,7 +53,7 @@ */ #define UNSHARE_EMPTY_MNTNS 0x00100000 /* Unshare an empty mount namespace. */ -#ifndef __ASSEMBLY__ +#ifndef __ASSEMBLER__ /** * struct clone_args - arguments for the clone3 syscall * @flags: Flags for the new process as listed above. diff --git a/tools/testing/selftests/timers/clocksource-switch.c b/tools/testing/selftests/timers/clocksource-switch.c index db62a764c29e..2e86f56d953e 100644 --- a/tools/testing/selftests/timers/clocksource-switch.c +++ b/tools/testing/selftests/timers/clocksource-switch.c @@ -40,16 +40,23 @@ int get_clocksources(char list[][30]) { int fd, i; - size_t size; + ssize_t size; char buf[512]; char *head, *tmp; fd = open("/sys/devices/system/clocksource/clocksource0/available_clocksource", O_RDONLY); + if (fd < 0) + return 0; - size = read(fd, buf, 512); + size = read(fd, buf, sizeof(buf) - 1); close(fd); + if (size <= 0) + return 0; + + buf[size] = '\0'; + for (i = 0; i < 10; i++) list[i][0] = '\0'; @@ -74,11 +81,21 @@ int get_clocksources(char list[][30]) int get_cur_clocksource(char *buf, size_t size) { + ssize_t len; int fd; fd = open("/sys/devices/system/clocksource/clocksource0/current_clocksource", O_RDONLY); + if (fd < 0) + return -1; + + len = read(fd, buf, size - 1); + + close(fd); + + if (len <= 0) + return -1; - size = read(fd, buf, size); + buf[len] = '\0'; return 0; } diff --git a/tools/testing/selftests/timers/posix_timers.c b/tools/testing/selftests/timers/posix_timers.c index a92d4b957747..52b7289e1450 100644 --- a/tools/testing/selftests/timers/posix_timers.c +++ b/tools/testing/selftests/timers/posix_timers.c @@ -113,9 +113,10 @@ static void check_itimer(int which, const char *name) done = 0; - if (which == ITIMER_VIRTUAL) + if (which == ITIMER_VIRTUAL) { + clock_id = CLOCK_THREAD_CPUTIME_ID; signal(SIGVTALRM, sig_handler); - else if (which == ITIMER_PROF) { + } else if (which == ITIMER_PROF) { clock_id = CLOCK_THREAD_CPUTIME_ID; signal(SIGPROF, sig_handler); } @@ -148,7 +149,7 @@ static void check_timer_create(int which) struct itimerspec val = { .it_value.tv_sec = DELAY, }; - int clock_id = CLOCK_REALTIME; + int clock_id = which; timer_t id; done = 0; diff --git a/tools/testing/selftests/timers/raw_skew.c b/tools/testing/selftests/timers/raw_skew.c index 0c87a8fb0d7f..dfca3e2d50d2 100644 --- a/tools/testing/selftests/timers/raw_skew.c +++ b/tools/testing/selftests/timers/raw_skew.c @@ -91,6 +91,7 @@ int main(int argc, char **argv) { struct timespec mon, raw, start, end; long long delta1, delta2, interval, eppm, ppm; + long tick_nom; struct timex tx1, tx2; setbuf(stdout, NULL); @@ -129,6 +130,17 @@ int main(int argc, char **argv) /* Avg the two actual freq samples adjtimex gave us */ ppm = (long long)(tx1.freq + tx2.freq) * 1000 / 2; ppm = shift_right(ppm, 16); + + /* + * The tick value holds the coarse part of the adjustment, a + * microsecond of the tick length per unit, and time sync daemons + * do put part of their correction there. What it is worth has to + * be counted in as well, or a clock disciplined through it looks + * off by a hundred ppm a unit against what the two clocks show. + */ + tick_nom = USEC_PER_SEC / sysconf(_SC_CLK_TCK); + ppm += (long long)(tx1.tick + tx2.tick - 2 * tick_nom) * + (USEC_PER_SEC / tick_nom) * 1000 / 2; printf(" %lld.%i(act)", ppm/1000, abs((int)(ppm%1000))); if (llabs(eppm - ppm) > 1000) { diff --git a/tools/testing/selftests/x86/Makefile b/tools/testing/selftests/x86/Makefile index d478b13cc8d5..7565d2cf6a3c 100644 --- a/tools/testing/selftests/x86/Makefile +++ b/tools/testing/selftests/x86/Makefile @@ -19,7 +19,8 @@ TARGETS_C_32BIT_ONLY := entry_from_vm86 test_syscall_vdso unwind_vdso \ test_FCMOV test_FCOMI test_FISTTP \ vdso_restorer TARGETS_C_64BIT_ONLY := fsgsbase sysret_rip syscall_numbering \ - corrupt_xstate_header amx lam test_shadow_stack avx apx + corrupt_xstate_header amx lam test_shadow_stack avx apx \ + sigframe_fpu_portability # Some selftests require 32bit support enabled also on 64bit systems TARGETS_C_32BIT_NEEDED := ldt_gdt ptrace_syscall @@ -138,3 +139,5 @@ $(OUTPUT)/avx_64: CFLAGS += -mno-avx -mno-avx512f $(OUTPUT)/amx_64: EXTRA_FILES += xstate.c $(OUTPUT)/avx_64: EXTRA_FILES += xstate.c $(OUTPUT)/apx_64: EXTRA_FILES += xstate.c + +$(OUTPUT)/sigframe_fpu_portability_64: CFLAGS += -mno-avx -mno-avx512f diff --git a/tools/testing/selftests/x86/sigframe_fpu_portability.c b/tools/testing/selftests/x86/sigframe_fpu_portability.c new file mode 100644 index 000000000000..8377de052032 --- /dev/null +++ b/tools/testing/selftests/x86/sigframe_fpu_portability.c @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: GPL-2.0-only +#define _GNU_SOURCE +#include <stdio.h> +#include <signal.h> +#include <string.h> +#include <sys/ucontext.h> +#include <stdlib.h> +#include <stdint.h> +#include <stdbool.h> +#include <cpuid.h> +#include <unistd.h> +#include <sys/syscall.h> +#include <stddef.h> +#include <setjmp.h> + +#include "helpers.h" +#include "xstate.h" + +#ifndef FP_XSTATE_MAGIC2_SIZE +#define FP_XSTATE_MAGIC2_SIZE sizeof(FP_XSTATE_MAGIC2) +#endif + +/* + * This test verifies the FPU portability and consistency of the signal frame. + * + * - test_valid_shrunk_xstate_size: + * Verifies that the kernel restores state from a frame with xstate_size + * shrunk to only include active features. + * + * - test_invalid_shrunk_xstate_size: + * Verifies that the kernel rejects a frame if xstate_size is too small for + * the features enabled in xfeatures. + */ + +#define SIGFRAME_XSTATE_HDR_OFFSET 512 +#define XSTATE_SSE_ONLY_SIZE (SIGFRAME_XSTATE_HDR_OFFSET + XSAVE_HDR_SIZE) +#define XFEATURE_MASK_FPSSE ((1 << XFEATURE_FP) | (1 << XFEATURE_SSE)) + +static uint32_t ymm_offset; +static uint32_t xstate_size_ymm; +static pid_t self_pid; + +/* + * Load %ymm0 from @v, invoke SYS_kill to deliver @sig, and store the + * restored %ymm0 state back into @v within a single inline assembly + * block so the compiler cannot clobber %xmm0/%ymm0 between steps. + */ +__attribute__((target("avx"))) +static void raise_with_ymm0(int sig, uint64_t *v) +{ + register long rax asm("rax") = SYS_kill; + register long rdi asm("rdi") = self_pid; + register long rsi asm("rsi") = sig; + + asm volatile ("vmovdqu %0, %%ymm0\n\t" + "syscall\n\t" + "vmovdqu %%ymm0, %0" + : "+m" (*(char (*)[32])v), "+r" (rax) + : "r" (rdi), "r" (rsi) + : "rcx", "r11", "ymm0", "memory"); +} + +/* + * Avoid using printf() in signal handlers as it is not + * async-signal-safe. + */ +#define SIGNAL_BUF_LEN 1024 +static char sig_err_buf[SIGNAL_BUF_LEN]; + +static void sig_print(const char *msg) +{ + int left = SIGNAL_BUF_LEN - strlen(sig_err_buf) - 1; + + strncat(sig_err_buf, msg, left); +} + +static void check_avx_support(void) +{ + uint32_t eax, ebx, ecx, edx; + struct xstate_info xstate; + + /* Check CPUID.01H:ECX.OSXSAVE[bit 27] before calling xgetbv to avoid #UD */ + __cpuid(1, eax, ebx, ecx, edx); + if (!(ecx & (1 << 27))) + ksft_exit_skip("OSXSAVE not enabled by OS\n"); + + /* Check XCR0[2] (YMM) is enabled by OS */ + if (!(xgetbv(0) & (1 << XFEATURE_YMM))) + ksft_exit_skip("AVX (YMM) not enabled in XCR0\n"); + + xstate = get_xstate_info(XFEATURE_YMM); + if (!xstate.size) + ksft_exit_skip("AVX not supported by hardware\n"); + + ymm_offset = xstate.xbuf_offset; + xstate_size_ymm = xstate.xbuf_offset + xstate.size; +} + +#define TEST_YMMH_VAL (0x5656565656565656UL) + +static void __handle_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp, bool valid_size) +{ + uint64_t xfeatures, *ymmh_p; + struct xsave_buffer *xbuf; + struct _fpx_sw_bytes *sw; + ucontext_t *uc = ucp; + void *fp; + + fp = uc->uc_mcontext.fpregs; + if (!fp) { + sig_print("fpregs is NULL\n"); + return; + } + + sw = get_fpx_sw_bytes(fp); + if (sw->magic1 != FP_XSTATE_MAGIC1) { + sig_print("magic1 is not valid\n"); + return; + } + + xbuf = (struct xsave_buffer *)fp; + + /* + * Both test cases shrink the frame to contain only AVX (FP + SSE + YMM). + * If valid_size is true, set xstate_size to match the enabled features. + * If valid_size is false, set xstate_size too small (SSE only), which + * the kernel must reject. + */ + if (valid_size) + sw->xstate_size = xstate_size_ymm; + else + sw->xstate_size = XSTATE_SSE_ONLY_SIZE; + + xfeatures = get_xstatebv(xbuf); + xfeatures &= XFEATURE_MASK_FPSSE | (1 << XFEATURE_YMM); + set_xstatebv(xbuf, xfeatures); + set_fpx_sw_bytes_features(fp, xfeatures); + + *(uint32_t *)(fp + sw->xstate_size) = FP_XSTATE_MAGIC2; + + if (valid_size) { + ymmh_p = (uint64_t *)(fp + ymm_offset); + ymmh_p[0] = TEST_YMMH_VAL; + ymmh_p[1] = TEST_YMMH_VAL + 1; + } + + /* clear everything after MAGIC2. */ + if (sw->xstate_size + FP_XSTATE_MAGIC2_SIZE < sw->extended_size) + memset(fp + sw->xstate_size + FP_XSTATE_MAGIC2_SIZE, 0, + sw->extended_size - sw->xstate_size - FP_XSTATE_MAGIC2_SIZE); +} + +static void handle_valid_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp) +{ + __handle_shrunk_xstate_size(sig, si, ucp, true); +} + +static void handle_invalid_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp) +{ + __handle_shrunk_xstate_size(sig, si, ucp, false); +} + +static void test_valid_shrunk_xstate_size(void) +{ + uint64_t v[4]; + + sig_err_buf[0] = 0; + sethandler(SIGUSR1, handle_valid_shrunk_xstate_size, 0); + + v[0] = 0x1111111111111111ULL; + v[1] = 0x2222222222222222ULL; + v[2] = 0x3333333333333333ULL; + v[3] = 0x4444444444444444ULL; + raise_with_ymm0(SIGUSR1, v); + + if (sig_err_buf[0]) + ksft_test_result_fail("%s\n", sig_err_buf); + else if (v[2] == TEST_YMMH_VAL && v[3] == (TEST_YMMH_VAL + 1)) + ksft_test_result_pass("YMM state restored correctly from shrunk frame\n"); + else + ksft_test_result_fail( + "Got upper bits: 0x%lx 0x%lx (expected %lx %lx)\n", + v[2], v[3], TEST_YMMH_VAL, TEST_YMMH_VAL + 1); + + clearhandler(SIGUSR1); +} + +static sigjmp_buf segv_jmpbuf; + +static void handle_segv(int sig, siginfo_t *si, void *ucp) +{ + siglongjmp(segv_jmpbuf, 1); +} + +static void test_invalid_shrunk_xstate_size(void) +{ + uint64_t v[4]; + + sig_err_buf[0] = 0; + sethandler(SIGUSR1, handle_invalid_shrunk_xstate_size, 0); + sethandler(SIGSEGV, handle_segv, 0); + + if (sigsetjmp(segv_jmpbuf, 1) == 0) { + v[0] = 0x1111111111111111ULL; + v[1] = 0x2222222222222222ULL; + v[2] = 0x3333333333333333ULL; + v[3] = 0x4444444444444444ULL; + raise_with_ymm0(SIGUSR1, v); + sig_print("Inconsistent size was NOT rejected\n"); + } + + clearhandler(SIGUSR1); + clearhandler(SIGSEGV); + + if (sig_err_buf[0]) + ksft_test_result_fail("%s\n", sig_err_buf); + else + ksft_test_result_pass("Inconsistent size correctly rejected\n"); +} + +int main(void) +{ + ksft_print_header(); + ksft_set_plan(2); + + self_pid = getpid(); + + check_avx_support(); + + test_valid_shrunk_xstate_size(); + test_invalid_shrunk_xstate_size(); + + ksft_finished(); + return 0; +} diff --git a/tools/testing/selftests/x86/xstate.c b/tools/testing/selftests/x86/xstate.c index 97fe4bd8bc77..0ab577157cd7 100644 --- a/tools/testing/selftests/x86/xstate.c +++ b/tools/testing/selftests/x86/xstate.c @@ -34,18 +34,6 @@ (1 << XFEATURE_XTILEDATA) | \ (1 << XFEATURE_APX)) -static inline uint64_t xgetbv(uint32_t index) -{ - uint32_t eax, edx; - - asm volatile("xgetbv" : "=a" (eax), "=d" (edx) : "c" (index)); - return eax + ((uint64_t)edx << 32); -} - -static inline uint64_t get_xstatebv(struct xsave_buffer *xbuf) -{ - return *(uint64_t *)(&xbuf->header); -} static struct xstate_info xstate; diff --git a/tools/testing/selftests/x86/xstate.h b/tools/testing/selftests/x86/xstate.h index 6ee816e7625a..eedf0cab7ccb 100644 --- a/tools/testing/selftests/x86/xstate.h +++ b/tools/testing/selftests/x86/xstate.h @@ -3,6 +3,8 @@ #define __SELFTESTS_X86_XSTATE_H #include <stdint.h> +#include <stdlib.h> +#include <string.h> #include "kselftest.h" @@ -94,6 +96,14 @@ static inline void xrstor(struct xsave_buffer *xbuf, uint64_t rfbm) : : "D" (xbuf), "a" (rfbm_lo), "d" (rfbm_hi)); } +static inline uint64_t xgetbv(uint32_t index) +{ + uint32_t eax, edx; + + asm volatile("xgetbv" : "=a" (eax), "=d" (edx) : "c" (index)); + return eax + ((uint64_t)edx << 32); +} + #define CPUID_LEAF_XSTATE 0xd #define CPUID_SUBLEAF_XSTATE_USER 0x0 @@ -160,6 +170,11 @@ static inline void set_xstatebv(struct xsave_buffer *xbuf, uint64_t bv) *(uint64_t *)(&xbuf->header) = bv; } +static inline uint64_t get_xstatebv(struct xsave_buffer *xbuf) +{ + return *(uint64_t *)(&xbuf->header); +} + /* See 'struct _fpx_sw_bytes' at sigcontext.h */ #define SW_BYTES_OFFSET 464 /* N.B. The struct's field name varies so read from the offset. */ @@ -175,6 +190,11 @@ static inline uint64_t get_fpx_sw_bytes_features(void *buffer) return *(uint64_t *)(buffer + SW_BYTES_BV_OFFSET); } +static inline void set_fpx_sw_bytes_features(void *buffer, uint64_t features) +{ + *(uint64_t *)(buffer + SW_BYTES_BV_OFFSET) = features; +} + static inline void set_rand_data(struct xstate_info *xstate, struct xsave_buffer *xbuf) { int *ptr = (int *)&xbuf->bytes[xstate->xbuf_offset]; |
