summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-09-30 13:15:48 +0100
committerMark Brown <broonie@kernel.org>2026-09-30 13:15:48 +0100
commit24bf019cbe7e44d1e933480933b8410886bf4c60 (patch)
tree68ba5a2536df9ec6d93b3f524b60b53011988d08
parentd366f5b1dbc9ab26c4575690274dd8f6412805b0 (diff)
parent1aeb52f7869a680c042fc9ae806281e8f60469f7 (diff)
downloadlinux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.tar.gz
linux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.zip
Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git
# Conflicts: # Documentation/scheduler/index.rst # arch/arm64/configs/defconfig
-rw-r--r--Documentation/ABI/testing/sysfs-devices-system-cpu24
-rw-r--r--Documentation/ABI/testing/sysfs-platform-ts550054
-rw-r--r--Documentation/arch/arm64/silicon-errata.rst3
-rw-r--r--Documentation/arch/x86/tdx.rst21
-rw-r--r--Documentation/arch/x86/xstate.rst54
-rw-r--r--Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml1
-rw-r--r--Documentation/driver-api/index.rst1
-rw-r--r--Documentation/driver-api/steal-governor.rst151
-rw-r--r--Documentation/filesystems/resctrl.rst19
-rw-r--r--Documentation/scheduler/index.rst1
-rw-r--r--Documentation/scheduler/sched-paravirt.rst67
-rw-r--r--MAINTAINERS14
-rw-r--r--arch/Kconfig38
-rw-r--r--arch/arm/kernel/perf_regs.c8
-rw-r--r--arch/arm64/Kconfig10
-rw-r--r--arch/arm64/Kconfig.platforms1
-rw-r--r--arch/arm64/boot/dts/qcom/purwa.dtsi5
-rw-r--r--arch/arm64/configs/defconfig2
-rw-r--r--arch/arm64/include/asm/preempt.h10
-rw-r--r--arch/arm64/kernel/paravirt.c4
-rw-r--r--arch/arm64/kernel/perf_regs.c8
-rw-r--r--arch/csky/kernel/perf_regs.c8
-rw-r--r--arch/loongarch/Kconfig1
-rw-r--r--arch/loongarch/kernel/paravirt.c4
-rw-r--r--arch/loongarch/kernel/perf_regs.c8
-rw-r--r--arch/mips/kernel/perf_regs.c8
-rw-r--r--arch/parisc/kernel/perf_regs.c8
-rw-r--r--arch/powerpc/Kconfig1
-rw-r--r--arch/powerpc/perf/perf_regs.c2
-rw-r--r--arch/powerpc/platforms/pseries/setup.c4
-rw-r--r--arch/riscv/Kconfig1
-rw-r--r--arch/riscv/include/asm/smp.h12
-rw-r--r--arch/riscv/kernel/paravirt.c4
-rw-r--r--arch/riscv/kernel/perf_regs.c8
-rw-r--r--arch/riscv/kernel/sbi-ipi.c4
-rw-r--r--arch/riscv/kernel/smp.c12
-rw-r--r--arch/s390/Kconfig1
-rw-r--r--arch/s390/include/asm/preempt.h11
-rw-r--r--arch/s390/kernel/hiperdispatch.c10
-rw-r--r--arch/s390/kernel/perf_regs.c2
-rw-r--r--arch/um/kernel/um_arch.c78
-rw-r--r--arch/x86/Kconfig10
-rw-r--r--arch/x86/boot/compressed/error.c19
-rw-r--r--arch/x86/boot/compressed/error.h1
-rw-r--r--arch/x86/boot/compressed/mem.c42
-rw-r--r--arch/x86/boot/compressed/sev.h2
-rw-r--r--arch/x86/boot/compressed/tdx-shared.c2
-rw-r--r--arch/x86/boot/early_serial_console.c13
-rw-r--r--arch/x86/boot/string.c13
-rw-r--r--arch/x86/coco/sev/core.c30
-rw-r--r--arch/x86/coco/tdx/tdx-shared.c31
-rw-r--r--arch/x86/coco/tdx/tdx.c41
-rw-r--r--arch/x86/crypto/aegis128-aesni-glue.c3
-rw-r--r--arch/x86/crypto/aesni-intel_glue.c7
-rw-r--r--arch/x86/crypto/aria_aesni_avx2_glue.c11
-rw-r--r--arch/x86/crypto/aria_aesni_avx_glue.c11
-rw-r--r--arch/x86/crypto/aria_gfni_avx512_glue.c11
-rw-r--r--arch/x86/crypto/camellia_aesni_avx2_glue.c11
-rw-r--r--arch/x86/crypto/camellia_aesni_avx_glue.c11
-rw-r--r--arch/x86/crypto/cast5_avx_glue.c7
-rw-r--r--arch/x86/crypto/cast6_avx_glue.c7
-rw-r--r--arch/x86/crypto/serpent_avx2_glue.c9
-rw-r--r--arch/x86/crypto/serpent_avx_glue.c7
-rw-r--r--arch/x86/crypto/sm4_aesni_avx2_glue.c11
-rw-r--r--arch/x86/crypto/sm4_aesni_avx_glue.c11
-rw-r--r--arch/x86/crypto/twofish_avx_glue.c6
-rw-r--r--arch/x86/events/amd/uncore.c39
-rw-r--r--arch/x86/events/core.c488
-rw-r--r--arch/x86/events/intel/core.c154
-rw-r--r--arch/x86/events/intel/ds.c232
-rw-r--r--arch/x86/events/intel/lbr.c2
-rw-r--r--arch/x86/events/perf_event.h217
-rw-r--r--arch/x86/include/asm/cpufeatures.h5
-rw-r--r--arch/x86/include/asm/cpuid/api.h2
-rw-r--r--arch/x86/include/asm/fpu/regset.h6
-rw-r--r--arch/x86/include/asm/fpu/sched.h6
-rw-r--r--arch/x86/include/asm/fpu/types.h25
-rw-r--r--arch/x86/include/asm/fpu/xstate.h3
-rw-r--r--arch/x86/include/asm/kvm-x86-ops.h1
-rw-r--r--arch/x86/include/asm/kvm_host.h10
-rw-r--r--arch/x86/include/asm/local.h4
-rw-r--r--arch/x86/include/asm/math_emu.h15
-rw-r--r--arch/x86/include/asm/msr-index.h10
-rw-r--r--arch/x86/include/asm/perf_event.h46
-rw-r--r--arch/x86/include/asm/preempt.h28
-rw-r--r--arch/x86/include/asm/processor.h2
-rw-r--r--arch/x86/include/asm/sev.h4
-rw-r--r--arch/x86/include/asm/shared/string.h52
-rw-r--r--arch/x86/include/asm/shared/tdx.h6
-rw-r--r--arch/x86/include/asm/string.h21
-rw-r--r--arch/x86/include/asm/string_64.h1
-rw-r--r--arch/x86/include/asm/tdx.h23
-rw-r--r--arch/x86/include/asm/tdx_global_metadata.h9
-rw-r--r--arch/x86/include/asm/traps.h2
-rw-r--r--arch/x86/include/uapi/asm/perf_regs.h53
-rw-r--r--arch/x86/include/uapi/asm/sigcontext.h15
-rw-r--r--arch/x86/kernel/asm-offsets.c1
-rw-r--r--arch/x86/kernel/cpu/amd.c21
-rw-r--r--arch/x86/kernel/cpu/bugs.c18
-rw-r--r--arch/x86/kernel/cpu/common.c98
-rw-r--r--arch/x86/kernel/cpu/microcode/intel-ucode-defs.h91
-rw-r--r--arch/x86/kernel/cpu/mtrr/amd.c9
-rw-r--r--arch/x86/kernel/cpu/resctrl/ctrlmondata.c6
-rw-r--r--arch/x86/kernel/cpu/scattered.c2
-rw-r--r--arch/x86/kernel/cpu/sgx/main.c8
-rw-r--r--arch/x86/kernel/cpu/vmware.c4
-rw-r--r--arch/x86/kernel/crash.c6
-rw-r--r--arch/x86/kernel/fpu/bugs.c4
-rw-r--r--arch/x86/kernel/fpu/core.c46
-rw-r--r--arch/x86/kernel/fpu/init.c6
-rw-r--r--arch/x86/kernel/fpu/regset.c6
-rw-r--r--arch/x86/kernel/fpu/signal.c136
-rw-r--r--arch/x86/kernel/fpu/xstate.c62
-rw-r--r--arch/x86/kernel/kvm.c4
-rw-r--r--arch/x86/kernel/perf_regs.c174
-rw-r--r--arch/x86/kernel/shstk.c2
-rw-r--r--arch/x86/kvm/mmu/mmu.c4
-rw-r--r--arch/x86/kvm/svm/sev.c2
-rw-r--r--arch/x86/kvm/vmx/pmu_intel.c28
-rw-r--r--arch/x86/kvm/vmx/tdx.c99
-rw-r--r--arch/x86/kvm/vmx/tdx.h2
-rw-r--r--arch/x86/kvm/vmx/vmx.c10
-rw-r--r--arch/x86/kvm/vmx/vmx.h15
-rw-r--r--arch/x86/platform/Makefile1
-rw-r--r--arch/x86/platform/olpc/olpc-xo15-sci.c4
-rw-r--r--arch/x86/platform/pvh/enlighten.c3
-rw-r--r--arch/x86/platform/ts5500/Makefile2
-rw-r--r--arch/x86/platform/ts5500/ts5500.c341
-rw-r--r--arch/x86/virt/svm/sev.c172
-rw-r--r--arch/x86/virt/vmx/tdx/seamcall_internal.h19
-rw-r--r--arch/x86/virt/vmx/tdx/tdx.c437
-rw-r--r--arch/x86/virt/vmx/tdx/tdx.h10
-rw-r--r--arch/x86/virt/vmx/tdx/tdx_global_metadata.c23
-rw-r--r--arch/x86/virt/vmx/tdx/tdxcall.S10
-rw-r--r--arch/x86/xen/pmu.c5
-rw-r--r--drivers/base/cpu.c12
-rw-r--r--drivers/clocksource/timer-clint.c4
-rw-r--r--drivers/crypto/ccp/sev-dev.c2
-rw-r--r--drivers/firmware/efi/libstub/x86-stub.c39
-rw-r--r--drivers/irqchip/Kconfig6
-rw-r--r--drivers/irqchip/irq-aclint-sswi.c4
-rw-r--r--drivers/irqchip/irq-al-fic.c2
-rw-r--r--drivers/irqchip/irq-gic-v3-its.c6
-rw-r--r--drivers/irqchip/irq-gic-v3.c12
-rw-r--r--drivers/irqchip/irq-gic-v5.c9
-rw-r--r--drivers/irqchip/irq-gic.c8
-rw-r--r--drivers/irqchip/irq-lan966x-oic.c1
-rw-r--r--drivers/irqchip/irq-mtk-cirq.c2
-rw-r--r--drivers/irqchip/irq-pruss-intc.c38
-rw-r--r--drivers/irqchip/irq-riscv-imsic-early.c4
-rw-r--r--drivers/irqchip/irq-riscv-imsic-state.c13
-rw-r--r--drivers/irqchip/irq-riscv-imsic-state.h1
-rw-r--r--drivers/irqchip/irq-sifive-plic.c2
-rw-r--r--drivers/irqchip/irq-vic.c2
-rw-r--r--drivers/irqchip/qcom-pdc.c3
-rw-r--r--drivers/resctrl/mpam_resctrl.c5
-rw-r--r--drivers/soc/fsl/qe/qe_ports_ic.c1
-rw-r--r--drivers/virt/Kconfig17
-rw-r--r--drivers/virt/Makefile1
-rw-r--r--drivers/virt/coco/tdx-guest/tdx-guest.c6
-rw-r--r--drivers/virt/steal_governor.c296
-rw-r--r--drivers/xen/time.c4
-rw-r--r--fs/aio.c2
-rw-r--r--fs/exec.c13
-rw-r--r--fs/proc/uptime.c6
-rw-r--r--fs/resctrl/ctrlmondata.c6
-rw-r--r--fs/resctrl/pseudo_lock.c9
-rw-r--r--fs/resctrl/rdtgroup.c8
-rw-r--r--include/asm-generic/preempt.h10
-rw-r--r--include/linux/bitmap.h14
-rw-r--r--include/linux/cpumask.h42
-rw-r--r--include/linux/hrtimer.h10
-rw-r--r--include/linux/interrupt.h6
-rw-r--r--include/linux/irq-entry-common.h17
-rw-r--r--include/linux/kernel.h20
-rw-r--r--include/linux/kernel_stat.h11
-rw-r--r--include/linux/list.h6
-rw-r--r--include/linux/perf_event.h23
-rw-r--r--include/linux/perf_regs.h36
-rw-r--r--include/linux/posix-timers.h39
-rw-r--r--include/linux/preempt.h20
-rw-r--r--include/linux/resctrl.h19
-rw-r--r--include/linux/sched.h32
-rw-r--r--include/linux/sched/cputime.h6
-rw-r--r--include/linux/sched/task.h1
-rw-r--r--include/linux/wait.h2
-rw-r--r--include/uapi/linux/perf_event.h49
-rw-r--r--include/uapi/linux/sched.h2
-rw-r--r--include/vdso/math64.h31
-rw-r--r--io_uring/rw.c2
-rw-r--r--kernel/Kconfig.kexec2
-rw-r--r--kernel/Kconfig.preempt13
-rw-r--r--kernel/cpu.c6
-rw-r--r--kernel/crash_core.c2
-rw-r--r--kernel/entry/common.c17
-rw-r--r--kernel/events/core.c177
-rw-r--r--kernel/exit.c13
-rw-r--r--kernel/futex/requeue.c2
-rw-r--r--kernel/futex/waitwake.c8
-rw-r--r--kernel/irq/irqdomain.c1
-rw-r--r--kernel/locking/rtmutex.c2
-rw-r--r--kernel/sched/core.c444
-rw-r--r--kernel/sched/cputime.c4
-rw-r--r--kernel/sched/deadline.c9
-rw-r--r--kernel/sched/debug.c21
-rw-r--r--kernel/sched/ext/ext.c18
-rw-r--r--kernel/sched/ext/ext.h7
-rw-r--r--kernel/sched/fair.c131
-rw-r--r--kernel/sched/idle.c5
-rw-r--r--kernel/sched/rt.c7
-rw-r--r--kernel/sched/sched.h80
-rw-r--r--kernel/sched/stop_task.c5
-rw-r--r--kernel/sched/wait.c22
-rw-r--r--kernel/time/hrtimer.c23
-rw-r--r--kernel/time/posix-cpu-timers.c103
-rw-r--r--kernel/time/posix-timers.c26
-rw-r--r--kernel/time/posix-timers.h3
-rw-r--r--kernel/time/sleep_timeout.c4
-rw-r--r--kernel/time/tick-sched.c30
-rw-r--r--kernel/time/time_test.c16
-rw-r--r--kernel/time/timeconv.c6
-rw-r--r--kernel/time/timekeeping.c4
-rw-r--r--kernel/time/timer.c2
-rw-r--r--kernel/time/timer_migration.c6
-rw-r--r--kernel/time/vsyscall.c26
-rw-r--r--lib/bitmap.c17
-rw-r--r--lib/crc/x86/crc-pclmul-template.h6
-rw-r--r--lib/crypto/x86/blake2s.h4
-rw-r--r--lib/crypto/x86/chacha.h3
-rw-r--r--lib/crypto/x86/nh.h4
-rw-r--r--lib/crypto/x86/poly1305.h7
-rw-r--r--lib/crypto/x86/sha1.h4
-rw-r--r--lib/crypto/x86/sha256.h4
-rw-r--r--lib/crypto/x86/sha512.h3
-rw-r--r--lib/crypto/x86/sm3.h3
-rw-r--r--lib/raid/xor/Makefile2
-rw-r--r--lib/raid/xor/x86/xor-avx512.c122
-rw-r--r--lib/raid/xor/x86/xor_arch.h31
-rw-r--r--lib/vdso/gettimeofday.c2
-rw-r--r--net/core/pktgen.c4
-rw-r--r--tools/objtool/Documentation/klp-test-design.txt286
-rw-r--r--tools/objtool/Documentation/klp-write-tests.txt266
-rw-r--r--tools/objtool/Makefile15
-rw-r--r--tools/objtool/tests/generic/fixtures/abs_and_addressable.c44
-rw-r--r--tools/objtool/tests/generic/fixtures/basic.c20
-rw-r--r--tools/objtool/tests/generic/fixtures/changed_data.c16
-rw-r--r--tools/objtool/tests/generic/fixtures/checksum_data.c116
-rw-r--r--tools/objtool/tests/generic/fixtures/checksum_insn.c78
-rw-r--r--tools/objtool/tests/generic/fixtures/checksum_position.c42
-rw-r--r--tools/objtool/tests/generic/fixtures/checksum_skip.c47
-rw-r--r--tools/objtool/tests/generic/fixtures/cold_function.c21
-rw-r--r--tools/objtool/tests/generic/fixtures/cross_module.c25
-rw-r--r--tools/objtool/tests/generic/fixtures/data_alignment.c29
-rw-r--r--tools/objtool/tests/generic/fixtures/function_removal.c25
-rw-r--r--tools/objtool/tests/generic/fixtures/init_reference.c17
-rw-r--r--tools/objtool/tests/generic/fixtures/jump_label.c76
-rw-r--r--tools/objtool/tests/generic/fixtures/klp_funcs.c31
-rw-r--r--tools/objtool/tests/generic/fixtures/local_to_global.c34
-rw-r--r--tools/objtool/tests/generic/fixtures/new_data.c23
-rw-r--r--tools/objtool/tests/generic/fixtures/new_export_ref.c35
-rw-r--r--tools/objtool/tests/generic/fixtures/new_function.c21
-rw-r--r--tools/objtool/tests/generic/fixtures/no_modinfo.c11
-rw-r--r--tools/objtool/tests/generic/fixtures/special_section.c24
-rw-r--r--tools/objtool/tests/generic/fixtures/special_section_shared.c31
-rw-r--r--tools/objtool/tests/generic/fixtures/static_call.c59
-rw-r--r--tools/objtool/tests/generic/fixtures/static_local.c17
-rw-r--r--tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c41
-rw-r--r--tools/objtool/tests/generic/fixtures/switch_rodata.c31
-rw-r--r--tools/objtool/tests/generic/fixtures/symid_discarded.c25
-rw-r--r--tools/objtool/tests/generic/fixtures/sympos_dup.c32
-rw-r--r--tools/objtool/tests/generic/fixtures/sympos_vmlinux.c40
-rw-r--r--tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c57
-rw-r--r--tools/objtool/tests/generic/fixtures/thinlto_local.c39
-rw-r--r--tools/objtool/tests/generic/fixtures/ubsan_noise.c49
-rwxr-xr-xtools/objtool/tests/generic/test-abs-and-addressable.sh50
-rwxr-xr-xtools/objtool/tests/generic/test-basic.sh17
-rwxr-xr-xtools/objtool/tests/generic/test-changed-data.sh18
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-data.sh61
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-debug.sh49
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-insn.sh49
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-position.sh51
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-skip.sh81
-rwxr-xr-xtools/objtool/tests/generic/test-checksum-value.sh37
-rwxr-xr-xtools/objtool/tests/generic/test-cold-function.sh39
-rwxr-xr-xtools/objtool/tests/generic/test-data-alignment.sh40
-rwxr-xr-xtools/objtool/tests/generic/test-export-symbol-for-modules.sh39
-rwxr-xr-xtools/objtool/tests/generic/test-function-removal.sh34
-rwxr-xr-xtools/objtool/tests/generic/test-init-reference.sh29
-rwxr-xr-xtools/objtool/tests/generic/test-jump-label-exempt-keys.sh51
-rwxr-xr-xtools/objtool/tests/generic/test-jump-label-key.sh44
-rwxr-xr-xtools/objtool/tests/generic/test-jump-label-module-key.sh22
-rwxr-xr-xtools/objtool/tests/generic/test-jump-label-module-static-key.sh45
-rwxr-xr-xtools/objtool/tests/generic/test-jump-label-new-key.sh51
-rwxr-xr-xtools/objtool/tests/generic/test-klp-funcs-content.sh45
-rwxr-xr-xtools/objtool/tests/generic/test-local-to-global-flip.sh63
-rwxr-xr-xtools/objtool/tests/generic/test-local-vs-export.sh32
-rwxr-xr-xtools/objtool/tests/generic/test-missing-checksum.sh18
-rwxr-xr-xtools/objtool/tests/generic/test-missing-modinfo.sh16
-rwxr-xr-xtools/objtool/tests/generic/test-modname-normalize.sh26
-rwxr-xr-xtools/objtool/tests/generic/test-module-object.sh31
-rwxr-xr-xtools/objtool/tests/generic/test-module-vmlinux-reloc.sh40
-rwxr-xr-xtools/objtool/tests/generic/test-new-data.sh27
-rwxr-xr-xtools/objtool/tests/generic/test-new-export-ref.sh46
-rwxr-xr-xtools/objtool/tests/generic/test-new-function.sh16
-rwxr-xr-xtools/objtool/tests/generic/test-post-link.sh42
-rwxr-xr-xtools/objtool/tests/generic/test-special-section-shared.sh26
-rwxr-xr-xtools/objtool/tests/generic/test-special-section.sh20
-rwxr-xr-xtools/objtool/tests/generic/test-static-call-annotate-stripped.sh42
-rwxr-xr-xtools/objtool/tests/generic/test-static-call-module-key.sh36
-rwxr-xr-xtools/objtool/tests/generic/test-static-call-new.sh45
-rwxr-xr-xtools/objtool/tests/generic/test-static-local-uncorrelated.sh41
-rwxr-xr-xtools/objtool/tests/generic/test-static-local.sh24
-rwxr-xr-xtools/objtool/tests/generic/test-switch-rodata.sh53
-rwxr-xr-xtools/objtool/tests/generic/test-symid-discarded.sh44
-rwxr-xr-xtools/objtool/tests/generic/test-sympos-vmlinux.sh57
-rwxr-xr-xtools/objtool/tests/generic/test-sympos.sh51
-rwxr-xr-xtools/objtool/tests/generic/test-symvers-parse-error.sh23
-rwxr-xr-xtools/objtool/tests/generic/test-thinlto-ambiguity.sh77
-rwxr-xr-xtools/objtool/tests/generic/test-thinlto-local.sh48
-rwxr-xr-xtools/objtool/tests/generic/test-ubsan-noise.sh48
-rw-r--r--tools/objtool/tests/lib.sh911
-rwxr-xr-xtools/objtool/tests/run-tests.sh239
-rw-r--r--tools/objtool/tests/x86/fixtures/alt_annotate.c57
-rw-r--r--tools/objtool/tests/x86/fixtures/checksum_alt.c66
-rw-r--r--tools/objtool/tests/x86/fixtures/empty_alternative.c77
-rw-r--r--tools/objtool/tests/x86/fixtures/kcfi.c39
-rw-r--r--tools/objtool/tests/x86/fixtures/special_sections.c77
-rw-r--r--tools/objtool/tests/x86/fixtures/static_call_no_key.c32
-rwxr-xr-xtools/objtool/tests/x86/test-alt-annotation.sh38
-rwxr-xr-xtools/objtool/tests/x86/test-checksum-alt.sh45
-rwxr-xr-xtools/objtool/tests/x86/test-empty-alternative.sh31
-rwxr-xr-xtools/objtool/tests/x86/test-kcfi.sh39
-rwxr-xr-xtools/objtool/tests/x86/test-manual-klp-static-call.sh40
-rwxr-xr-xtools/objtool/tests/x86/test-special-sections.sh42
-rw-r--r--tools/perf/trace/beauty/include/uapi/linux/sched.h2
-rw-r--r--tools/testing/selftests/timers/clocksource-switch.c23
-rw-r--r--tools/testing/selftests/timers/posix_timers.c7
-rw-r--r--tools/testing/selftests/timers/raw_skew.c12
-rw-r--r--tools/testing/selftests/x86/Makefile5
-rw-r--r--tools/testing/selftests/x86/sigframe_fpu_portability.c235
-rw-r--r--tools/testing/selftests/x86/xstate.c12
-rw-r--r--tools/testing/selftests/x86/xstate.h20
342 files changed, 10161 insertions, 2240 deletions
diff --git a/Documentation/ABI/testing/sysfs-devices-system-cpu b/Documentation/ABI/testing/sysfs-devices-system-cpu
index 73c8c204820d..545687ebadcc 100644
--- a/Documentation/ABI/testing/sysfs-devices-system-cpu
+++ b/Documentation/ABI/testing/sysfs-devices-system-cpu
@@ -693,15 +693,21 @@ Description: Umwait control
Low order two bits must be zero.
What: /sys/devices/system/cpu/sev
+ /sys/devices/system/cpu/sev/sev_status
/sys/devices/system/cpu/sev/vmpl
Date: May 2024
Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org>
Description: Secure Encrypted Virtualization (SEV) information
- This directory is only present when running as an SEV-SNP guest.
+ This directory is only present when running as an SEV guest.
+
+ sev_status: Reports the value of the SEV_STATUS MSR which
+ enumerates the enabled features of an SEV
+ environment.
vmpl: Reports the Virtual Machine Privilege Level (VMPL) at which
- the SEV-SNP guest is running.
+ the SEV-SNP guest is running. This file is only present
+ when running as an SEV-SNP guest.
What: /sys/devices/system/cpu/svm
@@ -810,3 +816,17 @@ Date: Nov 2022
Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org>
Description:
(RO) the list of CPUs that can be brought online.
+
+What: /sys/devices/system/cpu/preferred
+Date: Sep 2026
+Contact: Linux kernel mailing list <linux-kernel@vger.kernel.org>
+Description:
+ (RO) the list of preferred CPUs applicable in
+ paravirtualized environments.
+
+ The steal governor driver dynamically adjusts this mask
+ based on observed steal time. Scheduling tasks on
+ CPUs outside of this list may lead to performance
+ degradations due to underlying physical CPU contention.
+
+ See Documentation/scheduler/sched-paravirt.rst for more details.
diff --git a/Documentation/ABI/testing/sysfs-platform-ts5500 b/Documentation/ABI/testing/sysfs-platform-ts5500
deleted file mode 100644
index e685957caa12..000000000000
--- a/Documentation/ABI/testing/sysfs-platform-ts5500
+++ /dev/null
@@ -1,54 +0,0 @@
-What: /sys/devices/platform/ts5500/adc
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Indicates the presence of an A/D Converter. If it is present,
- it will display "1", otherwise "0".
-
-What: /sys/devices/platform/ts5500/ereset
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Indicates the presence of an external reset. If it is present,
- it will display "1", otherwise "0".
-
-What: /sys/devices/platform/ts5500/id
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Product ID of the TS board. TS-5500 ID is 0x60.
-
-What: /sys/devices/platform/ts5500/jumpers
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Bitfield showing the jumpers' state. If a jumper is present,
- the corresponding bit is set. For instance, 0x0e means jumpers
- 2, 3 and 4 are set.
-
-What: /sys/devices/platform/ts5500/name
-Date: July 2014
-KernelVersion: 3.16
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Model name of the TS board, e.g. "TS-5500".
-
-What: /sys/devices/platform/ts5500/rs485
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Indicates the presence of the RS485 option. If it is present,
- it will display "1", otherwise "0".
-
-What: /sys/devices/platform/ts5500/sram
-Date: January 2013
-KernelVersion: 3.7
-Contact: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-Description:
- Indicates the presence of the SRAM option. If it is present,
- it will display "1", otherwise "0".
diff --git a/Documentation/arch/arm64/silicon-errata.rst b/Documentation/arch/arm64/silicon-errata.rst
index ac3248b9f2f3..25afa79266a4 100644
--- a/Documentation/arch/arm64/silicon-errata.rst
+++ b/Documentation/arch/arm64/silicon-errata.rst
@@ -53,6 +53,9 @@ stable kernels.
| Allwinner | A64/R18 | UNKNOWN1 | SUN50I_ERRATUM_UNKNOWN1 |
+----------------+-----------------+-----------------+-----------------------------+
+----------------+-----------------+-----------------+-----------------------------+
+| Altera | SoCFPGA Agilex5 | 2.1.23 |ALTERA_ERRATUM_AGILEX5_2_1_23|
++----------------+-----------------+-----------------+-----------------------------+
++----------------+-----------------+-----------------+-----------------------------+
| Ampere | AmpereOne | AC03_CPU_38 | AMPERE_ERRATUM_AC03_CPU_38 |
+----------------+-----------------+-----------------+-----------------------------+
| Ampere | AmpereOne | AC03_CPU_57 | N/A |
diff --git a/Documentation/arch/x86/tdx.rst b/Documentation/arch/x86/tdx.rst
index 3303499ad4c6..a36ea2bd4301 100644
--- a/Documentation/arch/x86/tdx.rst
+++ b/Documentation/arch/x86/tdx.rst
@@ -200,6 +200,27 @@ reflects the TCB of the currently running TDX module and therefore
changes after an update. By contrast, TEE_TCB_SVN reflects the TCB at TD
launch time and is not affected.
+Dynamic PAMT
+------------
+
+The Physical Address Metadata Table (PAMT) is metadata in which the TDX
+module keeps data about each physical page (think struct page). Space
+for it is allocated by the VMM, consumes up to about 0.4% of system
+memory and needs to be supplied to the TDX module when the TDX module is
+first loaded.
+
+Dynamic PAMT is an add-on feature that allows a VMM to dynamically
+allocate the part of the PAMT which tracks 4KB pages. This reduces the
+amount of memory that TDX consumes while TDs are not in use.
+
+When Dynamic PAMT is in use, dmesg shows it like::
+
+ [..] virt/tdx: Enable Dynamic PAMT
+ [..] virt/tdx: 10092 KB allocated for PAMT
+ [..] virt/tdx: TDX-Module initialized
+
+Dynamic PAMT is enabled automatically if supported.
+
TDX Interaction to Other Kernel Components
------------------------------------------
diff --git a/Documentation/arch/x86/xstate.rst b/Documentation/arch/x86/xstate.rst
index cec05ac464c1..e2944f744255 100644
--- a/Documentation/arch/x86/xstate.rst
+++ b/Documentation/arch/x86/xstate.rst
@@ -172,3 +172,57 @@ are extended to control the guest permission:
Note that some VMMs may have already established a set of supported state
components. These options are not presumed to support any particular VMM.
+
+Signal Frame Layout and Portability
+-----------------------------------
+
+The signal frame is designed to be self-describing and portable. This is
+especially important for checkpoint/restore tools like CRIU, which may restore
+a process on a different host than where it was checkpointed. A signal frame
+created on a machine with fewer CPU features can be successfully restored on a
+machine with more CPU features, but not vice-versa.
+
+Signal Frame Software Reserved Bytes
+^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+
+On CPUs supporting XSAVE, bytes 464..511 in the 512-byte FXSAVE/FXRSTOR frame
+are reserved for software use and contain ``struct _fpx_sw_bytes`` (defined in
+``<uapi/asm/sigcontext.h>``)::
+
+ struct _fpx_sw_bytes {
+ __u32 magic1;
+ __u32 extended_size;
+ __u64 xfeatures;
+ __u32 xstate_size;
+ __u32 padding[7];
+ };
+
+- ``magic1``: Set to ``FP_XSTATE_MAGIC1`` (``0x46505853U``) if an extended
+ xstate context is present; 0 for a legacy frame.
+- ``extended_size``: The total size allocated on the stack for the frame,
+ measured from the ``fpstate`` pointer. In 32-bit signal frames, this also
+ includes the 112-byte legacy FPU state prefix of ``struct _fpstate_32``.
+- ``xfeatures``: The mask of xstate features saved in the frame.
+- ``xstate_size``: The actual size of the xstate context for the enabled
+ features (including the 512-byte FXSAVE area and the 64-byte XSAVE header).
+
+The kernel uses ``xstate_size`` in conjunction with the pointer to the xstate
+context to locate the ``FP_XSTATE_MAGIC2`` (``0x46505845U``) marker right after
+the xstate context (at ``xstate_context + xstate_size``). In 64-bit signal frames,
+the ``fpstate`` pointer points directly to the xstate context. In 32-bit signal
+frames (including 32-bit compat tasks on 64-bit kernels), the ``fpstate``
+pointer points to ``struct _fpstate_32``, which contains the 112-byte legacy
+FPU state followed by the 512-byte FXSR state (and any extended xstate). Since
+there is no standalone UAPI structure defined for just the 112-byte legacy
+state, the xstate context starts at ``fpstate + 112`` (and ``extended_size``
+spans the entire allocation from ``fpstate``).
+
+Portability Constraints
+^^^^^^^^^^^^^^^^^^^^^^^
+
+Signal frame portability is constrained by the architectural XSAVE layout.
+Restoration is supported only if the destination host supports all features
+present in the frame and uses matching component offsets and sizes for them.
+While layout compatibility is generally maintained across CPUs from the same
+vendor, differences can occur across vendors or if the XSAVE space of a
+deprecated feature (e.g. MPX) is repurposed for a newer feature (e.g. APX).
diff --git a/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml b/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml
index 518fbf6b2761..be4dc95e3916 100644
--- a/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml
+++ b/Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml
@@ -59,6 +59,7 @@ properties:
- qcom,sm8650-pdc
- qcom,sm8750-pdc
- qcom,x1e80100-pdc
+ - qcom,x1p42100-pdc
- const: qcom,pdc
reg:
diff --git a/Documentation/driver-api/index.rst b/Documentation/driver-api/index.rst
index 6601a258690f..26b7638a327d 100644
--- a/Documentation/driver-api/index.rst
+++ b/Documentation/driver-api/index.rst
@@ -139,6 +139,7 @@ Subsystem-specific APIs
sm501
soundwire/index
spi
+ steal-governor
surface_aggregator/index
switchtec
sync_file
diff --git a/Documentation/driver-api/steal-governor.rst b/Documentation/driver-api/steal-governor.rst
new file mode 100644
index 000000000000..3817eedb38d7
--- /dev/null
+++ b/Documentation/driver-api/steal-governor.rst
@@ -0,0 +1,151 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+Steal Governor
+==============
+
+:Author: Shrikanth Hegde <sshegde@linux.ibm.com>
+
+Introduction
+============
+
+The steal governor is aimed at mitigating the Noisy Neighbour problem
+which occurs in paravirtualized environments with CPU overcommit.
+The performance of a workload running in one VM gets degraded by
+the activity of other VMs on the same host. As a result, all VMs
+collectively make slower forward progress.
+
+In such systems, high utilization in all VMs causes the hypervisor to
+frequently preempt vCPUs. This vCPU preemption is expensive.
+To mitigate this, the kernel aims to restrict workloads to a subset of
+Preferred CPUs to reduce physical CPU contention.
+A detailed explanation of Preferred CPUs is available in
+``Documentation/scheduler/sched-paravirt.rst``.
+
+The steal governor selects ``CONFIG_PREFERRED_CPU=y`` which enables the
+scheduler core infrastructure to move the tasks to Preferred CPUs where
+possible. The driver controls the policy decisions regarding the state of
+preferred CPUs. That is, this driver decides which CPUs are preferred
+and which CPUs are non-preferred.
+
+The driver code is available at ``drivers/virt/steal_governor.c``.
+
+Core idea
+=========
+
+steal time is an indication available today in Guest which shows contention
+for underlying physical CPU. Use it as a hint in the guest to fold the
+workload to a reduced set of vCPUs. When there is contention, steal time
+will show up in all the guests. When each guest honors the hint and folds
+the workload to a smaller set of vCPUs (Preferred CPUs), it reduces the
+contention and thereby reduces vCPU preemption.
+This is achieved without any cross-guest communication.
+
+Steal governor driver effectively does:
+
+1. Periodically computes the steal ratio using accumulated steal time
+ across possible CPUs, normalized by the number of active CPUs.
+
+2. If steal ratio is greater than high threshold, reduce the number of
+ preferred CPUs by 1 core. Ensure at least one core is left always.
+ Skip changing the state of offline CPUs in that core.
+
+3. If steal ratio is less than or equal to low threshold, increase the
+ number of preferred CPUs by 1 core. If preferred is same as active,
+ nothing to be done. Skip changing the state of offline CPUs.
+ This helps to handle cases where few CPUs are offline in a core and
+ those offline CPUs will not be marked as preferred.
+
+4. Ensure preferred CPUs is always subset of active CPUs.
+ On feature disable it is same as active CPUs.
+
+This feature works best only when all the VMs enable the feature as
+it is a co-operative scheme. If a specific VM doesn't enable this feature
+it may end up with more CPUs than others, still should lead to better
+performance when seen from system view. Those who enable this driver must
+ensure it is enabled in all VMs.
+
+Note that this driver is strictly intended for actual guests; for example,
+loading this module in a privileged VM like Xen Dom0 is blocked.
+
+Workload considerations
+=======================
+
+The steal governor is useful for workloads where vCPU preemption has
+costs beyond the lost CPU time, such as lock-holder preemption, critical
+sections, communicating threads, and cache or TLB disruption.
+
+Pure CPU-time workloads with independent workers may not benefit and
+could see a small regression due to additional guest scheduling overhead.
+
+Module Parameters
+=================
+
+interval_ms
+-----------
+
+How often steal governor checks for steal time.
+Default: 1000 i.e. 1 second. Value should be in between 100ms to 100sec.
+
+This controls how fast steal governor driver reacts to changes to the
+contention of physical CPUs. Since it does a fair amount of work, setting
+too low may have overhead. Setting it too high might render it ineffective.
+
+low_threshold
+-------------
+
+lower threshold value in percentage * 100.
+Default: 200, i.e. 2% steal is considered as low threshold.
+Can't be higher than high_threshold.
+
+This determines what values should be considered as nil/no steal values.
+When steal governor sees steal ratio is less than or equal to this value,
+it will increase the preferred CPUs by 1 core.
+Using zero might cause oscillations.
+
+high_threshold
+--------------
+
+higher threshold value in percentage * 100
+Default: 500, i.e. 5% steal is considered as high threshold.
+Can't be lower than low_threshold. Must be less than 10000.
+
+This determines what values should be considered as high steal values.
+When steal governor sees steal ratio is higher than this value, it will
+reduce the preferred CPUs by 1 core.
+
+Limitations of default values
+-----------------------------
+
+Because of the vast diversity in VM configurations and different
+architectures, the default thresholds may not be optimal for all systems.
+Users may need to tune these parameters based on the system under
+test to achieve the best results.
+
+The governor sums the steal time across all possible CPUs, which ensures
+the accumulated steal time remains a monotonically increasing value.
+However, to calculate the effective steal ratio, it divides this sum
+by the number of active CPUs. Because only active CPUs contribute to
+the steal time delta, this prevents threshold dilution on sparsely
+populated systems.
+
+The driver reduces/increases preferred CPUs by core-level. This could provide
+faster convergence for hypervisors such as powerVM. But on KVM and Xen
+convergence could be slower depending on the configuration.
+Using a smaller interval_ms could help one to expedite it.
+
+Reasons for CONFIG_STEAL_GOVERNOR=m
+===================================
+
+Selecting this driver makes CONFIG_PREFERRED_CPU=y. That makes configs
+driven by user preference. Though one can have CONFIG_STEAL_GOVERNOR=y,
+It is recommended to build CONFIG_STEAL_GOVERNOR=m due to below reasons:
+
+1. Doing periodic work has additional overheads. Enabling this driver
+ in systems where steal time cannot happen is of no use. There is no
+ benefit with additional overheads in such systems.
+
+2. This works well when all VMs work in a co-operative manner. When an
+ administrative user enables it in one VM, he/she will likely enable
+ it in all VMs.
+
+3. User can tweak the module parameters by reloading the module.
diff --git a/Documentation/filesystems/resctrl.rst b/Documentation/filesystems/resctrl.rst
index e4b66af55ffb..b52795e03303 100644
--- a/Documentation/filesystems/resctrl.rst
+++ b/Documentation/filesystems/resctrl.rst
@@ -236,12 +236,11 @@ with respect to allocation:
user can request.
"bandwidth_gran":
- The granularity in which the memory bandwidth
- percentage is allocated. The allocated
- b/w percentage is rounded off to the next
- control step available on the hardware. The
- available bandwidth control steps are:
- min_bandwidth + N * bandwidth_gran.
+ The approximate granularity in which the memory bandwidth
+ percentage is allocated. The allocated bandwidth percentage is
+ rounded up or down to the closest control step available on the
+ hardware. The available hardware steps are no larger than this
+ value.
"delay_linear":
Indicates if the delay scale is linear or
@@ -643,7 +642,7 @@ When monitoring is enabled all MON groups will also contain:
during execution of instructions summed across all logical CPUs on a
package for the current monitoring group.
- "activity" also reports a floating point value (in Farads). This provides
+ "activity" also reports a floating point value (in nanofarads). This provides
an estimate of work done independent of the frequency that the CPUs used
for execution.
@@ -881,8 +880,10 @@ The minimum bandwidth percentage value for each cpu model is predefined
and can be looked up through "info/MB/min_bandwidth". The bandwidth
granularity that is allocated is also dependent on the cpu model and can
be looked up at "info/MB/bandwidth_gran". The available bandwidth
-control steps are: min_bw + N * bw_gran. Intermediate values are rounded
-to the next control step available on the hardware.
+control steps are, approximately, min_bw + N * bw_gran. The steps may
+appear irregular due to rounding to an exact percentage: bw_gran is the
+maximum interval between the percentage values corresponding to any two
+adjacent steps in the hardware.
The bandwidth throttling is a core specific mechanism on some of Intel
SKUs. Using a high bandwidth and a low bandwidth setting on two threads
diff --git a/Documentation/scheduler/index.rst b/Documentation/scheduler/index.rst
index d6d75421756a..791178ac8bec 100644
--- a/Documentation/scheduler/index.rst
+++ b/Documentation/scheduler/index.rst
@@ -23,6 +23,7 @@ Scheduler
sched-stats
sched-ext
sched-debug
+ sched-paravirt
sched-preemption
text_files
diff --git a/Documentation/scheduler/sched-paravirt.rst b/Documentation/scheduler/sched-paravirt.rst
new file mode 100644
index 000000000000..311cdbf2722b
--- /dev/null
+++ b/Documentation/scheduler/sched-paravirt.rst
@@ -0,0 +1,67 @@
+.. SPDX-License-Identifier: GPL-2.0
+.. _sched-paravirt:
+
+Preferred CPUs
+==============
+
+In paravirtualized environments CPU overcommit is a common scenario.
+i.e. the sum of virtual CPUs (vCPUs) of all VMs is greater than number of
+physical CPUs (pCPUs). Under such conditions when all or many VMs have
+high utilization, hypervisor won't be able to satisfy the CPU requirement
+and has to context switch within or across VMs. The hypervisor needs to
+preempt one vCPU to run another. This is called vCPU preemption.
+This is more expensive compared to task context switch within a vCPU, since
+hypervisor lacks vCPU context and could preempt a critical section which
+slows forward progress.
+
+In such cases it is better that combined vCPU demand from all VMs is reduced
+by not using some of the vCPUs in each VM. vCPUs where workload can be safely
+scheduled which won't increase any contention for pCPU are called
+"Preferred CPUs".
+
+One of the main design constructs is that preferred CPUs are always
+a subset of active CPUs. In most cases preferred CPUs will be same as
+active CPUs. When there is pCPU contention, Preferred CPUs will reduce
+based on the steal time. When the pCPU contention goes away as indicated
+by steal time, Preferred CPUs could become same as active CPUs again.
+The policy decisions are to be taken by driver.
+For example, steal_governor. Look at its documentation for more
+details. (``drivers/virt/steal_governor.c``)
+
+Scheduling decisions such as wakeup, pushing the task etc, need this
+CPU state info. This is maintained in ``cpu_preferred_mask``.
+vCPUs which are not in ``cpu_preferred_mask`` should be treated as vCPUs which
+should not be used at this moment provided it doesn't break user affinity.
+
+This is achieved by:
+
+1. Selecting a preferred CPU at wakeup using fallback mechanism.
+2. Pushing the task away from non-preferred CPU at tick.
+3. Selecting only preferred CPUs for load balance.
+
+``/sys/devices/system/cpu/preferred`` prints the current ``cpu_preferred_mask``
+in cpulist format.
+
+Notes:
+
+1. This feature is available under ``CONFIG_PREFERRED_CPU``. Driver which
+ makes decisions should enable it. For example, steal_governor driver
+ (``CONFIG_STEAL_GOVERNOR``). On enabling the driver, CPU preferred state
+ can change based on steal time. Without the driver, preferred CPUs is
+ same as active CPUs.
+
+2. This feature works for the FAIR class only.
+
+3. A pinned task, which can't be moved to preferred CPUs will continue
+ to run based on its affinity. But no load balancing happens if it is affined
+ only on non-preferred CPUs.
+
+4. Decision to change the preferred CPU state is driven by the kernel.
+ Hence it shouldn't break user affinities. One of the main reasons why
+ CPU hotplug or Isolated cpuset partitions was not a solution.
+
+5. This feature works best only when all the Guest VMs enable the feature as
+ it is a co-operative scheme. If a specific VM doesn't enable this feature
+ it may end up with more CPUs than others, still should lead to better
+ performance when seen from system view.
+ Users who enable this driver must ensure it is enabled in all Guest VMs.
diff --git a/MAINTAINERS b/MAINTAINERS
index 11e412d802b1..d4a65bd29a4a 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -26514,6 +26514,15 @@ F: rust/helpers/jump_label.c
F: rust/kernel/generated_arch_static_branch_asm.rs.S
F: rust/kernel/jump_label.rs
+STEAL GOVERNOR DRIVER
+M: Shrikanth Hegde <sshegde@linux.ibm.com>
+R: Yury Norov <yury.norov@gmail.com>
+L: linux-kernel@vger.kernel.org
+S: Maintained
+T: git git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git sched/core
+F: Documentation/driver-api/steal-governor.rst
+F: drivers/virt/steal_governor.c
+
STI AUDIO (ASoC) DRIVERS
M: Arnaud Pouliquen <arnaud.pouliquen@foss.st.com>
L: linux-sound@vger.kernel.org
@@ -27104,11 +27113,6 @@ S: Maintained
F: Documentation/process/contribution-maturity-model.rst
F: Documentation/process/researcher-guidelines.rst
-TECHNOLOGIC SYSTEMS TS-5500 PLATFORM SUPPORT
-M: "Savoir-faire Linux Inc." <kernel@savoirfairelinux.com>
-S: Maintained
-F: arch/x86/platform/ts5500/
-
TECHNOTREND USB IR RECEIVER
M: Sean Young <sean@mess.org>
L: linux-media@vger.kernel.org
diff --git a/arch/Kconfig b/arch/Kconfig
index a438eda84cd4..1163736de1f9 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -1696,44 +1696,6 @@ config HAVE_STATIC_CALL_INLINE
depends on HAVE_STATIC_CALL
select OBJTOOL
-config HAVE_PREEMPT_DYNAMIC
- bool
-
-config HAVE_PREEMPT_DYNAMIC_CALL
- bool
- depends on HAVE_STATIC_CALL
- select HAVE_PREEMPT_DYNAMIC
- help
- An architecture should select this if it can handle the preemption
- model being selected at boot time using static calls.
-
- Where an architecture selects HAVE_STATIC_CALL_INLINE, any call to a
- preemption function will be patched directly.
-
- Where an architecture does not select HAVE_STATIC_CALL_INLINE, any
- call to a preemption function will go through a trampoline, and the
- trampoline will be patched.
-
- It is strongly advised to support inline static call to avoid any
- overhead.
-
-config HAVE_PREEMPT_DYNAMIC_KEY
- bool
- depends on HAVE_ARCH_JUMP_LABEL
- select HAVE_PREEMPT_DYNAMIC
- help
- An architecture should select this if it can handle the preemption
- model being selected at boot time using static keys.
-
- Each preemption function will be given an early return based on a
- static key. This should have slightly lower overhead than non-inline
- static calls, as this effectively inlines each trampoline into the
- start of its callee. This may avoid redundant work, and may
- integrate better with CFI schemes.
-
- This will have greater overhead than using inline static calls as
- the call to the preemption function cannot be entirely elided.
-
config ARCH_WANT_LD_ORPHAN_WARN
bool
help
diff --git a/arch/arm/kernel/perf_regs.c b/arch/arm/kernel/perf_regs.c
index 0529f90395c9..838d701adf4d 100644
--- a/arch/arm/kernel/perf_regs.c
+++ b/arch/arm/kernel/perf_regs.c
@@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1ULL << PERF_REG_ARM_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
@@ -31,9 +31,3 @@ u64 perf_reg_abi(struct task_struct *task)
return PERF_SAMPLE_REGS_ABI_32;
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig
index 88de433a78ef..1ec35aada772 100644
--- a/arch/arm64/Kconfig
+++ b/arch/arm64/Kconfig
@@ -217,7 +217,6 @@ config ARM64
select HAVE_PERF_EVENTS_NMI if ARM64_PSEUDO_NMI
select HAVE_PERF_REGS
select HAVE_PERF_USER_STACK_DUMP
- select HAVE_PREEMPT_DYNAMIC_KEY
select HAVE_REGS_AND_STACK_ACCESS_API
select HAVE_RELIABLE_STACKTRACE
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
@@ -1422,6 +1421,15 @@ config NVIDIA_OLYMPUS_1027_ERRATUM
If unsure, say Y.
+config ALTERA_ERRATUM_AGILEX5_2_1_23
+ bool
+ help
+ The Altera SoCFPGA Agilex5 GIC600 SoC integration has ACE-lite
+ addressing limited to the first 32bit of physical address space,
+ so the ITS cannot access memory above 4GB.
+
+ Selected by ARCH_INTEL_SOCFPGA, as all Agilex5 devices are affected.
+
config RENESAS_ERRATUM_GEN4GICITS1
bool "Renesas R-Car Gen4: GIC600 can not access physical addresses above 4 GiB"
default y
diff --git a/arch/arm64/Kconfig.platforms b/arch/arm64/Kconfig.platforms
index 962e67fcb5fb..c52dc955df5f 100644
--- a/arch/arm64/Kconfig.platforms
+++ b/arch/arm64/Kconfig.platforms
@@ -370,6 +370,7 @@ config ARCH_SEATTLE
config ARCH_INTEL_SOCFPGA
bool "Intel's SoCFPGA ARMv8 Families"
+ select ALTERA_ERRATUM_AGILEX5_2_1_23
help
This enables support for Intel's SoCFPGA ARMv8 families:
Stratix 10 (ex. Altera), Stratix10 Software Virtual Platform,
diff --git a/arch/arm64/boot/dts/qcom/purwa.dtsi b/arch/arm64/boot/dts/qcom/purwa.dtsi
index c698e6cb2543..4348dd3d1dc5 100644
--- a/arch/arm64/boot/dts/qcom/purwa.dtsi
+++ b/arch/arm64/boot/dts/qcom/purwa.dtsi
@@ -224,6 +224,11 @@
compatible = "qcom,x1p42100-qmp-gen4x4-pcie-phy";
};
+/* X1P42100 PDC is same as X1E80100, but without hardware register bug */
+&pdc {
+ compatible = "qcom,x1p42100-pdc", "qcom,pdc";
+};
+
&qfprom {
gpu_speed_bin: gpu-speed-bin@119 {
reg = <0x119 0x2>;
diff --git a/arch/arm64/configs/defconfig b/arch/arm64/configs/defconfig
index cbb43e5890bc..fce418fe6ff6 100644
--- a/arch/arm64/configs/defconfig
+++ b/arch/arm64/configs/defconfig
@@ -1638,8 +1638,6 @@ CONFIG_PWM_VISCONTI=m
CONFIG_PWM_XILINX=m
CONFIG_SL28CPLD_INTC=y
CONFIG_XILINX_INTC=y
-CONFIG_QCOM_PDC=y
-CONFIG_QCOM_MPM=y
CONFIG_TI_SCI_INTR_IRQCHIP=m
CONFIG_TI_SCI_INTA_IRQCHIP=m
CONFIG_RESET_GPIO=m
diff --git a/arch/arm64/include/asm/preempt.h b/arch/arm64/include/asm/preempt.h
index 3c85cacc1d19..75a592b7f47f 100644
--- a/arch/arm64/include/asm/preempt.h
+++ b/arch/arm64/include/asm/preempt.h
@@ -97,19 +97,9 @@ static inline bool __preempt_count_dec_and_test(void)
void preempt_schedule(void);
void preempt_schedule_notrace(void);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-
-void dynamic_preempt_schedule(void);
-#define __preempt_schedule() dynamic_preempt_schedule()
-void dynamic_preempt_schedule_notrace(void);
-#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace()
-
-#else /* CONFIG_PREEMPT_DYNAMIC */
-
#define __preempt_schedule() preempt_schedule()
#define __preempt_schedule_notrace() preempt_schedule_notrace()
-#endif /* CONFIG_PREEMPT_DYNAMIC */
#endif /* CONFIG_PREEMPTION */
#endif /* __ASM_PREEMPT_H */
diff --git a/arch/arm64/kernel/paravirt.c b/arch/arm64/kernel/paravirt.c
index 572efb96b23f..30bf61d031eb 100644
--- a/arch/arm64/kernel/paravirt.c
+++ b/arch/arm64/kernel/paravirt.c
@@ -157,9 +157,9 @@ int __init pv_time_init(void)
static_call_update(pv_steal_clock, para_steal_clock);
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
pr_info("using stolen time PV\n");
diff --git a/arch/arm64/kernel/perf_regs.c b/arch/arm64/kernel/perf_regs.c
index b4eece3eb17d..71a3e0238de4 100644
--- a/arch/arm64/kernel/perf_regs.c
+++ b/arch/arm64/kernel/perf_regs.c
@@ -77,7 +77,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1ULL << PERF_REG_ARM64_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
u64 reserved_mask = REG_RESERVED;
@@ -98,9 +98,3 @@ u64 perf_reg_abi(struct task_struct *task)
return PERF_SAMPLE_REGS_ABI_64;
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/csky/kernel/perf_regs.c b/arch/csky/kernel/perf_regs.c
index 09b7f88a2d6a..c932a96afc56 100644
--- a/arch/csky/kernel/perf_regs.c
+++ b/arch/csky/kernel/perf_regs.c
@@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1ULL << PERF_REG_CSKY_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
@@ -31,9 +31,3 @@ u64 perf_reg_abi(struct task_struct *task)
return PERF_SAMPLE_REGS_ABI_32;
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig
index 0bd8503fd5c4..0aad0004b010 100644
--- a/arch/loongarch/Kconfig
+++ b/arch/loongarch/Kconfig
@@ -169,7 +169,6 @@ config LOONGARCH
select HAVE_PERF_REGS
select HAVE_PERF_USER_STACK_DUMP
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
- select HAVE_PREEMPT_DYNAMIC_KEY
select HAVE_REGS_AND_STACK_ACCESS_API
select HAVE_RELIABLE_STACKTRACE if UNWINDER_ORC
select HAVE_RETHOOK
diff --git a/arch/loongarch/kernel/paravirt.c b/arch/loongarch/kernel/paravirt.c
index 10821cce554c..e8965a3f8082 100644
--- a/arch/loongarch/kernel/paravirt.c
+++ b/arch/loongarch/kernel/paravirt.c
@@ -308,10 +308,10 @@ int __init pv_time_init(void)
static_call_update(pv_steal_clock, paravt_steal_clock);
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
#endif
if (static_key_enabled(&virt_preempt_key))
diff --git a/arch/loongarch/kernel/perf_regs.c b/arch/loongarch/kernel/perf_regs.c
index 263ac4ab5af6..164514f40ae0 100644
--- a/arch/loongarch/kernel/perf_regs.c
+++ b/arch/loongarch/kernel/perf_regs.c
@@ -25,7 +25,7 @@ u64 perf_reg_abi(struct task_struct *tsk)
}
#endif /* CONFIG_32BIT */
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask)
return -EINVAL;
@@ -45,9 +45,3 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
return regs->regs[idx];
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/mips/kernel/perf_regs.c b/arch/mips/kernel/perf_regs.c
index e686780d1647..00a5201dbd5d 100644
--- a/arch/mips/kernel/perf_regs.c
+++ b/arch/mips/kernel/perf_regs.c
@@ -28,7 +28,7 @@ u64 perf_reg_abi(struct task_struct *tsk)
}
#endif /* CONFIG_32BIT */
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask)
return -EINVAL;
@@ -60,9 +60,3 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
return (s64)v; /* Sign extend if 32-bit. */
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/parisc/kernel/perf_regs.c b/arch/parisc/kernel/perf_regs.c
index 10a1a5f06a18..4f21aab5405c 100644
--- a/arch/parisc/kernel/perf_regs.c
+++ b/arch/parisc/kernel/perf_regs.c
@@ -34,7 +34,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1ULL << PERF_REG_PARISC_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
@@ -53,9 +53,3 @@ u64 perf_reg_abi(struct task_struct *task)
return PERF_SAMPLE_REGS_ABI_64;
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig
index 3f59b201b62f..c3b6cd655606 100644
--- a/arch/powerpc/Kconfig
+++ b/arch/powerpc/Kconfig
@@ -278,7 +278,6 @@ config PPC
select HAVE_PERF_EVENTS_NMI if PPC64
select HAVE_PERF_REGS
select HAVE_PERF_USER_STACK_DUMP
- select HAVE_PREEMPT_DYNAMIC_KEY
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
select HAVE_RETHOOK if KPROBES
select HAVE_REGS_AND_STACK_ACCESS_API
diff --git a/arch/powerpc/perf/perf_regs.c b/arch/powerpc/perf/perf_regs.c
index 350dccb0143c..a01d8a903640 100644
--- a/arch/powerpc/perf/perf_regs.c
+++ b/arch/powerpc/perf/perf_regs.c
@@ -125,7 +125,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
return regs_get_register(regs, pt_regs_offset[idx]);
}
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
diff --git a/arch/powerpc/platforms/pseries/setup.c b/arch/powerpc/platforms/pseries/setup.c
index f29e7547c995..c181df4dec70 100644
--- a/arch/powerpc/platforms/pseries/setup.c
+++ b/arch/powerpc/platforms/pseries/setup.c
@@ -876,9 +876,9 @@ static void __init pSeries_setup_arch(void)
static_branch_enable(&shared_processor);
pv_spinlocks_init();
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
#endif
}
diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig
index f409f264d8c4..0ccb72101d28 100644
--- a/arch/riscv/Kconfig
+++ b/arch/riscv/Kconfig
@@ -194,7 +194,6 @@ config RISCV
select HAVE_PERF_REGS
select HAVE_PERF_USER_STACK_DUMP
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
- select HAVE_PREEMPT_DYNAMIC_KEY
select HAVE_REGS_AND_STACK_ACCESS_API
select HAVE_RETHOOK
select HAVE_RSEQ
diff --git a/arch/riscv/include/asm/smp.h b/arch/riscv/include/asm/smp.h
index 0ecc67641b09..bed39fff1f8a 100644
--- a/arch/riscv/include/asm/smp.h
+++ b/arch/riscv/include/asm/smp.h
@@ -15,6 +15,18 @@
struct seq_file;
extern unsigned long boot_cpu_hartid;
+enum ipi_message_type {
+ IPI_RESCHEDULE,
+ IPI_CALL_FUNC,
+ IPI_CPU_STOP,
+ IPI_CPU_CRASH_STOP,
+ IPI_IRQ_WORK,
+ IPI_TIMER,
+ IPI_CPU_BACKTRACE,
+ IPI_KGDB_ROUNDUP,
+ IPI_MAX
+};
+
#ifdef CONFIG_SMP
#include <linux/jump_label.h>
diff --git a/arch/riscv/kernel/paravirt.c b/arch/riscv/kernel/paravirt.c
index 5f56be79cd06..9c13a6f1ea2a 100644
--- a/arch/riscv/kernel/paravirt.c
+++ b/arch/riscv/kernel/paravirt.c
@@ -116,9 +116,9 @@ int __init pv_time_init(void)
static_call_update(pv_steal_clock, pv_time_steal_clock);
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
pr_info("Computing paravirt steal-time\n");
diff --git a/arch/riscv/kernel/perf_regs.c b/arch/riscv/kernel/perf_regs.c
index fd304a248de6..1ecc8760b88b 100644
--- a/arch/riscv/kernel/perf_regs.c
+++ b/arch/riscv/kernel/perf_regs.c
@@ -18,7 +18,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1ULL << PERF_REG_RISCV_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
@@ -35,9 +35,3 @@ u64 perf_reg_abi(struct task_struct *task)
#endif
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
diff --git a/arch/riscv/kernel/sbi-ipi.c b/arch/riscv/kernel/sbi-ipi.c
index 0cc5559c08d8..eeec178a9b95 100644
--- a/arch/riscv/kernel/sbi-ipi.c
+++ b/arch/riscv/kernel/sbi-ipi.c
@@ -57,7 +57,7 @@ void __init sbi_ipi_init(void)
return;
}
- virq = ipi_mux_create(BITS_PER_BYTE, sbi_send_ipi);
+ virq = ipi_mux_create(IPI_MAX, sbi_send_ipi);
if (virq <= 0) {
pr_err("unable to create muxed IPIs\n");
irq_dispose_mapping(sbi_ipi_virq);
@@ -75,7 +75,7 @@ void __init sbi_ipi_init(void)
"irqchip/sbi-ipi:starting",
sbi_ipi_starting_cpu, NULL);
- riscv_ipi_set_virq_range(virq, BITS_PER_BYTE);
+ riscv_ipi_set_virq_range(virq, IPI_MAX);
pr_info("providing IPIs using SBI IPI extension\n");
/*
diff --git a/arch/riscv/kernel/smp.c b/arch/riscv/kernel/smp.c
index fa66f9c97d74..8930b62b15e7 100644
--- a/arch/riscv/kernel/smp.c
+++ b/arch/riscv/kernel/smp.c
@@ -28,18 +28,6 @@
#include <asm/cacheflush.h>
#include <asm/cpu_ops.h>
-enum ipi_message_type {
- IPI_RESCHEDULE,
- IPI_CALL_FUNC,
- IPI_CPU_STOP,
- IPI_CPU_CRASH_STOP,
- IPI_IRQ_WORK,
- IPI_TIMER,
- IPI_CPU_BACKTRACE,
- IPI_KGDB_ROUNDUP,
- IPI_MAX
-};
-
static const char * const ipi_names[] = {
[IPI_RESCHEDULE] = "Rescheduling interrupts",
[IPI_CALL_FUNC] = "Function call interrupts",
diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig
index 55b074e748f8..2b2b51224d5d 100644
--- a/arch/s390/Kconfig
+++ b/arch/s390/Kconfig
@@ -245,7 +245,6 @@ config S390
select HAVE_PERF_REGS
select HAVE_PERF_USER_STACK_DUMP
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
- select HAVE_PREEMPT_DYNAMIC_KEY
select HAVE_REGS_AND_STACK_ACCESS_API
select HAVE_RELIABLE_STACKTRACE
select HAVE_RETHOOK
diff --git a/arch/s390/include/asm/preempt.h b/arch/s390/include/asm/preempt.h
index 5560d5fca2a3..60c6f019ec45 100644
--- a/arch/s390/include/asm/preempt.h
+++ b/arch/s390/include/asm/preempt.h
@@ -155,20 +155,9 @@ static __always_inline int __preempt_count_sub_return(int val)
void preempt_schedule(void);
void preempt_schedule_notrace(void);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-
-void dynamic_preempt_schedule(void);
-void dynamic_preempt_schedule_notrace(void);
-#define __preempt_schedule() dynamic_preempt_schedule()
-#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace()
-
-#else /* CONFIG_PREEMPT_DYNAMIC */
-
#define __preempt_schedule() preempt_schedule()
#define __preempt_schedule_notrace() preempt_schedule_notrace()
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
#endif /* CONFIG_PREEMPTION */
#endif /* __ASM_PREEMPT_H */
diff --git a/arch/s390/kernel/hiperdispatch.c b/arch/s390/kernel/hiperdispatch.c
index 8494823559b1..1397dab8a677 100644
--- a/arch/s390/kernel/hiperdispatch.c
+++ b/arch/s390/kernel/hiperdispatch.c
@@ -207,16 +207,12 @@ static unsigned long hd_calculate_steal_percentage(void)
{
unsigned long time_delta, steal_delta, steal, percentage;
static ktime_t prev;
- int cpus, cpu;
+ int cpus;
ktime_t now;
- cpus = 0;
- steal = 0;
percentage = 0;
- for_each_cpu(cpu, &hd_vmvl_cpumask) {
- steal += kcpustat_cpu(cpu).cpustat[CPUTIME_STEAL];
- cpus++;
- }
+ steal = kcpustat_field_total(CPUTIME_STEAL, &hd_vmvl_cpumask);
+ cpus = cpumask_weight(&hd_vmvl_cpumask);
/*
* If there is no vertical medium and low CPUs steal time
* is 0 as vertical high CPUs shouldn't experience steal time.
diff --git a/arch/s390/kernel/perf_regs.c b/arch/s390/kernel/perf_regs.c
index 7b305f1456f8..6496fd23c540 100644
--- a/arch/s390/kernel/perf_regs.c
+++ b/arch/s390/kernel/perf_regs.c
@@ -34,7 +34,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
#define REG_RESERVED (~((1UL << PERF_REG_S390_MAX) - 1))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
if (!mask || mask & REG_RESERVED)
return -EINVAL;
diff --git a/arch/um/kernel/um_arch.c b/arch/um/kernel/um_arch.c
index 3dbe3acc4833..c3f83e805392 100644
--- a/arch/um/kernel/um_arch.c
+++ b/arch/um/kernel/um_arch.c
@@ -269,12 +269,88 @@ unsigned long brk_start;
#define MIN_VMALLOC (32 * 1024 * 1024)
+static u64 __init read_xcr0(void)
+{
+ u32 a, b, c, d;
+
+ asm volatile("cpuid"
+ : "=a"(a), "=b"(b), "=c"(c), "=d"(d)
+ : "a"(0), "c"(0));
+ if (a >= 1) { /* max_leaf >= 1 */
+ asm volatile("cpuid"
+ : "=a"(a), "=b"(b), "=c"(c), "=d"(d)
+ : "a"(1), "c"(0));
+ if (c & (1 << 27)) { /* XSAVE enabled by OS */
+ asm volatile("xgetbv" : "=d"(d), "=a"(a) : "c"(0));
+ return ((u64)d << 32) | a;
+ }
+ }
+ return 0;
+}
+
+static void __init validate_and_set_cpu_cap(int cap, u64 xcr0)
+{
+ /*
+ * Check for missing xstate features right away, so that there's no
+ * perceived need for all optimized code in the kernel to do so.
+ */
+ switch (cap) {
+ case X86_FEATURE_AVX:
+ case X86_FEATURE_AVX2:
+ case X86_FEATURE_AVX_VNNI:
+ case X86_FEATURE_FMA:
+ case X86_FEATURE_VAES:
+ case X86_FEATURE_VPCLMULQDQ:
+ if ((xcr0 & 0x7) != 0x7) {
+ static bool warned;
+
+ if (!warned) {
+ os_warn("Disabling AVX support due to missing xstate features\n");
+ warned = true;
+ }
+ return;
+ }
+ break;
+ case X86_FEATURE_AVX512F:
+ case X86_FEATURE_AVX512BW:
+ case X86_FEATURE_AVX512CD:
+ case X86_FEATURE_AVX512DQ:
+ case X86_FEATURE_AVX512ER:
+ case X86_FEATURE_AVX512IFMA:
+ case X86_FEATURE_AVX512PF:
+ case X86_FEATURE_AVX512VBMI:
+ case X86_FEATURE_AVX512VL:
+ case X86_FEATURE_AVX512_4FMAPS:
+ case X86_FEATURE_AVX512_4VNNIW:
+ case X86_FEATURE_AVX512_BF16:
+ case X86_FEATURE_AVX512_BITALG:
+ case X86_FEATURE_AVX512_FP16:
+ case X86_FEATURE_AVX512_VBMI2:
+ case X86_FEATURE_AVX512_VNNI:
+ case X86_FEATURE_AVX512_VP2INTERSECT:
+ case X86_FEATURE_AVX512_VPOPCNTDQ:
+ if ((xcr0 & 0xe7) != 0xe7) {
+ static bool warned;
+
+ if (!warned) {
+ os_warn("Disabling AVX-512 support due to missing xstate features\n");
+ warned = true;
+ }
+ return;
+ }
+ break;
+ }
+ set_cpu_cap(&boot_cpu_data, cap);
+}
+
static void __init parse_host_cpu_flags(char *line)
{
+ u64 xcr0 = read_xcr0();
int i;
+
for (i = 0; i < 32*NCAPINTS; i++) {
if ((x86_cap_flags[i] != NULL) && strstr(line, x86_cap_flags[i]))
- set_cpu_cap(&boot_cpu_data, i);
+ validate_and_set_cpu_cap(i, xcr0);
}
}
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index a8781db3be70..58dce0b66179 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -295,7 +295,6 @@ config X86
select HAVE_STACK_VALIDATION if HAVE_OBJTOOL
select HAVE_STATIC_CALL
select HAVE_STATIC_CALL_INLINE if HAVE_OBJTOOL
- select HAVE_PREEMPT_DYNAMIC_CALL
select HAVE_RSEQ
select HAVE_RUST if X86_64
select HAVE_SYSCALL_TRACEPOINTS
@@ -3073,15 +3072,6 @@ config GEOS
help
This option enables system support for the Traverse Technologies GEOS.
-config TS5500
- bool "Technologic Systems TS-5500 platform support"
- depends on MELAN
- select CHECK_SIGNATURE
- select NEW_LEDS
- select LEDS_CLASS
- help
- This option enables system support for the Technologic Systems TS-5500.
-
endif # X86_32
config AMD_NB
diff --git a/arch/x86/boot/compressed/error.c b/arch/x86/boot/compressed/error.c
index 19a8251de506..ce5ed7d8265e 100644
--- a/arch/x86/boot/compressed/error.c
+++ b/arch/x86/boot/compressed/error.c
@@ -22,22 +22,3 @@ void error(char *m)
while (1)
asm("hlt");
}
-
-/* EFI libstub provides vsnprintf() */
-#ifdef CONFIG_EFI_STUB
-void panic(const char *fmt, ...)
-{
- static char buf[1024];
- va_list args;
- int len;
-
- va_start(args, fmt);
- len = vsnprintf(buf, sizeof(buf), fmt, args);
- va_end(args);
-
- if (len && buf[len - 1] == '\n')
- buf[len - 1] = '\0';
-
- error(buf);
-}
-#endif
diff --git a/arch/x86/boot/compressed/error.h b/arch/x86/boot/compressed/error.h
index 31f9e080d61a..87062dea9a20 100644
--- a/arch/x86/boot/compressed/error.h
+++ b/arch/x86/boot/compressed/error.h
@@ -6,6 +6,5 @@
void warn(const char *m);
void error(char *m) __noreturn;
-void panic(const char *fmt, ...) __noreturn __cold;
#endif /* BOOT_COMPRESSED_ERROR_H */
diff --git a/arch/x86/boot/compressed/mem.c b/arch/x86/boot/compressed/mem.c
index 0e9f84ab4bdc..1721af3a8039 100644
--- a/arch/x86/boot/compressed/mem.c
+++ b/arch/x86/boot/compressed/mem.c
@@ -2,48 +2,6 @@
#include "error.h"
#include "misc.h"
-#include "tdx.h"
-#include "sev.h"
-#include <asm/shared/tdx.h>
-
-/*
- * accept_memory() and process_unaccepted_memory() called from EFI stub which
- * runs before decompressor and its early_tdx_detect().
- *
- * Enumerate TDX directly from the early users.
- */
-static bool early_is_tdx_guest(void)
-{
- static bool once;
- static bool is_tdx;
-
- if (!IS_ENABLED(CONFIG_INTEL_TDX_GUEST))
- return false;
-
- if (!once) {
- u32 eax, sig[3];
-
- cpuid_count(TDX_CPUID_LEAF_ID, 0, &eax,
- &sig[0], &sig[2], &sig[1]);
- is_tdx = !memcmp(TDX_IDENT, sig, sizeof(sig));
- once = true;
- }
-
- return is_tdx;
-}
-
-void arch_accept_memory(phys_addr_t start, phys_addr_t end)
-{
- /* Platform-specific memory-acceptance call goes here */
- if (early_is_tdx_guest()) {
- if (!tdx_accept_memory(start, end))
- panic("TDX: Failed to accept memory\n");
- } else if (early_is_sevsnp_guest()) {
- snp_accept_memory(start, end);
- } else {
- error("Cannot accept memory: unknown platform\n");
- }
-}
bool init_unaccepted_memory(void)
{
diff --git a/arch/x86/boot/compressed/sev.h b/arch/x86/boot/compressed/sev.h
index 22637b416b46..62e50c2e71ed 100644
--- a/arch/x86/boot/compressed/sev.h
+++ b/arch/x86/boot/compressed/sev.h
@@ -14,7 +14,6 @@
void snp_accept_memory(phys_addr_t start, phys_addr_t end);
u64 sev_get_status(void);
-bool early_is_sevsnp_guest(void);
static inline u64 sev_es_rd_ghcb_msr(void)
{
@@ -37,7 +36,6 @@ static inline void sev_es_wr_ghcb_msr(u64 val)
static inline void snp_accept_memory(phys_addr_t start, phys_addr_t end) { }
static inline u64 sev_get_status(void) { return 0; }
-static inline bool early_is_sevsnp_guest(void) { return false; }
#endif
diff --git a/arch/x86/boot/compressed/tdx-shared.c b/arch/x86/boot/compressed/tdx-shared.c
index 5ac43762fe13..dc38047647cc 100644
--- a/arch/x86/boot/compressed/tdx-shared.c
+++ b/arch/x86/boot/compressed/tdx-shared.c
@@ -1,2 +1,4 @@
+#define __NO_FORTIFY
+
#include "error.h"
#include "../../coco/tdx/tdx-shared.c"
diff --git a/arch/x86/boot/early_serial_console.c b/arch/x86/boot/early_serial_console.c
index 5b83beab89e1..39fcd551fc81 100644
--- a/arch/x86/boot/early_serial_console.c
+++ b/arch/x86/boot/early_serial_console.c
@@ -22,6 +22,7 @@
#define DLH 1 /* Divisor latch High */
#define DEFAULT_BAUD 9600
+#define BASE_BAUD (1843200 / 16)
static void early_serial_init(int port, int baud)
{
@@ -33,7 +34,7 @@ static void early_serial_init(int port, int baud)
outb(0, port + FCR); /* no fifo */
outb(0x3, port + MCR); /* DTR + RTS */
- divisor = 115200 / baud;
+ divisor = BASE_BAUD / baud;
c = inb(port + LCR);
outb(c | DLAB, port + LCR);
outb(divisor & 0xff, port + DLL);
@@ -74,16 +75,13 @@ static void parse_earlyprintk(void)
else
pos = e - arg;
} else if (!strncmp(arg + pos, "ttyS", 4)) {
- static const int bases[] = { 0x3f8, 0x2f8 };
- int idx = 0;
-
/* += strlen("ttyS"); */
pos += 4;
if (arg[pos++] == '1')
- idx = 1;
-
- port = bases[idx];
+ port = 0x2f8; /* ttyS1 */
+ else
+ port = DEFAULT_SERIAL_PORT;
}
if (arg[pos] == ',')
@@ -98,7 +96,6 @@ static void parse_earlyprintk(void)
early_serial_init(port, baud);
}
-#define BASE_BAUD (1843200/16)
static unsigned int probe_baud(int port)
{
unsigned char lcr, dll, dlh;
diff --git a/arch/x86/boot/string.c b/arch/x86/boot/string.c
index 1632d40e1f54..be454a686422 100644
--- a/arch/x86/boot/string.c
+++ b/arch/x86/boot/string.c
@@ -15,6 +15,7 @@
#include <linux/errno.h>
#include <linux/limits.h>
#include <asm/asm.h>
+#include <asm/shared/string.h>
#include "ctype.h"
#include "string.h"
@@ -31,17 +32,7 @@
int memcmp(const void *s1, const void *s2, size_t len)
{
- bool diff;
-
- /*
- * Make sure ZF is properly set in the len==0 case because in it,
- * RCX==0 and the REPE; CMPSB won't get executed.
- */
- asm volatile("test %3, %3\n\t"
- "repe cmpsb"
- : "=@ccnz" (diff), "+D" (s1), "+S" (s2), "+c" (len)
- : : "cc", "memory");
- return diff;
+ return __inline_memcmp(s1, s2, len);
}
/*
diff --git a/arch/x86/coco/sev/core.c b/arch/x86/coco/sev/core.c
index cc292d7c6fd1..eb2e853e950a 100644
--- a/arch/x86/coco/sev/core.c
+++ b/arch/x86/coco/sev/core.c
@@ -1431,15 +1431,22 @@ static ssize_t vmpl_show(struct kobject *kobj,
return sysfs_emit(buf, "%d\n", snp_vmpl);
}
+static ssize_t sev_status_show(struct kobject *kobj,
+ struct kobj_attribute *attr, char *buf)
+{
+ return sysfs_emit(buf, "0x%llx\n", sev_status);
+}
+
static struct kobj_attribute vmpl_attr = __ATTR_RO(vmpl);
+static struct kobj_attribute sev_status_attr = __ATTR_RO(sev_status);
-static struct attribute *vmpl_attrs[] = {
- &vmpl_attr.attr,
+static struct attribute *sev_status_attrs[] = {
+ &sev_status_attr.attr,
NULL
};
static struct attribute_group sev_attr_group = {
- .attrs = vmpl_attrs,
+ .attrs = sev_status_attrs,
};
static int __init sev_sysfs_init(void)
@@ -1448,7 +1455,7 @@ static int __init sev_sysfs_init(void)
struct device *dev_root;
int ret;
- if (!cc_platform_has(CC_ATTR_GUEST_SEV_SNP))
+ if (!(sev_status & MSR_AMD64_SEV_ENABLED))
return -ENODEV;
dev_root = bus_get_dev_root(&cpu_subsys);
@@ -1463,7 +1470,20 @@ static int __init sev_sysfs_init(void)
ret = sysfs_create_group(sev_kobj, &sev_attr_group);
if (ret)
- kobject_put(sev_kobj);
+ goto drop_kobj;
+
+ if (sev_status & MSR_AMD64_SEV_SNP_ENABLED) {
+ ret = sysfs_add_file_to_group(sev_kobj, &vmpl_attr.attr, NULL);
+ if (ret)
+ goto drop_sysfs;
+ }
+
+ return 0;
+
+drop_sysfs:
+ sysfs_remove_group(sev_kobj, &sev_attr_group);
+drop_kobj:
+ kobject_put(sev_kobj);
return ret;
}
diff --git a/arch/x86/coco/tdx/tdx-shared.c b/arch/x86/coco/tdx/tdx-shared.c
index 1655aa56a0a5..29661d8dfe8e 100644
--- a/arch/x86/coco/tdx/tdx-shared.c
+++ b/arch/x86/coco/tdx/tdx-shared.c
@@ -89,3 +89,34 @@ noinstr u64 __tdx_hypercall(struct tdx_module_args *args)
/* TDVMCALL leaf return code is in R10 */
return args->r10;
}
+
+void __noreturn tdx_panic(const char *msg)
+{
+ struct tdx_module_args args = {
+ .r10 = TDX_HYPERCALL_STANDARD,
+ .r11 = TDVMCALL_REPORT_FATAL_ERROR,
+ .r12 = 0, /* Error code: 0 is Panic */
+ };
+ /* Define register order according to the GHCI */
+ struct { u64 r14, r15, rbx, rdi, rsi, r8, r9, rdx; } message = {};
+
+ /* VMM assumes '\0' in byte 65, if the message took all 64 bytes */
+ memcpy(&message, msg, strnlen(msg, sizeof(message)));
+
+ args.r8 = message.r8;
+ args.r9 = message.r9;
+ args.r14 = message.r14;
+ args.r15 = message.r15;
+ args.rdi = message.rdi;
+ args.rsi = message.rsi;
+ args.rbx = message.rbx;
+ args.rdx = message.rdx;
+
+ /*
+ * This hypercall should never return and it is not safe
+ * to keep the guest running. Call it forever if it
+ * happens to return.
+ */
+ while (1)
+ __tdx_hypercall(&args);
+}
diff --git a/arch/x86/coco/tdx/tdx.c b/arch/x86/coco/tdx/tdx.c
index f904a636d449..ad0131813a30 100644
--- a/arch/x86/coco/tdx/tdx.c
+++ b/arch/x86/coco/tdx/tdx.c
@@ -139,7 +139,7 @@ int tdx_mcall_get_report0(u8 *reportdata, u8 *tdreport)
return 0;
}
-EXPORT_SYMBOL_GPL(tdx_mcall_get_report0);
+EXPORT_SYMBOL_FOR_MODULES(tdx_mcall_get_report0, "tdx-guest");
/**
* tdx_mcall_extend_rtmr() - Wrapper to extend RTMR registers using
@@ -175,7 +175,7 @@ int tdx_mcall_extend_rtmr(u8 index, u8 *data)
return 0;
}
-EXPORT_SYMBOL_GPL(tdx_mcall_extend_rtmr);
+EXPORT_SYMBOL_FOR_MODULES(tdx_mcall_extend_rtmr, "tdx-guest");
/**
* tdx_hcall_get_quote() - Wrapper to request TD Quote using GetQuote
@@ -196,42 +196,7 @@ u64 tdx_hcall_get_quote(u8 *buf, size_t size)
/* Since buf is a shared memory, set the shared (decrypted) bits */
return _tdx_hypercall(TDVMCALL_GET_QUOTE, cc_mkdec(virt_to_phys(buf)), size, 0, 0);
}
-EXPORT_SYMBOL_GPL(tdx_hcall_get_quote);
-
-static void __noreturn tdx_panic(const char *msg)
-{
- struct tdx_module_args args = {
- .r10 = TDX_HYPERCALL_STANDARD,
- .r11 = TDVMCALL_REPORT_FATAL_ERROR,
- .r12 = 0, /* Error code: 0 is Panic */
- };
- union {
- /* Define register order according to the GHCI */
- struct { u64 r14, r15, rbx, rdi, rsi, r8, r9, rdx; };
-
- char bytes[64] __nonstring;
- } message;
-
- /* VMM assumes '\0' in byte 65, if the message took all 64 bytes */
- strtomem_pad(message.bytes, msg, '\0');
-
- args.r8 = message.r8;
- args.r9 = message.r9;
- args.r14 = message.r14;
- args.r15 = message.r15;
- args.rdi = message.rdi;
- args.rsi = message.rsi;
- args.rbx = message.rbx;
- args.rdx = message.rdx;
-
- /*
- * This hypercall should never return and it is not safe
- * to keep the guest running. Call it forever if it
- * happens to return.
- */
- while (1)
- __tdx_hypercall(&args);
-}
+EXPORT_SYMBOL_FOR_MODULES(tdx_hcall_get_quote, "tdx-guest");
/*
* The kernel cannot handle #VEs when accessing normal kernel memory. Ensure
diff --git a/arch/x86/crypto/aegis128-aesni-glue.c b/arch/x86/crypto/aegis128-aesni-glue.c
index f1adfba1a76e..09fc0b15b0e9 100644
--- a/arch/x86/crypto/aegis128-aesni-glue.c
+++ b/arch/x86/crypto/aegis128-aesni-glue.c
@@ -265,8 +265,7 @@ static struct aead_alg crypto_aegis128_aesni_alg = {
static int __init crypto_aegis128_aesni_module_init(void)
{
if (!boot_cpu_has(X86_FEATURE_XMM4_1) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !cpu_has_xfeatures(XFEATURE_MASK_SSE, NULL))
+ !boot_cpu_has(X86_FEATURE_AES))
return -ENODEV;
return crypto_register_aead(&crypto_aegis128_aesni_alg);
diff --git a/arch/x86/crypto/aesni-intel_glue.c b/arch/x86/crypto/aesni-intel_glue.c
index 1903ed8bbe3b..5135bd4e6ea6 100644
--- a/arch/x86/crypto/aesni-intel_glue.c
+++ b/arch/x86/crypto/aesni-intel_glue.c
@@ -1548,8 +1548,7 @@ static int __init register_avx_algs(void)
if (!boot_cpu_has(X86_FEATURE_AVX2) ||
!boot_cpu_has(X86_FEATURE_VAES) ||
!boot_cpu_has(X86_FEATURE_VPCLMULQDQ) ||
- !boot_cpu_has(X86_FEATURE_PCLMULQDQ) ||
- !cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL))
+ !boot_cpu_has(X86_FEATURE_PCLMULQDQ))
return 0;
err = crypto_register_skciphers(skcipher_algs_vaes_avx2,
ARRAY_SIZE(skcipher_algs_vaes_avx2));
@@ -1562,9 +1561,7 @@ static int __init register_avx_algs(void)
if (!boot_cpu_has(X86_FEATURE_AVX512BW) ||
!boot_cpu_has(X86_FEATURE_AVX512VL) ||
- !boot_cpu_has(X86_FEATURE_BMI2) ||
- !cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM |
- XFEATURE_MASK_AVX512, NULL))
+ !boot_cpu_has(X86_FEATURE_BMI2))
return 0;
if (boot_cpu_has(X86_FEATURE_PREFER_YMM)) {
diff --git a/arch/x86/crypto/aria_aesni_avx2_glue.c b/arch/x86/crypto/aria_aesni_avx2_glue.c
index 1487a49bfbac..371be2fb6469 100644
--- a/arch/x86/crypto/aria_aesni_avx2_glue.c
+++ b/arch/x86/crypto/aria_aesni_avx2_glue.c
@@ -195,22 +195,13 @@ static struct skcipher_alg aria_algs[] = {
static int __init aria_avx2_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
!boot_cpu_has(X86_FEATURE_AVX2) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX2 or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
if (boot_cpu_has(X86_FEATURE_GFNI)) {
aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way;
aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way;
diff --git a/arch/x86/crypto/aria_aesni_avx_glue.c b/arch/x86/crypto/aria_aesni_avx_glue.c
index e4e3d78915a5..d23fc91c0ebd 100644
--- a/arch/x86/crypto/aria_aesni_avx_glue.c
+++ b/arch/x86/crypto/aria_aesni_avx_glue.c
@@ -182,21 +182,12 @@ static struct skcipher_alg aria_algs[] = {
static int __init aria_avx_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
if (boot_cpu_has(X86_FEATURE_GFNI)) {
aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way;
aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way;
diff --git a/arch/x86/crypto/aria_gfni_avx512_glue.c b/arch/x86/crypto/aria_gfni_avx512_glue.c
index 363cbf4399cc..e05bbeb22d4a 100644
--- a/arch/x86/crypto/aria_gfni_avx512_glue.c
+++ b/arch/x86/crypto/aria_gfni_avx512_glue.c
@@ -196,24 +196,15 @@ static struct skcipher_alg aria_algs[] = {
static int __init aria_avx512_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
!boot_cpu_has(X86_FEATURE_AVX2) ||
!boot_cpu_has(X86_FEATURE_AVX512F) ||
!boot_cpu_has(X86_FEATURE_AVX512VL) ||
- !boot_cpu_has(X86_FEATURE_GFNI) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_GFNI)) {
pr_info("AVX512/GFNI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM |
- XFEATURE_MASK_AVX512, &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
aria_ops.aria_encrypt_16way = aria_aesni_avx_gfni_encrypt_16way;
aria_ops.aria_decrypt_16way = aria_aesni_avx_gfni_decrypt_16way;
aria_ops.aria_ctr_crypt_16way = aria_aesni_avx_gfni_ctr_crypt_16way;
diff --git a/arch/x86/crypto/camellia_aesni_avx2_glue.c b/arch/x86/crypto/camellia_aesni_avx2_glue.c
index 2d2f4e16537c..073fa3bb8388 100644
--- a/arch/x86/crypto/camellia_aesni_avx2_glue.c
+++ b/arch/x86/crypto/camellia_aesni_avx2_glue.c
@@ -97,22 +97,13 @@ static struct skcipher_alg camellia_algs[] = {
static int __init camellia_aesni_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
!boot_cpu_has(X86_FEATURE_AVX2) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX2 or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
return crypto_register_skciphers(camellia_algs,
ARRAY_SIZE(camellia_algs));
}
diff --git a/arch/x86/crypto/camellia_aesni_avx_glue.c b/arch/x86/crypto/camellia_aesni_avx_glue.c
index 5c321f255eb7..872e5e07220f 100644
--- a/arch/x86/crypto/camellia_aesni_avx_glue.c
+++ b/arch/x86/crypto/camellia_aesni_avx_glue.c
@@ -98,21 +98,12 @@ static struct skcipher_alg camellia_algs[] = {
static int __init camellia_aesni_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
return crypto_register_skciphers(camellia_algs,
ARRAY_SIZE(camellia_algs));
}
diff --git a/arch/x86/crypto/cast5_avx_glue.c b/arch/x86/crypto/cast5_avx_glue.c
index 3aca04d43b34..5de35e863370 100644
--- a/arch/x86/crypto/cast5_avx_glue.c
+++ b/arch/x86/crypto/cast5_avx_glue.c
@@ -92,11 +92,8 @@ static struct skcipher_alg cast5_algs[] = {
static int __init cast5_init(void)
{
- const char *feature_name;
-
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
+ if (!boot_cpu_has(X86_FEATURE_AVX)) {
+ pr_info("AVX instructions are not detected.\n");
return -ENODEV;
}
diff --git a/arch/x86/crypto/cast6_avx_glue.c b/arch/x86/crypto/cast6_avx_glue.c
index c4dd28c30303..3d7ea48007bc 100644
--- a/arch/x86/crypto/cast6_avx_glue.c
+++ b/arch/x86/crypto/cast6_avx_glue.c
@@ -92,11 +92,8 @@ static struct skcipher_alg cast6_algs[] = {
static int __init cast6_init(void)
{
- const char *feature_name;
-
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
+ if (!boot_cpu_has(X86_FEATURE_AVX)) {
+ pr_info("AVX instructions are not detected.\n");
return -ENODEV;
}
diff --git a/arch/x86/crypto/serpent_avx2_glue.c b/arch/x86/crypto/serpent_avx2_glue.c
index f5f2121b7956..72a9e2b306d6 100644
--- a/arch/x86/crypto/serpent_avx2_glue.c
+++ b/arch/x86/crypto/serpent_avx2_glue.c
@@ -93,17 +93,10 @@ static struct skcipher_alg serpent_algs[] = {
static int __init serpent_avx2_init(void)
{
- const char *feature_name;
-
- if (!boot_cpu_has(X86_FEATURE_AVX2) || !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ if (!boot_cpu_has(X86_FEATURE_AVX2)) {
pr_info("AVX2 instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
return crypto_register_skciphers(serpent_algs,
ARRAY_SIZE(serpent_algs));
diff --git a/arch/x86/crypto/serpent_avx_glue.c b/arch/x86/crypto/serpent_avx_glue.c
index 9c8b3a335d5c..42c4e1569674 100644
--- a/arch/x86/crypto/serpent_avx_glue.c
+++ b/arch/x86/crypto/serpent_avx_glue.c
@@ -100,11 +100,8 @@ static struct skcipher_alg serpent_algs[] = {
static int __init serpent_init(void)
{
- const char *feature_name;
-
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
+ if (!boot_cpu_has(X86_FEATURE_AVX)) {
+ pr_info("AVX instructions are not detected.\n");
return -ENODEV;
}
diff --git a/arch/x86/crypto/sm4_aesni_avx2_glue.c b/arch/x86/crypto/sm4_aesni_avx2_glue.c
index fec0ab7a63dd..eef73894e777 100644
--- a/arch/x86/crypto/sm4_aesni_avx2_glue.c
+++ b/arch/x86/crypto/sm4_aesni_avx2_glue.c
@@ -98,22 +98,13 @@ static struct skcipher_alg sm4_aesni_avx2_skciphers[] = {
static int __init sm4_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
!boot_cpu_has(X86_FEATURE_AVX2) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX2 or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
return crypto_register_skciphers(sm4_aesni_avx2_skciphers,
ARRAY_SIZE(sm4_aesni_avx2_skciphers));
}
diff --git a/arch/x86/crypto/sm4_aesni_avx_glue.c b/arch/x86/crypto/sm4_aesni_avx_glue.c
index 88caf418a06f..ed383da5ff46 100644
--- a/arch/x86/crypto/sm4_aesni_avx_glue.c
+++ b/arch/x86/crypto/sm4_aesni_avx_glue.c
@@ -314,21 +314,12 @@ static struct skcipher_alg sm4_aesni_avx_skciphers[] = {
static int __init sm4_init(void)
{
- const char *feature_name;
-
if (!boot_cpu_has(X86_FEATURE_AVX) ||
- !boot_cpu_has(X86_FEATURE_AES) ||
- !boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ !boot_cpu_has(X86_FEATURE_AES)) {
pr_info("AVX or AES-NI instructions are not detected.\n");
return -ENODEV;
}
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
- return -ENODEV;
- }
-
return crypto_register_skciphers(sm4_aesni_avx_skciphers,
ARRAY_SIZE(sm4_aesni_avx_skciphers));
}
diff --git a/arch/x86/crypto/twofish_avx_glue.c b/arch/x86/crypto/twofish_avx_glue.c
index 9e20db013750..985bc54a2340 100644
--- a/arch/x86/crypto/twofish_avx_glue.c
+++ b/arch/x86/crypto/twofish_avx_glue.c
@@ -102,10 +102,8 @@ static struct skcipher_alg twofish_algs[] = {
static int __init twofish_init(void)
{
- const char *feature_name;
-
- if (!cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, &feature_name)) {
- pr_info("CPU feature '%s' is not supported.\n", feature_name);
+ if (!boot_cpu_has(X86_FEATURE_AVX)) {
+ pr_info("AVX instructions are not detected.\n");
return -ENODEV;
}
diff --git a/arch/x86/events/amd/uncore.c b/arch/x86/events/amd/uncore.c
index 7181973b5b12..7aa5a5b652a2 100644
--- a/arch/x86/events/amd/uncore.c
+++ b/arch/x86/events/amd/uncore.c
@@ -39,11 +39,11 @@ static int pmu_version;
struct amd_uncore_ctx {
int refcnt;
int cpu;
- struct perf_event **events;
unsigned long active_mask[BITS_TO_LONGS(NUM_COUNTERS_MAX)];
int nr_active;
struct hrtimer hrtimer;
u64 hrtimer_duration;
+ struct perf_event *events[];
};
struct amd_uncore_pmu {
@@ -206,17 +206,15 @@ static int amd_uncore_add(struct perf_event *event, int flags)
struct amd_uncore_ctx *ctx = *per_cpu_ptr(pmu->ctx, event->cpu);
struct hw_perf_event *hwc = &event->hw;
- /* are we already assigned? */
+ /*
+ * Perf serializes ->add() and ->del() for an event. A successful
+ * ->add() records the claimed slot in hwc->idx before returning, and
+ * ->del() clears that slot before resetting hwc->idx. Therefore, an
+ * existing assignment must be at hwc->idx.
+ */
if (hwc->idx != -1 && ctx->events[hwc->idx] == event)
goto out;
- for (i = 0; i < pmu->num_counters; i++) {
- if (ctx->events[i] == event) {
- hwc->idx = i;
- goto out;
- }
- }
-
/* if not, take the first available counter */
hwc->idx = -1;
for (i = 0; i < pmu->num_counters; i++) {
@@ -248,19 +246,15 @@ out:
static void amd_uncore_del(struct perf_event *event, int flags)
{
- int i;
struct amd_uncore_pmu *pmu = event_to_amd_uncore_pmu(event);
struct amd_uncore_ctx *ctx = *per_cpu_ptr(pmu->ctx, event->cpu);
struct hw_perf_event *hwc = &event->hw;
+ struct perf_event *old = event;
event->pmu->stop(event, PERF_EF_UPDATE);
- for (i = 0; i < pmu->num_counters; i++) {
- struct perf_event *tmp = event;
-
- if (try_cmpxchg(&ctx->events[i], &tmp, NULL))
- break;
- }
+ /* ->del() follows a successful ->add(), so hwc->idx owns this slot. */
+ WARN_ON_ONCE(!try_cmpxchg(&ctx->events[hwc->idx], &old, NULL));
hwc->idx = -1;
}
@@ -519,10 +513,8 @@ static void amd_uncore_ctx_free(struct amd_uncore *uncore, unsigned int cpu)
if (cpu == ctx->cpu)
cpumask_clear_cpu(cpu, &pmu->active_mask);
- if (!--ctx->refcnt) {
- kfree(ctx->events);
+ if (!--ctx->refcnt)
kfree(ctx);
- }
*per_cpu_ptr(pmu->ctx, cpu) = NULL;
}
@@ -567,18 +559,11 @@ static int amd_uncore_ctx_init(struct amd_uncore *uncore, unsigned int cpu)
/* Allocate context if sibling does not exist */
if (!curr) {
node = cpu_to_node(cpu);
- curr = kzalloc_node(sizeof(*curr), GFP_KERNEL, node);
+ curr = kzalloc_node(struct_size(curr, events, pmu->num_counters), GFP_KERNEL, node);
if (!curr)
goto fail;
curr->cpu = cpu;
- curr->events = kzalloc_node(sizeof(*curr->events) *
- pmu->num_counters,
- GFP_KERNEL, node);
- if (!curr->events) {
- kfree(curr);
- goto fail;
- }
amd_uncore_init_hrtimer(curr);
curr->hrtimer_duration = (u64)update_interval * NSEC_PER_MSEC;
diff --git a/arch/x86/events/core.c b/arch/x86/events/core.c
index 8b3ea0adb965..9b1a9032e36d 100644
--- a/arch/x86/events/core.c
+++ b/arch/x86/events/core.c
@@ -93,7 +93,7 @@ DEFINE_STATIC_CALL_NULL(x86_pmu_stop_scheduling, *x86_pmu.stop_scheduling);
DEFINE_STATIC_CALL_NULL(x86_pmu_sched_task, *x86_pmu.sched_task);
-DEFINE_STATIC_CALL_NULL(x86_pmu_drain_pebs, *x86_pmu.drain_pebs);
+DEFINE_STATIC_CALL_RET0(x86_pmu_drain_pebs, *x86_pmu.drain_pebs);
DEFINE_STATIC_CALL_NULL(x86_pmu_pebs_aliases, *x86_pmu.pebs_aliases);
DEFINE_STATIC_CALL_NULL(x86_pmu_filter, *x86_pmu.filter);
@@ -408,6 +408,53 @@ set_ext_hw_attr(struct hw_perf_event *hwc, struct perf_event *event)
return x86_pmu_extra_regs(val, event);
}
+static DEFINE_PER_CPU(struct xregs_state *, ext_regs_buf);
+
+static void release_ext_regs_buffers(void)
+{
+ int cpu;
+
+ if (!x86_pmu.ext_regs_mask)
+ return;
+
+ for_each_possible_cpu(cpu) {
+ kfree(per_cpu(ext_regs_buf, cpu));
+ per_cpu(ext_regs_buf, cpu) = NULL;
+ }
+}
+
+static void reserve_ext_regs_buffers(void)
+{
+ bool compacted = cpu_feature_enabled(X86_FEATURE_XCOMPACTED);
+ unsigned int size;
+ int cpu;
+
+ if (!x86_pmu.ext_regs_mask)
+ return;
+
+ /* Add 64 bytes to satisfy the XSAVE area's 64-byte alignment. */
+ size = xstate_calculate_size(x86_pmu.ext_regs_mask, compacted) + 64;
+
+ for_each_possible_cpu(cpu) {
+ per_cpu(ext_regs_buf, cpu) = kzalloc_node(size, GFP_KERNEL,
+ cpu_to_node(cpu));
+ if (WARN_ON_ONCE(!per_cpu(ext_regs_buf, cpu)))
+ goto err;
+ }
+
+ return;
+
+err:
+ release_ext_regs_buffers();
+}
+
+static inline struct xregs_state *get_ext_regs_buf(int cpu)
+{
+ void *buf = per_cpu(ext_regs_buf, cpu);
+
+ return buf ? PTR_ALIGN(buf, 64) : NULL;
+}
+
int x86_reserve_hardware(void)
{
int err = 0;
@@ -420,6 +467,7 @@ int x86_reserve_hardware(void)
} else {
reserve_ds_buffers();
reserve_lbr_buffers();
+ reserve_ext_regs_buffers();
}
}
if (!err)
@@ -436,6 +484,7 @@ void x86_release_hardware(void)
release_pmc_hardware();
release_ds_buffers();
release_lbr_buffers();
+ release_ext_regs_buffers();
mutex_unlock(&pmc_reserve_mutex);
}
}
@@ -583,6 +632,79 @@ int x86_pmu_max_precise(struct pmu *pmu)
return precise;
}
+static int pebs_simd_regs_validate(struct perf_event *event)
+{
+ u64 caps = hybrid(event->pmu, arch_pebs_cap).caps;
+
+ if (event_needs_xmm(event) &&
+ !x86_pmu.arch_pebs && !x86_pmu.intel_cap.pebs_baseline)
+ return -EINVAL;
+ if (event_needs_xmm(event) &&
+ x86_pmu.arch_pebs && !(caps & ARCH_PEBS_VECR_XMM))
+ return -EINVAL;
+
+ if (event_needs_ssp(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_GPR)))
+ return -EINVAL;
+ if (event_needs_ymm(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_YMMH)))
+ return -EINVAL;
+ if (event_needs_egprs(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_EGPRS)))
+ return -EINVAL;
+ if (event_needs_opmask(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_OPMASK)))
+ return -EINVAL;
+ if (event_needs_low16_zmm(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_ZMMH)))
+ return -EINVAL;
+ if (event_needs_high16_zmm(event) &&
+ !(x86_pmu.arch_pebs && (caps & ARCH_PEBS_VECR_H16ZMM)))
+ return -EINVAL;
+
+ return 0;
+}
+
+static int event_simd_regs_validate(struct perf_event *event)
+{
+ u64 reserved = ~GENMASK_ULL(PERF_REG_MISC_MAX - 1, 0);
+
+ if (!get_ext_regs_buf(raw_smp_processor_id()))
+ return -ENOMEM;
+ /*
+ * The XMM space in the perf_event_x86_regs is reclaimed
+ * for eGPRs and other general registers.
+ */
+ if (((event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_regs_intr & reserved)) ||
+ ((event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_regs_user & reserved)))
+ return -EINVAL;
+ if (event_needs_egprs(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_APX))
+ return -EINVAL;
+ if (event_needs_ssp(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_CET_USER))
+ return -EINVAL;
+ if (event_needs_xmm(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_SSE))
+ return -EINVAL;
+ if (event_needs_ymm(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_YMM))
+ return -EINVAL;
+ if (event_needs_low16_zmm(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_ZMM_Hi256))
+ return -EINVAL;
+ if (event_needs_high16_zmm(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_Hi16_ZMM))
+ return -EINVAL;
+ if (event_needs_opmask(event) &&
+ !(x86_pmu.ext_regs_mask & XFEATURE_MASK_OPMASK))
+ return -EINVAL;
+
+ return 0;
+}
+
int x86_pmu_hw_config(struct perf_event *event)
{
if (event->attr.precise_ip) {
@@ -653,19 +775,38 @@ int x86_pmu_hw_config(struct perf_event *event)
return -EINVAL;
}
- /* sample_regs_user never support XMM registers */
- if (unlikely(event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK))
- return -EINVAL;
- /*
- * Besides the general purpose registers, XMM registers may
- * be collected in PEBS on some platforms, e.g. Icelake
- */
- if (unlikely(event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK)) {
- if (!(event->pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS))
- return -EINVAL;
+ if (event->attr.sample_type & (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)) {
+ int ret;
- if (!event->attr.precise_ip)
- return -EINVAL;
+ if (event->attr.sample_simd_regs_enabled) {
+ if (!(event->pmu->capabilities & PERF_PMU_CAP_SIMD_REGS))
+ return -EINVAL;
+
+ if (event->attr.precise_ip) {
+ ret = pebs_simd_regs_validate(event);
+ if (ret)
+ return ret;
+ }
+ ret = event_simd_regs_validate(event);
+ if (ret)
+ return ret;
+ } else if (event_has_extended_regs(event)) {
+ if (!(event->pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS))
+ return -EINVAL;
+
+ if (event->attr.precise_ip) {
+ u64 caps = hybrid(event->pmu, arch_pebs_cap).caps;
+
+ if (x86_pmu.arch_pebs && !(caps & ARCH_PEBS_VECR_XMM))
+ return -EINVAL;
+ if (!x86_pmu.arch_pebs && !x86_pmu.intel_cap.pebs_baseline)
+ return -EINVAL;
+ }
+ if (!get_ext_regs_buf(raw_smp_processor_id()))
+ return -ENOMEM;
+ if (!(x86_pmu.ext_regs_mask & XFEATURE_MASK_SSE))
+ return -EINVAL;
+ }
}
return x86_setup_perfctr(event);
@@ -721,9 +862,10 @@ void x86_pmu_disable_all(void)
}
}
-struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data)
+struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr,
+ struct x86_guest_pebs *guest_pebs)
{
- return static_call(x86_pmu_guest_get_msrs)(nr, data);
+ return static_call(x86_pmu_guest_get_msrs)(nr, guest_pebs);
}
EXPORT_SYMBOL_FOR_KVM(perf_guest_get_msrs);
@@ -1717,6 +1859,302 @@ do_del:
static_call_cond(x86_pmu_del)(event);
}
+void x86_pmu_clear_perf_regs(struct pt_regs *regs)
+{
+ struct x86_perf_regs *perf_regs = container_of(regs, struct x86_perf_regs, regs);
+
+ perf_regs->abi = PERF_SAMPLE_REGS_ABI_NONE;
+ perf_regs->xmm_regs = NULL;
+ perf_regs->ymmh_regs = NULL;
+ perf_regs->zmmh_regs = NULL;
+ perf_regs->h16zmm_regs = NULL;
+ perf_regs->opmask_regs = NULL;
+ perf_regs->egpr_regs = NULL;
+ perf_regs->ssp = NULL;
+}
+
+static void update_perf_regs(struct x86_perf_regs *perf_regs,
+ struct xregs_state *xsave, u64 bitmap)
+{
+ struct cet_user_state *cet;
+ u64 mask;
+
+ if (!xsave)
+ return;
+
+ /* Restrict to features actually saved by XSAVES */
+ mask = bitmap & xsave->header.xfeatures;
+
+ if (mask & XFEATURE_MASK_SSE)
+ perf_regs->xmm_space = xsave->i387.xmm_space;
+ if (mask & XFEATURE_MASK_YMM)
+ perf_regs->ymmh = get_xsave_addr(xsave, XFEATURE_YMM);
+ if (mask & XFEATURE_MASK_ZMM_Hi256)
+ perf_regs->zmmh = get_xsave_addr(xsave, XFEATURE_ZMM_Hi256);
+ if (mask & XFEATURE_MASK_Hi16_ZMM)
+ perf_regs->h16zmm = get_xsave_addr(xsave, XFEATURE_Hi16_ZMM);
+ if (mask & XFEATURE_MASK_OPMASK)
+ perf_regs->opmask = get_xsave_addr(xsave, XFEATURE_OPMASK);
+ if (mask & XFEATURE_MASK_APX)
+ perf_regs->egpr = get_xsave_addr(xsave, XFEATURE_APX);
+ if (mask & XFEATURE_MASK_CET_USER) {
+ cet = get_xsave_addr(xsave, XFEATURE_CET_USER);
+ perf_regs->ssp = cet ? &cet->user_ssp : NULL;
+ }
+}
+
+/*
+ * The x86 specific variant of perf_sample_regs_intr().
+ * Update data->regs_intr fields for extended registers (e.g., SIMD).
+ */
+static void x86_pmu_update_regs_intr(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs,
+ bool exclude_kernel)
+{
+ struct x86_perf_regs *perf_regs;
+
+ if (exclude_kernel && !user_mode(regs)) {
+ data->regs_intr.regs = NULL;
+ data->regs_intr.abi = PERF_SAMPLE_REGS_ABI_NONE;
+ } else {
+ data->regs_intr.regs = regs;
+ data->regs_intr.abi = perf_reg_abi(current);
+ }
+
+ data->dyn_size += sizeof(u64);
+ if (data->regs_intr.regs) {
+ data->dyn_size += hweight64(event->attr.sample_regs_intr) *
+ sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ data->dyn_size += perf_update_xregs_size(event, true);
+ data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
+
+ perf_regs = container_of(data->regs_intr.regs,
+ struct x86_perf_regs, regs);
+ perf_regs->abi = data->regs_intr.abi;
+ }
+
+ /*
+ * Set PERF_SAMPLE_REGS_INTR to bypass perf_sample_regs_intr() call
+ * in perf_prepare_sample() function.
+ */
+ data->sample_flags |= PERF_SAMPLE_REGS_INTR;
+}
+
+static DEFINE_PER_CPU(struct x86_perf_regs, x86_user_regs);
+
+static void x86_pmu_get_regs_user(struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct x86_perf_regs *x86_regs_user = this_cpu_ptr(&x86_user_regs);
+ struct perf_regs regs_user;
+
+ x86_pmu_clear_perf_regs(&x86_regs_user->regs);
+
+ perf_get_regs_user(&regs_user, regs);
+ data->regs_user.abi = regs_user.abi;
+ if (regs_user.regs) {
+ x86_regs_user->regs = *regs_user.regs;
+ data->regs_user.regs = &x86_regs_user->regs;
+ } else {
+ data->regs_user.regs = NULL;
+ }
+}
+
+/*
+ * The x86 specific variant of perf_sample_regs_user().
+ * Update data->regs_user fields for extended registers (e.g., SIMD).
+ */
+static void x86_pmu_update_regs_user(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct x86_perf_regs *x86_regs_user = this_cpu_ptr(&x86_user_regs);
+ struct perf_event_attr *attr = &event->attr;
+ struct x86_perf_regs *perf_regs;
+
+ /*
+ * PERF_SAMPLE_REGS_INTR and PERF_SAMPLE_REGS_USER can both be
+ * requested for one event. Keep user regs in a separate x86_perf_regs
+ * instance, so intr-reg collection does not overwrite user-reg data.
+ */
+ if (user_mode(regs)) {
+ x86_pmu_clear_perf_regs(&x86_regs_user->regs);
+ perf_regs = container_of(regs, struct x86_perf_regs, regs);
+ /* Copy all sampled regs data to x86_regs_user. */
+ *x86_regs_user = *perf_regs;
+ data->regs_user.regs = &x86_regs_user->regs;
+ data->regs_user.abi = perf_reg_abi(current);
+ } else if (is_user_task(current)) {
+ x86_pmu_get_regs_user(data, regs);
+ } else {
+ data->regs_user.abi = PERF_SAMPLE_REGS_ABI_NONE;
+ data->regs_user.regs = NULL;
+ }
+
+ data->dyn_size += sizeof(u64);
+ if (data->regs_user.regs) {
+ data->dyn_size += hweight64(attr->sample_regs_user) * sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ data->dyn_size += perf_update_xregs_size(event, false);
+ data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
+
+ x86_regs_user->abi = data->regs_user.abi;
+ }
+
+ /*
+ * Set PERF_SAMPLE_REGS_USER to bypass perf_sample_regs_user() call
+ * in perf_prepare_sample() function.
+ */
+ data->sample_flags |= PERF_SAMPLE_REGS_USER;
+}
+
+/*
+ * This function retrieves cached user-space fpu registers (XMM/YMM/ZMM).
+ * If TIF_NEED_FPU_LOAD is set or PMI hits into guest, it indicates that
+ * the user-space FPU state is cached. Otherwise, the data should be read
+ * directly from the hardware registers.
+ */
+static inline u64 x86_pmu_update_user_xregs(struct perf_sample_data *data,
+ struct pt_regs *regs,
+ u64 mask, bool from_pebs)
+{
+ struct x86_perf_regs *perf_regs;
+ struct xregs_state *xsave;
+ struct fpu *fpu;
+ struct fpstate *fps;
+ u64 user_mask = mask;
+
+ if (!is_user_task(current))
+ return 0;
+
+ if (data->regs_user.abi == PERF_SAMPLE_REGS_ABI_NONE)
+ return 0;
+
+ /*
+ * If PEBS hits kernel space, need to re-sample extended
+ * registers for user space.
+ */
+ if (user_mode(regs))
+ user_mask = from_pebs ? 0 : mask;
+
+ fpu = x86_task_fpu(current);
+ fps = READ_ONCE(fpu->__task_fpstate);
+ /*
+ * If fpu->__task_fpstate is set, it points to the cached user
+ * FPU state (e.g. with KVM guest-state swapping). Otherwise,
+ * when TIF_NEED_FPU_LOAD is set, fpu->fpstate holds the cached
+ * user state. If neither is true, the user state is live in hardware.
+ */
+ if (user_mask && (test_thread_flag(TIF_NEED_FPU_LOAD) || fps)) {
+ perf_regs = container_of(data->regs_user.regs,
+ struct x86_perf_regs, regs);
+ if (!fps)
+ fps = fpu->fpstate;
+ xsave = &fps->regs.xsave;
+
+ update_perf_regs(perf_regs, xsave, user_mask);
+ return 0;
+ }
+
+ return user_mask;
+}
+
+static u64 get_simd_sample_mask(struct perf_event *event, u64 sample_type)
+{
+ u64 mask = 0;
+
+ if (__event_needs_xmm(event, sample_type))
+ mask |= XFEATURE_MASK_SSE;
+ if (__event_needs_ymm(event, sample_type))
+ mask |= XFEATURE_MASK_YMM;
+ if (__event_needs_low16_zmm(event, sample_type))
+ mask |= XFEATURE_MASK_ZMM_Hi256;
+ if (__event_needs_high16_zmm(event, sample_type))
+ mask |= XFEATURE_MASK_Hi16_ZMM;
+ if (__event_needs_opmask(event, sample_type))
+ mask |= XFEATURE_MASK_OPMASK;
+ if (__event_needs_egprs(event, sample_type))
+ mask |= XFEATURE_MASK_APX;
+ if (__event_needs_ssp(event, sample_type))
+ mask |= XFEATURE_MASK_CET_USER;
+
+ return mask;
+}
+
+static void x86_pmu_sample_xregs(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs,
+ bool from_pebs)
+{
+ struct xregs_state *xsave = get_ext_regs_buf(smp_processor_id());
+ u64 sample_type = event->attr.sample_type;
+ struct x86_perf_regs *perf_regs;
+ u64 intr_mask = 0;
+ u64 user_mask = 0;
+
+ if (WARN_ON_ONCE(!xsave) || !in_nmi())
+ return;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) && data->regs_intr.regs) {
+ intr_mask |= get_simd_sample_mask(event, PERF_SAMPLE_REGS_INTR);
+ intr_mask &= x86_pmu.ext_regs_mask;
+ intr_mask = from_pebs ? 0 : intr_mask;
+ }
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) && data->regs_user.regs) {
+ user_mask |= get_simd_sample_mask(event, PERF_SAMPLE_REGS_USER);
+ user_mask &= x86_pmu.ext_regs_mask;
+ user_mask = x86_pmu_update_user_xregs(data, regs,
+ user_mask, from_pebs);
+ }
+
+ if (user_mask | intr_mask) {
+ xsave->header.xfeatures = 0;
+ xsaves_nmi(xsave, user_mask | intr_mask);
+ }
+
+ if (intr_mask) {
+ perf_regs = container_of(data->regs_intr.regs,
+ struct x86_perf_regs, regs);
+ update_perf_regs(perf_regs, xsave, intr_mask);
+ }
+
+ if (user_mask) {
+ perf_regs = container_of(data->regs_user.regs,
+ struct x86_perf_regs, regs);
+ update_perf_regs(perf_regs, xsave, user_mask);
+ }
+}
+
+void x86_pmu_update_perf_regs(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs,
+ bool from_pebs)
+{
+ u64 sample_type = event->attr.sample_type;
+
+ if (!(sample_type &
+ (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)))
+ return;
+
+ if (!event_needs_xmm(event) &&
+ !event_has_simd_regs(event))
+ return;
+
+ if (sample_type & PERF_SAMPLE_REGS_INTR) {
+ x86_pmu_update_regs_intr(event, data, regs,
+ event->attr.exclude_kernel);
+ }
+ if (sample_type & PERF_SAMPLE_REGS_USER)
+ x86_pmu_update_regs_user(event, data, regs);
+
+ x86_pmu_sample_xregs(event, data, regs, from_pebs);
+}
+
int x86_pmu_handle_irq(struct pt_regs *regs)
{
struct perf_sample_data data;
@@ -1800,9 +2238,11 @@ void perf_put_guest_lvtpc(void)
EXPORT_SYMBOL_FOR_KVM(perf_put_guest_lvtpc);
#endif /* CONFIG_PERF_GUEST_MEDIATED_PMU */
+static DEFINE_PER_CPU(struct x86_perf_regs, x86_intr_regs);
static int
perf_event_nmi_handler(unsigned int cmd, struct pt_regs *regs)
{
+ struct x86_perf_regs *x86_regs = this_cpu_ptr(&x86_intr_regs);
u64 start_clock;
u64 finish_clock;
int ret;
@@ -1826,7 +2266,8 @@ perf_event_nmi_handler(unsigned int cmd, struct pt_regs *regs)
return NMI_DONE;
start_clock = sched_clock();
- ret = static_call(x86_pmu_handle_irq)(regs);
+ x86_regs->regs = *regs;
+ ret = static_call(x86_pmu_handle_irq)(&x86_regs->regs);
finish_clock = sched_clock();
perf_sample_event_took(finish_clock - start_clock);
@@ -2218,8 +2659,20 @@ static int __init init_hw_perf_events(void)
pmu.attr_update = x86_pmu.attr_update;
- if (!is_hybrid())
+ if (!is_hybrid()) {
x86_pmu_show_pmu_cap(NULL);
+ } else {
+ int i;
+
+ /*
+ * Init default ops.
+ * Must be called before registering x86_pmu_starting_cpu(),
+ * otherwise some key PMU fields, e.g., capabilities
+ * initialized in x86_pmu_starting_cpu(), would be overwritten.
+ */
+ for (i = 0; i < x86_pmu.num_hybrid_pmus; i++)
+ x86_pmu.hybrid_pmu[i].pmu = pmu;
+ }
if (!x86_pmu.read)
x86_pmu.read = _x86_pmu_read;
@@ -2266,7 +2719,6 @@ static int __init init_hw_perf_events(void)
for (i = 0; i < x86_pmu.num_hybrid_pmus; i++) {
hybrid_pmu = &x86_pmu.hybrid_pmu[i];
- hybrid_pmu->pmu = pmu;
hybrid_pmu->pmu.type = -1;
hybrid_pmu->pmu.attr_update = x86_pmu.attr_update;
hybrid_pmu->pmu.capabilities |= PERF_PMU_CAP_EXTENDED_HW_TYPE;
diff --git a/arch/x86/events/intel/core.c b/arch/x86/events/intel/core.c
index 3ef80882843e..0a34d674df59 100644
--- a/arch/x86/events/intel/core.c
+++ b/arch/x86/events/intel/core.c
@@ -14,7 +14,6 @@
#include <linux/slab.h>
#include <linux/export.h>
#include <linux/nmi.h>
-#include <linux/kvm_host.h>
#include <asm/cpufeature.h>
#include <asm/cpuid/api.h>
@@ -2797,7 +2796,7 @@ static void __intel_pmu_enable_all(int added, bool pmi)
}
wrmsrq(MSR_CORE_PERF_GLOBAL_CTRL,
- intel_ctrl & ~cpuc->intel_ctrl_guest_mask);
+ intel_ctrl & ~cpuc->intel_ctrl_exclude_host_mask);
if (test_bit(INTEL_PMC_IDX_FIXED_BTS, cpuc->active_mask)) {
struct perf_event *event =
@@ -2995,9 +2994,9 @@ static inline void intel_set_masks(struct perf_event *event, int idx)
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
if (event->attr.exclude_host)
- __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_guest_mask);
+ __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask);
if (event->attr.exclude_guest)
- __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_host_mask);
+ __set_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_guest_mask);
if (event_is_checkpointed(event))
__set_bit(idx, (unsigned long *)&cpuc->intel_cp_status);
}
@@ -3006,8 +3005,8 @@ static inline void intel_clear_masks(struct perf_event *event, int idx)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
- __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_guest_mask);
- __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_host_mask);
+ __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask);
+ __clear_bit(idx, (unsigned long *)&cpuc->intel_ctrl_exclude_guest_mask);
__clear_bit(idx, (unsigned long *)&cpuc->intel_cp_status);
}
@@ -3499,6 +3498,21 @@ static void intel_pmu_enable_event_ext(struct perf_event *event)
if (pebs_data_cfg & PEBS_DATACFG_XMMS)
ext |= ARCH_PEBS_VECR_XMM & cap.caps;
+ if (pebs_data_cfg & PEBS_DATACFG_YMMHS)
+ ext |= ARCH_PEBS_VECR_YMMH & cap.caps;
+
+ if (pebs_data_cfg & PEBS_DATACFG_EGPRS)
+ ext |= ARCH_PEBS_VECR_EGPRS & cap.caps;
+
+ if (pebs_data_cfg & PEBS_DATACFG_OPMASKS)
+ ext |= ARCH_PEBS_VECR_OPMASK & cap.caps;
+
+ if (pebs_data_cfg & PEBS_DATACFG_ZMMHS)
+ ext |= ARCH_PEBS_VECR_ZMMH & cap.caps;
+
+ if (pebs_data_cfg & PEBS_DATACFG_H16ZMMS)
+ ext |= ARCH_PEBS_VECR_H16ZMM & cap.caps;
+
if (pebs_data_cfg & PEBS_DATACFG_LBRS)
ext |= ARCH_PEBS_LBR & cap.caps;
@@ -3770,20 +3784,20 @@ static void intel_pmu_reset(void)
*
* The contents and other behavior of the guest event do not matter.
*/
-static void x86_pmu_handle_guest_pebs(struct pt_regs *regs,
- struct perf_sample_data *data)
+static int x86_pmu_handle_guest_pebs(struct pt_regs *regs,
+ struct perf_sample_data *data)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
- u64 guest_pebs_idxs = cpuc->pebs_enabled & ~cpuc->intel_ctrl_host_mask;
+ u64 guest_pebs_idxs = cpuc->pebs_enabled & ~cpuc->intel_ctrl_exclude_guest_mask;
struct perf_event *event = NULL;
int bit;
if (!unlikely(perf_guest_state()))
- return;
+ return 0;
if (!x86_pmu.pebs_ept || !x86_pmu.pebs_active ||
!guest_pebs_idxs)
- return;
+ return 0;
for_each_set_bit(bit, (unsigned long *)&guest_pebs_idxs, X86_PMC_IDX_MAX) {
event = cpuc->events[bit];
@@ -3793,9 +3807,14 @@ static void x86_pmu_handle_guest_pebs(struct pt_regs *regs,
perf_sample_data_init(data, 0, event->hw.last_period);
perf_event_overflow(event, data, regs);
- /* Inject one fake event is enough. */
- break;
+ /*
+ * Inject one fake event is enough.
+ * Returning 1 to inform PMI is handled.
+ */
+ return 1;
}
+
+ return 0;
}
static int handle_pmi_common(struct pt_regs *regs, u64 status)
@@ -3844,9 +3863,11 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status)
if (__test_and_clear_bit(GLOBAL_STATUS_BUFFER_OVF_BIT, (unsigned long *)&status)) {
u64 pebs_enabled = cpuc->pebs_enabled;
- handled++;
- x86_pmu_handle_guest_pebs(regs, &data);
- static_call(x86_pmu_drain_pebs)(regs, &data);
+ handled += x86_pmu_handle_guest_pebs(regs, &data);
+ handled += static_call(x86_pmu_drain_pebs)(regs, &data);
+ /* Ensure no "suspicious NMI" warning for empty PEBS buffer. */
+ if (!handled)
+ handled++;
/*
* PMI throttle may be triggered, which stops the PEBS event.
@@ -3873,8 +3894,10 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status)
*/
if (__test_and_clear_bit(GLOBAL_STATUS_ARCH_PEBS_THRESHOLD_BIT,
(unsigned long *)&status)) {
- handled++;
- static_call(x86_pmu_drain_pebs)(regs, &data);
+ handled += static_call(x86_pmu_drain_pebs)(regs, &data);
+ /* Ensure no "suspicious NMI" warning for empty PEBS buffer. */
+ if (!handled)
+ handled++;
if (cpuc->events[INTEL_PMC_IDX_FIXED_SLOTS] &&
is_pebs_counter_event_group(cpuc->events[INTEL_PMC_IDX_FIXED_SLOTS]))
@@ -3950,6 +3973,9 @@ static int handle_pmi_common(struct pt_regs *regs, u64 status)
if (has_branch_stack(event))
intel_pmu_lbr_save_brstack(&data, cpuc, event);
+ x86_pmu_clear_perf_regs(regs);
+ x86_pmu_update_perf_regs(event, &data, regs, false);
+
perf_event_overflow(event, &data, regs);
}
@@ -4717,14 +4743,20 @@ static void intel_pebs_aliases_skl(struct perf_event *event)
static unsigned long intel_pmu_large_pebs_flags(struct perf_event *event)
{
unsigned long flags = x86_pmu.large_pebs_flags;
+ u64 gprs_mask = event->attr.sample_simd_regs_enabled ?
+ PEBS_GP_REGS | PERF_X86_EGPRS_MASK |
+ BIT_ULL(PERF_REG_X86_SSP) :
+ PEBS_GP_REGS | PERF_REG_EXTENDED_MASK;
if (event->attr.use_clockid)
flags &= ~PERF_SAMPLE_TIME;
if (!event->attr.exclude_kernel)
flags &= ~PERF_SAMPLE_REGS_USER;
- if (event->attr.sample_regs_user & ~PEBS_GP_REGS)
+ if ((event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_regs_user & ~gprs_mask))
flags &= ~PERF_SAMPLE_REGS_USER;
- if (event->attr.sample_regs_intr & ~PEBS_GP_REGS)
+ if ((event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_regs_intr & ~gprs_mask))
flags &= ~PERF_SAMPLE_REGS_INTR;
return flags;
}
@@ -5295,13 +5327,13 @@ static int intel_pmu_hw_config(struct perf_event *event)
* when it uses {RD,WR}MSR, which should be handled by the KVM context,
* specifically in the intel_pmu_{get,set}_msr().
*/
-static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
+static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr,
+ struct x86_guest_pebs *guest_pebs)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct perf_guest_switch_msr *arr = cpuc->guest_switch_msrs;
- struct kvm_pmu *kvm_pmu = (struct kvm_pmu *)data;
u64 intel_ctrl = hybrid(cpuc->pmu, intel_ctrl);
- u64 pebs_mask = cpuc->pebs_enabled & x86_pmu.pebs_capable;
+ u64 pebs_mask = intel_ctrl & cpuc->pebs_enabled & x86_pmu.pebs_capable;
u64 guest_pebs_mask;
int global_ctrl;
@@ -5316,8 +5348,8 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
global_ctrl = (*nr)++;
arr[global_ctrl] = (struct perf_guest_switch_msr){
.msr = MSR_CORE_PERF_GLOBAL_CTRL,
- .host = intel_ctrl & ~cpuc->intel_ctrl_guest_mask,
- .guest = intel_ctrl & ~cpuc->intel_ctrl_host_mask & ~pebs_mask,
+ .host = intel_ctrl & ~cpuc->intel_ctrl_exclude_host_mask,
+ .guest = intel_ctrl & ~cpuc->intel_ctrl_exclude_guest_mask & ~pebs_mask,
};
if (!x86_pmu.ds_pebs)
@@ -5353,17 +5385,9 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
* the guest wants to use for PEBS, (c) are not excluded from counting
* in the guest, and (d) _are_ excluded from counting in the host.
*/
- guest_pebs_mask = pebs_mask & intel_ctrl & kvm_pmu->pebs_enable &
- ~cpuc->intel_ctrl_host_mask &
- cpuc->intel_ctrl_guest_mask;
-
- /*
- * Disable counters where the guest PMC is different than the host PMC
- * being used on behalf of the guest, as the PEBS record includes
- * PERF_GLOBAL_STATUS, i.e. the guest will see overflow status for the
- * wrong counter(s).
- */
- guest_pebs_mask &= ~kvm_pmu->host_cross_mapped_mask;
+ guest_pebs_mask = pebs_mask & guest_pebs->enable &
+ ~cpuc->intel_ctrl_exclude_guest_mask &
+ cpuc->intel_ctrl_exclude_host_mask;
/*
* FIXME: Allow guest and host usage of PEBS events to co-exist instead
@@ -5371,7 +5395,7 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
* What exactly goes wrong if guest and host are using PEBS is
* unknown.
*/
- if (pebs_mask & ~cpuc->intel_ctrl_guest_mask)
+ if (pebs_mask & ~cpuc->intel_ctrl_exclude_host_mask)
guest_pebs_mask = 0;
/*
@@ -5382,14 +5406,14 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
arr[(*nr)++] = (struct perf_guest_switch_msr){
.msr = MSR_IA32_DS_AREA,
.host = (unsigned long)cpuc->ds,
- .guest = guest_pebs_mask ? kvm_pmu->ds_area : (unsigned long)cpuc->ds,
+ .guest = guest_pebs_mask ? guest_pebs->ds_area : (unsigned long)cpuc->ds,
};
if (x86_pmu.intel_cap.pebs_baseline) {
arr[(*nr)++] = (struct perf_guest_switch_msr){
.msr = MSR_PEBS_DATA_CFG,
.host = cpuc->active_pebs_data_cfg,
- .guest = guest_pebs_mask ? kvm_pmu->pebs_data_cfg :
+ .guest = guest_pebs_mask ? guest_pebs->data_cfg :
cpuc->active_pebs_data_cfg,
};
}
@@ -5405,7 +5429,8 @@ static struct perf_guest_switch_msr *intel_guest_get_msrs(int *nr, void *data)
return arr;
}
-static struct perf_guest_switch_msr *core_guest_get_msrs(int *nr, void *data)
+static struct perf_guest_switch_msr *core_guest_get_msrs(int *nr,
+ struct x86_guest_pebs *guest_pebs)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct perf_guest_switch_msr *arr = cpuc->guest_switch_msrs;
@@ -6216,12 +6241,46 @@ static inline bool intel_pmu_broken_perf_cap(void)
return false;
}
-static inline void __intel_update_pmu_caps(struct pmu *pmu)
+static inline void __intel_update_pmu_xregs_caps(struct pmu *pmu)
{
struct pmu *dest_pmu = pmu ? pmu : x86_get_pmu(smp_processor_id());
- if (hybrid(pmu, arch_pebs_cap).caps & ARCH_PEBS_VECR_XMM)
- dest_pmu->capabilities |= PERF_PMU_CAP_EXTENDED_REGS;
+ /* Only support the extension when XSAVES is available. */
+ if (!boot_cpu_has(X86_FEATURE_XSAVES))
+ return;
+
+ if (!boot_cpu_has(X86_FEATURE_XMM) ||
+ !cpu_has_xfeatures(XFEATURE_MASK_SSE, NULL))
+ return;
+
+ /*
+ * On current hybrid platforms, P-cores and E-cores expose the same
+ * XSAVE feature set. Therefore, using the global x86_pmu.ext_regs_mask
+ * is sufficient to represent the hardware-supported XSAVE features.
+ */
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_SSE;
+
+ if (boot_cpu_has(X86_FEATURE_AVX) &&
+ cpu_has_xfeatures(XFEATURE_MASK_YMM, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_YMM;
+ if (boot_cpu_has(X86_FEATURE_APX) &&
+ cpu_has_xfeatures(XFEATURE_MASK_APX, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_APX;
+ if (boot_cpu_has(X86_FEATURE_AVX512F)) {
+ if (cpu_has_xfeatures(XFEATURE_MASK_OPMASK, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_OPMASK;
+ if (cpu_has_xfeatures(XFEATURE_MASK_ZMM_Hi256, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_ZMM_Hi256;
+ if (cpu_has_xfeatures(XFEATURE_MASK_Hi16_ZMM, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_Hi16_ZMM;
+ }
+ if (cpu_feature_enabled(X86_FEATURE_USER_SHSTK) &&
+ cpu_has_xfeatures(XFEATURE_MASK_CET_USER, NULL))
+ x86_pmu.ext_regs_mask |= XFEATURE_MASK_CET_USER;
+
+ dest_pmu->capabilities |= PERF_PMU_CAP_EXTENDED_REGS;
+ if (x86_pmu.ext_regs_mask > XFEATURE_MASK_SSE)
+ dest_pmu->capabilities |= PERF_PMU_CAP_SIMD_REGS;
}
static inline void __intel_update_large_pebs_flags(struct pmu *pmu)
@@ -6292,12 +6351,10 @@ static void update_pmu_cap_from_perfmonext(struct pmu *pmu)
hybrid(pmu, arch_pebs_cap).counters = pebs_mask;
hybrid(pmu, arch_pebs_cap).pdists = pdists_mask;
- if (WARN_ON((pebs_mask | pdists_mask) & ~cntrs_mask)) {
+ if (WARN_ON((pebs_mask | pdists_mask) & ~cntrs_mask))
x86_pmu.arch_pebs = 0;
- } else {
- __intel_update_pmu_caps(pmu);
+ else
__intel_update_large_pebs_flags(pmu);
- }
} else {
WARN_ON(x86_pmu.arch_pebs == 1);
x86_pmu.arch_pebs = 0;
@@ -6321,6 +6378,7 @@ static void intel_update_pmu_caps(struct pmu *pmu)
hybrid_pmu(pmu)->pmu_type == hybrid_big)
hybrid(pmu, intel_cap).perf_metrics = 1;
}
+ __intel_update_pmu_xregs_caps(pmu);
}
static void intel_pmu_check_hybrid_pmus(struct x86_hybrid_pmu *pmu)
@@ -6474,8 +6532,6 @@ static void intel_pmu_cpu_starting(int cpu)
}
}
- __intel_update_pmu_caps(cpuc->pmu);
-
if (!cpuc->shared_regs)
return;
diff --git a/arch/x86/events/intel/ds.c b/arch/x86/events/intel/ds.c
index d0329a7eb8a5..8444670cee4a 100644
--- a/arch/x86/events/intel/ds.c
+++ b/arch/x86/events/intel/ds.c
@@ -1704,6 +1704,7 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event)
u64 sample_type = attr->sample_type;
u64 pebs_data_cfg = 0;
bool gprs, tsx_weight;
+ u64 xgprs_mask;
if (!(sample_type & ~(PERF_SAMPLE_IP|PERF_SAMPLE_TIME)) &&
attr->precise_ip > 1)
@@ -1718,10 +1719,13 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event)
* + precise_ip < 2 for the non event IP
* + For RTM TSX weight we need GPRs for the abort code.
*/
+ xgprs_mask = event->attr.sample_simd_regs_enabled ?
+ PEBS_GP_REGS | BIT_ULL(PERF_REG_X86_SSP) :
+ PEBS_GP_REGS;
gprs = ((sample_type & PERF_SAMPLE_REGS_INTR) &&
- (attr->sample_regs_intr & PEBS_GP_REGS)) ||
+ (attr->sample_regs_intr & xgprs_mask)) ||
((sample_type & PERF_SAMPLE_REGS_USER) &&
- (attr->sample_regs_user & PEBS_GP_REGS));
+ (attr->sample_regs_user & xgprs_mask));
tsx_weight = (sample_type & PERF_SAMPLE_WEIGHT_TYPE) &&
((attr->config & INTEL_ARCH_EVENT_MASK) ==
@@ -1730,9 +1734,20 @@ static u64 pebs_update_adaptive_cfg(struct perf_event *event)
if (gprs || (attr->precise_ip < 2) || tsx_weight)
pebs_data_cfg |= PEBS_DATACFG_GP;
- if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
- (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK))
- pebs_data_cfg |= PEBS_DATACFG_XMMS;
+ if (sample_type & (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)) {
+ if (event_needs_xmm(event))
+ pebs_data_cfg |= PEBS_DATACFG_XMMS;
+ if (x86_pmu.arch_pebs && event_needs_ymm(event))
+ pebs_data_cfg |= PEBS_DATACFG_YMMHS;
+ if (x86_pmu.arch_pebs && event_needs_low16_zmm(event))
+ pebs_data_cfg |= PEBS_DATACFG_ZMMHS;
+ if (x86_pmu.arch_pebs && event_needs_high16_zmm(event))
+ pebs_data_cfg |= PEBS_DATACFG_H16ZMMS;
+ if (x86_pmu.arch_pebs && event_needs_opmask(event))
+ pebs_data_cfg |= PEBS_DATACFG_OPMASKS;
+ if (x86_pmu.arch_pebs && event_needs_egprs(event))
+ pebs_data_cfg |= PEBS_DATACFG_EGPRS;
+ }
if (sample_type & PERF_SAMPLE_BRANCH_STACK) {
/*
@@ -2523,7 +2538,7 @@ static void setup_pebs_adaptive_sample_data(struct perf_event *event,
return;
perf_regs = container_of(regs, struct x86_perf_regs, regs);
- perf_regs->xmm_regs = NULL;
+ x86_pmu_clear_perf_regs(regs);
format_group = basic->format_group;
@@ -2608,6 +2623,8 @@ static void setup_pebs_adaptive_sample_data(struct perf_event *event,
next_record += nr * sizeof(u64);
}
+ x86_pmu_update_perf_regs(event, data, regs, true);
+
WARN_ONCE(next_record != __pebs + basic->format_size,
"PEBS record size %u, expected %llu, config %llx\n",
basic->format_size,
@@ -2640,7 +2657,7 @@ static void setup_arch_pebs_sample_data(struct perf_event *event,
return;
perf_regs = container_of(regs, struct x86_perf_regs, regs);
- perf_regs->xmm_regs = NULL;
+ x86_pmu_clear_perf_regs(regs);
__setup_perf_sample_data(event, iregs, data);
@@ -2648,6 +2665,9 @@ static void setup_arch_pebs_sample_data(struct perf_event *event,
again:
header = at;
+ if (!header->size)
+ return;
+
next_record = at + sizeof(struct arch_pebs_header);
if (header->basic) {
struct arch_pebs_basic *basic = next_record;
@@ -2678,6 +2698,11 @@ again:
__setup_pebs_gpr_group(event, regs,
(struct pebs_gprs *)gprs,
sample_type);
+
+ /* Currently only user space mode enables SSP. */
+ if (user_mode(regs) && (sample_type &
+ (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)))
+ perf_regs->ssp = &gprs->ssp;
}
if (header->aux) {
@@ -2690,14 +2715,63 @@ again:
meminfo->tsx_tuning, ax);
}
- if (header->xmm) {
+ if (header->xmm || header->ymmh || header->egpr ||
+ header->opmask || header->zmmh || header->h16zmm) {
+ struct arch_pebs_xer_header *xer_header = next_record;
struct pebs_xmm *xmm;
+ struct ymmh_struct *ymmh;
+ struct avx_512_zmm_uppers_state *zmmh;
+ struct avx_512_hi16_state *h16zmm;
+ struct avx_512_opmask_state *opmask;
+ struct apx_state *egpr;
next_record += sizeof(struct arch_pebs_xer_header);
- xmm = next_record;
- perf_regs->xmm_regs = xmm->xmm;
- next_record = xmm + 1;
+ if (header->xmm) {
+ xmm = next_record;
+ /*
+ * Only output XMM regs to user space when arch-PEBS
+ * really writes data into xstate area.
+ */
+ if (xer_header->xstate & XFEATURE_MASK_SSE)
+ perf_regs->xmm_regs = xmm->xmm;
+ next_record = xmm + 1;
+ }
+
+ if (header->ymmh) {
+ ymmh = next_record;
+ if (xer_header->xstate & XFEATURE_MASK_YMM)
+ perf_regs->ymmh = ymmh;
+ next_record = ymmh + 1;
+ }
+
+ if (header->egpr) {
+ egpr = next_record;
+ if (xer_header->xstate & XFEATURE_MASK_APX)
+ perf_regs->egpr = egpr;
+ next_record = egpr + 1;
+ }
+
+ if (header->opmask) {
+ opmask = next_record;
+ if (xer_header->xstate & XFEATURE_MASK_OPMASK)
+ perf_regs->opmask = opmask;
+ next_record = opmask + 1;
+ }
+
+ if (header->zmmh) {
+ zmmh = next_record;
+ if (xer_header->xstate & XFEATURE_MASK_ZMM_Hi256)
+ perf_regs->zmmh = zmmh;
+ next_record = zmmh + 1;
+ }
+
+ if (header->h16zmm) {
+ h16zmm = next_record;
+ if (xer_header->xstate & XFEATURE_MASK_Hi16_ZMM)
+ perf_regs->h16zmm = h16zmm;
+ next_record = h16zmm + 1;
+ }
}
if (header->lbr) {
@@ -2742,6 +2816,8 @@ again:
at = at + header->size;
goto again;
}
+
+ x86_pmu_update_perf_regs(event, data, regs, true);
}
static inline void *
@@ -2865,13 +2941,21 @@ __intel_pmu_pebs_last_event(struct perf_event *event,
struct pt_regs *iregs,
struct pt_regs *regs,
struct perf_sample_data *data,
- void *at,
- int count,
+ void *at, int count, bool corrupted,
setup_fn setup_sample)
{
struct hw_perf_event *hwc = &event->hw;
- setup_sample(event, iregs, at, data, regs);
+ /* Skip parsing corrupted PEBS record. */
+ if (corrupted) {
+ /* Clear stale register states in previous records. */
+ memset(regs, 0, sizeof(*regs));
+ x86_pmu_clear_perf_regs(regs);
+ perf_sample_data_init(data, 0, event->hw.last_period);
+ } else {
+ setup_sample(event, iregs, at, data, regs);
+ }
+
if (iregs == &dummy_iregs) {
/*
* The PEBS records may be drained in the non-overflow context,
@@ -2889,12 +2973,16 @@ __intel_pmu_pebs_last_event(struct perf_event *event,
}
if (hwc->flags & PERF_X86_EVENT_AUTO_RELOAD) {
- if ((is_pebs_counter_event_group(event))) {
- /*
- * The value of each sample has been updated when setup
- * the corresponding sample data.
- */
- perf_event_update_userpage(event);
+ if (is_pebs_counter_event_group(event)) {
+ if (corrupted) {
+ intel_pmu_save_and_restart_reload(event, 1);
+ } else {
+ /*
+ * The value of each sample has been updated
+ * when setup the corresponding sample data.
+ */
+ perf_event_update_userpage(event);
+ }
} else {
/*
* Now, auto-reload is only enabled in fixed period mode.
@@ -2918,13 +3006,15 @@ __intel_pmu_pebs_last_event(struct perf_event *event,
* counters-snapshotting record, only needs to set the new
* period for the counter.
*/
- if (is_pebs_counter_event_group(event))
+ if (is_pebs_counter_event_group(event) && !corrupted)
static_call(x86_pmu_set_period)(event);
else
intel_pmu_save_and_restart(event);
}
}
+static DEFINE_PER_CPU(struct x86_perf_regs, x86_pebs_regs);
+
static __always_inline void
__intel_pmu_pebs_events(struct perf_event *event,
struct pt_regs *iregs,
@@ -2934,25 +3024,29 @@ __intel_pmu_pebs_events(struct perf_event *event,
setup_fn setup_sample)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
- struct x86_perf_regs perf_regs;
- struct pt_regs *regs = &perf_regs.regs;
+ struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs);
+ struct pt_regs *regs = &perf_regs->regs;
void *at = get_next_pebs_record_by_bit(base, top, bit);
int cnt = count;
+ x86_pmu_clear_perf_regs(regs);
+
if (!iregs)
iregs = &dummy_iregs;
while (cnt > 1) {
- __intel_pmu_pebs_event(event, iregs, regs, data, at, setup_sample);
+ __intel_pmu_pebs_event(event, iregs, regs, data,
+ at, setup_sample);
at += cpuc->pebs_record_size;
at = get_next_pebs_record_by_bit(at, top, bit);
cnt--;
}
- __intel_pmu_pebs_last_event(event, iregs, regs, data, at, count, setup_sample);
+ __intel_pmu_pebs_last_event(event, iregs, regs, data, at,
+ count, false, setup_sample);
}
-static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_data *data)
+static int intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_data *data)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct debug_store *ds = cpuc->ds;
@@ -2961,7 +3055,7 @@ static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_
int n;
if (!x86_pmu.pebs_active)
- return;
+ return 0;
at = (struct pebs_record_core *)(unsigned long)ds->pebs_buffer_base;
top = (struct pebs_record_core *)(unsigned long)ds->pebs_index;
@@ -2972,22 +3066,25 @@ static void intel_pmu_drain_pebs_core(struct pt_regs *iregs, struct perf_sample_
ds->pebs_index = ds->pebs_buffer_base;
if (!test_bit(0, cpuc->active_mask))
- return;
+ return 0;
WARN_ON_ONCE(!event);
if (!event->attr.precise_ip)
- return;
+ return 0;
n = top - at;
if (n <= 0) {
if (event->hw.flags & PERF_X86_EVENT_AUTO_RELOAD)
intel_pmu_save_and_restart_reload(event, 0);
- return;
+ return 0;
}
__intel_pmu_pebs_events(event, iregs, data, at, top, 0, n,
setup_pebs_fixed_sample_data);
+
+ /* PMC0 only */
+ return 1;
}
static void intel_pmu_pebs_event_update_no_drain(struct cpu_hw_events *cpuc, u64 mask)
@@ -3010,7 +3107,7 @@ static void intel_pmu_pebs_event_update_no_drain(struct cpu_hw_events *cpuc, u64
}
}
-static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_data *data)
+static int intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_data *data)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct debug_store *ds = cpuc->ds;
@@ -3019,11 +3116,12 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d
short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
short error[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
int max_pebs_events = intel_pmu_max_num_pebs(NULL);
+ u64 events_bitmap = 0;
int bit, i, size;
u64 mask;
if (!x86_pmu.pebs_active)
- return;
+ return 0;
base = (struct pebs_record_nhm *)(unsigned long)ds->pebs_buffer_base;
top = (struct pebs_record_nhm *)(unsigned long)ds->pebs_index;
@@ -3039,7 +3137,7 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d
if (unlikely(base >= top)) {
intel_pmu_pebs_event_update_no_drain(cpuc, mask);
- return;
+ return 0;
}
for (at = base; at < top; at += x86_pmu.pebs_record_size) {
@@ -3103,6 +3201,7 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d
if ((counts[bit] == 0) && (error[bit] == 0))
continue;
+ events_bitmap |= BIT_ULL(bit);
event = cpuc->events[bit];
if (WARN_ON_ONCE(!event))
continue;
@@ -3124,6 +3223,8 @@ static void intel_pmu_drain_pebs_nhm(struct pt_regs *iregs, struct perf_sample_d
setup_pebs_fixed_sample_data);
}
}
+
+ return hweight64(events_bitmap);
}
static __always_inline void
@@ -3158,39 +3259,46 @@ static __always_inline void
__intel_pmu_handle_last_pebs_record(struct pt_regs *iregs,
struct pt_regs *regs,
struct perf_sample_data *data,
- u64 mask, short *counts, void **last,
+ u64 mask, short *counts,
+ void **last, bool corrupted,
setup_fn setup_sample)
{
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct perf_event *event;
+ bool handled = false;
int bit;
for_each_set_bit(bit, (unsigned long *)&mask, X86_PMC_IDX_MAX) {
if (!counts[bit])
continue;
+ handled = true;
event = cpuc->events[bit];
-
__intel_pmu_pebs_last_event(event, iregs, regs, data, last[bit],
- counts[bit], setup_sample);
+ counts[bit], corrupted, setup_sample);
}
+ /* All records are corrupted, reset sampling period. */
+ if (!handled)
+ intel_pmu_pebs_event_update_no_drain(cpuc, mask);
}
-static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_data *data)
+static int intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_data *data)
{
short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS];
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
struct debug_store *ds = cpuc->ds;
- struct x86_perf_regs perf_regs;
- struct pt_regs *regs = &perf_regs.regs;
+ struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs);
+ struct pt_regs *regs = &perf_regs->regs;
struct pebs_basic *basic;
void *base, *at, *top;
+ u64 events_bitmap = 0;
+ bool corrupted = false;
u64 mask;
if (!x86_pmu.pebs_active)
- return;
+ return 0;
base = (struct pebs_basic *)(unsigned long)ds->pebs_buffer_base;
top = (struct pebs_basic *)(unsigned long)ds->pebs_index;
@@ -3203,7 +3311,7 @@ static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_d
if (unlikely(base >= top)) {
intel_pmu_pebs_event_update_no_drain(cpuc, mask);
- return;
+ return 0;
}
if (!iregs)
@@ -3214,36 +3322,45 @@ static void intel_pmu_drain_pebs_icl(struct pt_regs *iregs, struct perf_sample_d
u64 pebs_status;
basic = at;
+ if (WARN_ON_ONCE(!basic->format_size)) {
+ corrupted = true;
+ break;
+ }
if (basic->format_size != cpuc->pebs_record_size)
continue;
pebs_status = mask & basic->applicable_counters;
+ events_bitmap |= pebs_status;
__intel_pmu_handle_pebs_record(iregs, regs, data, at,
pebs_status, counts, last,
setup_pebs_adaptive_sample_data);
}
__intel_pmu_handle_last_pebs_record(iregs, regs, data, mask, counts, last,
- setup_pebs_adaptive_sample_data);
+ corrupted, setup_pebs_adaptive_sample_data);
+
+ return hweight64(events_bitmap);
}
-static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
- struct perf_sample_data *data)
+static int intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
+ struct perf_sample_data *data)
{
short counts[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS] = {};
void *last[INTEL_PMC_IDX_FIXED + MAX_FIXED_PEBS_EVENTS];
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
union arch_pebs_index index;
- struct x86_perf_regs perf_regs;
- struct pt_regs *regs = &perf_regs.regs;
+ struct x86_perf_regs *perf_regs = this_cpu_ptr(&x86_pebs_regs);
+ struct pt_regs *regs = &perf_regs->regs;
void *base, *at, *top;
+ u64 events_bitmap = 0;
+ bool corrupted = false;
u64 mask;
rdmsrq(MSR_IA32_PEBS_INDEX, index.whole);
if (unlikely(!index.wr)) {
intel_pmu_pebs_event_update_no_drain(cpuc, X86_PMC_IDX_MAX);
- return;
+ return 0;
}
base = cpuc->pebs_vaddr;
@@ -3271,8 +3388,10 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
header = at;
- if (WARN_ON_ONCE(!header->size))
- break;
+ if (WARN_ON_ONCE(!header->size)) {
+ corrupted = true;
+ goto done;
+ }
/* 1st fragment or single record must have basic group */
if (!header->basic) {
@@ -3282,6 +3401,7 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
basic = at + sizeof(struct arch_pebs_header);
pebs_status = mask & basic->applicable_counters;
+ events_bitmap |= pebs_status;
__intel_pmu_handle_pebs_record(iregs, regs, data, at,
pebs_status, counts, last,
setup_arch_pebs_sample_data);
@@ -3291,16 +3411,27 @@ static void intel_pmu_drain_arch_pebs(struct pt_regs *iregs,
if (!header->size)
break;
at += header->size;
+ if (WARN_ON_ONCE(at >= top)) {
+ corrupted = true;
+ goto done;
+ }
header = at;
}
/* Skip last fragment or the single record */
at += header->size;
+ if (WARN_ON_ONCE(at > top)) {
+ corrupted = true;
+ goto done;
+ }
}
+done:
__intel_pmu_handle_last_pebs_record(iregs, regs, data, mask,
- counts, last,
+ counts, last, corrupted,
setup_arch_pebs_sample_data);
+
+ return hweight64(events_bitmap);
}
static void __init intel_arch_pebs_init(void)
@@ -3402,7 +3533,6 @@ static void __init intel_ds_pebs_init(void)
x86_pmu.flags |= PMU_FL_PEBS_ALL;
x86_pmu.pebs_capable = ~0ULL;
pebs_qual = "-baseline";
- x86_get_pmu(smp_processor_id())->capabilities |= PERF_PMU_CAP_EXTENDED_REGS;
} else {
/* Only basic record supported */
x86_pmu.large_pebs_flags &=
diff --git a/arch/x86/events/intel/lbr.c b/arch/x86/events/intel/lbr.c
index 22e2a06d5786..0b446d8912e4 100644
--- a/arch/x86/events/intel/lbr.c
+++ b/arch/x86/events/intel/lbr.c
@@ -714,7 +714,7 @@ static inline bool vlbr_exclude_host(void)
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
return test_bit(INTEL_PMC_IDX_FIXED_VLBR,
- (unsigned long *)&cpuc->intel_ctrl_guest_mask);
+ (unsigned long *)&cpuc->intel_ctrl_exclude_host_mask);
}
void intel_pmu_lbr_enable_all(bool pmi)
diff --git a/arch/x86/events/perf_event.h b/arch/x86/events/perf_event.h
index fab9da78a5c7..e274802ef062 100644
--- a/arch/x86/events/perf_event.h
+++ b/arch/x86/events/perf_event.h
@@ -147,6 +147,199 @@ static inline bool is_acr_self_reload_event(struct perf_event *event)
return test_bit(hwc->idx, (unsigned long *)&hwc->config1);
}
+static inline bool __event_needs_xmm(struct perf_event *event, u64 sample_type)
+{
+ if (event->attr.sample_simd_regs_enabled) {
+ if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_XMM_QWORDS)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_simd_vec_reg_user > 0))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_simd_vec_reg_intr > 0))
+ return true;
+ } else {
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK))
+ return true;
+ }
+
+ return false;
+}
+
+static inline bool event_needs_xmm(struct perf_event *event)
+{
+ return __event_needs_xmm(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_ymm(struct perf_event *event, u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+ if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_YMM_QWORDS)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_simd_vec_reg_user > 0))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_simd_vec_reg_intr > 0))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_ymm(struct perf_event *event)
+{
+ return __event_needs_ymm(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_low16_zmm(struct perf_event *event,
+ u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+ if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_simd_vec_reg_user > 0))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_simd_vec_reg_intr > 0))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_low16_zmm(struct perf_event *event)
+{
+ return __event_needs_low16_zmm(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_high16_zmm(struct perf_event *event,
+ u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+ if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (fls64(event->attr.sample_simd_vec_reg_user) > PERF_X86_H16ZMM_BASE))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (fls64(event->attr.sample_simd_vec_reg_intr) > PERF_X86_H16ZMM_BASE))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_high16_zmm(struct perf_event *event)
+{
+ return __event_needs_high16_zmm(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_opmask(struct perf_event *event,
+ u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+ if (event->attr.sample_simd_pred_reg_qwords != PERF_X86_OPMASK_QWORDS)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_simd_pred_reg_user > 0))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_simd_pred_reg_intr > 0))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_opmask(struct perf_event *event)
+{
+ return __event_needs_opmask(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_egprs(struct perf_event *event,
+ u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_regs_user & PERF_X86_EGPRS_MASK))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_regs_intr & PERF_X86_EGPRS_MASK))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_egprs(struct perf_event *event)
+{
+ return __event_needs_egprs(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_ssp(struct perf_event *event,
+ u64 sample_type)
+{
+ if (!event->attr.sample_simd_regs_enabled)
+ return false;
+
+ if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+ (event->attr.sample_regs_user & BIT_ULL(PERF_REG_X86_SSP)))
+ return true;
+
+ if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (event->attr.sample_regs_intr & BIT_ULL(PERF_REG_X86_SSP)))
+ return true;
+
+ return false;
+}
+
+static inline bool event_needs_ssp(struct perf_event *event)
+{
+ return __event_needs_ssp(event,
+ PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
struct amd_nb {
int nb_id; /* NorthBridge id */
int refcnt; /* reference count */
@@ -347,8 +540,8 @@ struct cpu_hw_events {
/*
* Intel host/guest exclude bits
*/
- u64 intel_ctrl_guest_mask;
- u64 intel_ctrl_host_mask;
+ u64 intel_ctrl_exclude_host_mask;
+ u64 intel_ctrl_exclude_guest_mask;
struct perf_guest_switch_msr guest_switch_msrs[X86_PMC_IDX_MAX];
/*
@@ -948,7 +1141,7 @@ struct x86_pmu {
int pebs_record_size;
int pebs_buffer_size;
u64 pebs_events_mask;
- void (*drain_pebs)(struct pt_regs *regs, struct perf_sample_data *data);
+ int (*drain_pebs)(struct pt_regs *regs, struct perf_sample_data *data);
struct event_constraint *pebs_constraints;
void (*pebs_aliases)(struct perf_event *event);
u64 (*pebs_latency_data)(struct perf_event *event, u64 status);
@@ -1025,9 +1218,16 @@ struct x86_pmu {
unsigned int flags;
/*
+ * Extended regs, e.g., vector registers
+ * Utilize the same format as the XFEATURE_MASK_*
+ */
+ u64 ext_regs_mask;
+
+ /*
* Intel host/guest support (KVM)
*/
- struct perf_guest_switch_msr *(*guest_get_msrs)(int *nr, void *data);
+ struct perf_guest_switch_msr *(*guest_get_msrs)(int *nr,
+ struct x86_guest_pebs *guest_pebs);
/*
* Check period value for PERF_EVENT_IOC_PERIOD ioctl.
@@ -1046,7 +1246,7 @@ struct x86_pmu {
* unique capabilities.
*/
int num_hybrid_pmus;
- struct x86_hybrid_pmu *hybrid_pmu;
+ struct x86_hybrid_pmu *hybrid_pmu __counted_by_ptr(num_hybrid_pmus);
enum intel_cpu_type (*get_hybrid_cpu_type) (void);
};
@@ -1311,6 +1511,13 @@ void x86_pmu_enable_event(struct perf_event *event);
int x86_pmu_handle_irq(struct pt_regs *regs);
+void x86_pmu_clear_perf_regs(struct pt_regs *regs);
+
+void x86_pmu_update_perf_regs(struct perf_event *event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs,
+ bool from_pebs);
+
void x86_pmu_show_pmu_cap(struct pmu *pmu);
static inline int x86_pmu_num_counters(struct pmu *pmu)
diff --git a/arch/x86/include/asm/cpufeatures.h b/arch/x86/include/asm/cpufeatures.h
index f70ee74b5f92..bce3cc6bbadd 100644
--- a/arch/x86/include/asm/cpufeatures.h
+++ b/arch/x86/include/asm/cpufeatures.h
@@ -76,7 +76,7 @@
#define X86_FEATURE_K8 ( 3*32+ 4) /* Opteron, Athlon64 */
#define X86_FEATURE_ZEN5 ( 3*32+ 5) /* CPU based on Zen5 microarchitecture */
#define X86_FEATURE_ZEN6 ( 3*32+ 6) /* CPU based on Zen6 microarchitecture */
-/* Free ( 3*32+ 7) */
+#define X86_FEATURE_RMPOPT ( 3*32+ 7) /* Support for AMD RMPOPT instruction */
#define X86_FEATURE_CONSTANT_TSC ( 3*32+ 8) /* "constant_tsc" TSC ticks at a constant rate */
/* free: was #define X86_FEATURE_UP ( 3*32+ 9) * "up" SMP kernel running on UP */
#define X86_FEATURE_ART ( 3*32+10) /* "art" Always running timer (ART) */
@@ -430,6 +430,7 @@
#define X86_FEATURE_SUCCOR (17*32+ 1) /* "succor" Uncorrectable error containment and recovery */
#define X86_FEATURE_CPPC_PERF_PRIO (17*32+ 2) /* CPPC Floor Perf support */
#define X86_FEATURE_SMCA (17*32+ 3) /* "smca" Scalable MCA */
+#define X86_FEATURE_BTB_CTX_ISOLATION (17*32+ 4) /* AMD: Branch predictions contexts isolated */
/* Intel-defined CPU features, CPUID level 0x00000007:0 (EDX), word 18 */
#define X86_FEATURE_AVX512_4VNNIW (18*32+ 2) /* "avx512_4vnniw" AVX-512 Neural Network Instructions */
@@ -483,6 +484,8 @@
#define X86_FEATURE_AUTOIBRS (20*32+ 8) /* Automatic IBRS */
#define X86_FEATURE_NO_SMM_CTL_MSR (20*32+ 9) /* SMM_CTL MSR is not present */
+#define X86_FEATURE_L2_TLB_SIZE_X32 (20*32+14) /* L2 TLB sizes are encoded as multiples of 32 */
+
#define X86_FEATURE_GP_ON_USER_CPUID (20*32+17) /* User CPUID faulting */
#define X86_FEATURE_PREFETCHI (20*32+20) /* Prefetch Data/Instruction to Cache Level */
diff --git a/arch/x86/include/asm/cpuid/api.h b/arch/x86/include/asm/cpuid/api.h
index 82eddfa2347b..2d9f3d4d63de 100644
--- a/arch/x86/include/asm/cpuid/api.h
+++ b/arch/x86/include/asm/cpuid/api.h
@@ -204,7 +204,7 @@ static inline u32 cpuid_base_hypervisor(const char *sig, u32 leaves)
* from PVH early boot code before instrumentation is set up
* and memcmp() itself may be instrumented.
*/
- if (!__builtin_memcmp(sig, signature, 12) &&
+ if (!__inline_memcmp(sig, signature, 12) &&
(leaves == 0 || ((eax - base) >= leaves)))
return base;
}
diff --git a/arch/x86/include/asm/fpu/regset.h b/arch/x86/include/asm/fpu/regset.h
index 697b77e96025..433720990f5d 100644
--- a/arch/x86/include/asm/fpu/regset.h
+++ b/arch/x86/include/asm/fpu/regset.h
@@ -9,10 +9,8 @@
extern user_regset_active_fn regset_fpregs_active, regset_xregset_fpregs_active,
ssp_active;
-extern user_regset_get2_fn fpregs_get, xfpregs_get, fpregs_soft_get,
- xstateregs_get, ssp_get;
-extern user_regset_set_fn fpregs_set, xfpregs_set, fpregs_soft_set,
- xstateregs_set, ssp_set;
+extern user_regset_get2_fn fpregs_get, xfpregs_get, xstateregs_get, ssp_get;
+extern user_regset_set_fn fpregs_set, xfpregs_set, xstateregs_set, ssp_set;
/*
* xstateregs_active == regset_fpregs_active. Please refer to the comment
diff --git a/arch/x86/include/asm/fpu/sched.h b/arch/x86/include/asm/fpu/sched.h
index 89004f4ca208..67b0dfa3530f 100644
--- a/arch/x86/include/asm/fpu/sched.h
+++ b/arch/x86/include/asm/fpu/sched.h
@@ -10,6 +10,8 @@
#include <asm/trace/fpu.h>
extern void save_fpregs_to_fpstate(struct fpu *fpu);
+extern void update_fpu_state_and_flag(struct fpu *fpu,
+ struct task_struct *task);
extern void fpu__drop(struct task_struct *tsk);
extern int fpu_clone(struct task_struct *dst, u64 clone_flags, bool minimal,
unsigned long shstk_addr);
@@ -32,12 +34,10 @@ extern void fpu_flush_thread(void);
static inline void switch_fpu(struct task_struct *old, int cpu)
{
if (!test_tsk_thread_flag(old, TIF_NEED_FPU_LOAD) &&
- cpu_feature_enabled(X86_FEATURE_FPU) &&
!(old->flags & (PF_KTHREAD | PF_USER_WORKER))) {
struct fpu *old_fpu = x86_task_fpu(old);
- set_tsk_thread_flag(old, TIF_NEED_FPU_LOAD);
- save_fpregs_to_fpstate(old_fpu);
+ update_fpu_state_and_flag(old_fpu, old);
/*
* The save operation preserved register state, so the
* fpu_fpregs_owner_ctx is still @old_fpu. Store the
diff --git a/arch/x86/include/asm/fpu/types.h b/arch/x86/include/asm/fpu/types.h
index 93e99d2583d6..ea02832f7043 100644
--- a/arch/x86/include/asm/fpu/types.h
+++ b/arch/x86/include/asm/fpu/types.h
@@ -75,30 +75,6 @@ struct fxregs_state {
#define MXCSR_AND_FLAGS_SIZE sizeof(u64)
/*
- * Software based FPU emulation state. This is arbitrary really,
- * it matches the x87 format to make it easier to understand:
- */
-struct swregs_state {
- u32 cwd;
- u32 swd;
- u32 twd;
- u32 fip;
- u32 fcs;
- u32 foo;
- u32 fos;
- /* 8*10 bytes for each FP-reg = 80 bytes: */
- u32 st_space[20];
- u8 ftop;
- u8 changed;
- u8 lookahead;
- u8 no_update;
- u8 rm;
- u8 alimit;
- struct math_emu_info *info;
- u32 entry_eip;
-};
-
-/*
* List of XSAVE features Linux knows about:
*/
enum xfeature {
@@ -369,7 +345,6 @@ struct xregs_state {
union fpregs_state {
struct fregs_state fsave;
struct fxregs_state fxsave;
- struct swregs_state soft;
struct xregs_state xsave;
u8 __padding[PAGE_SIZE];
};
diff --git a/arch/x86/include/asm/fpu/xstate.h b/arch/x86/include/asm/fpu/xstate.h
index 7a7dc9d56027..19dec5f0b1c7 100644
--- a/arch/x86/include/asm/fpu/xstate.h
+++ b/arch/x86/include/asm/fpu/xstate.h
@@ -110,6 +110,9 @@ int xfeature_size(int xfeature_nr);
void xsaves(struct xregs_state *xsave, u64 mask);
void xrstors(struct xregs_state *xsave, u64 mask);
+void xsaves_nmi(struct xregs_state *xsave, u64 mask);
+
+unsigned int xstate_calculate_size(u64 xfeatures, bool compacted);
int xfd_enable_feature(u64 xfd_err);
diff --git a/arch/x86/include/asm/kvm-x86-ops.h b/arch/x86/include/asm/kvm-x86-ops.h
index e213c9ae3e30..5c358c40eae8 100644
--- a/arch/x86/include/asm/kvm-x86-ops.h
+++ b/arch/x86/include/asm/kvm-x86-ops.h
@@ -99,6 +99,7 @@ KVM_X86_OP_OPTIONAL_RET0(tdp_has_smep)
KVM_X86_OP(load_mmu_pgd)
KVM_X86_OP_OPTIONAL_RET0(set_external_spte)
KVM_X86_OP_OPTIONAL(free_external_spt)
+KVM_X86_OP_OPTIONAL_RET0(topup_external_cache)
KVM_X86_OP(has_wbinvd_exit)
KVM_X86_OP(get_l2_tsc_offset)
KVM_X86_OP(get_l2_tsc_multiplier)
diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_host.h
index 683bb8bf43a9..20b9db3f5203 100644
--- a/arch/x86/include/asm/kvm_host.h
+++ b/arch/x86/include/asm/kvm_host.h
@@ -609,15 +609,6 @@ struct kvm_pmu {
u64 pebs_data_cfg_rsvd;
/*
- * If a guest counter is cross-mapped to host counter with different
- * index, its PEBS capability will be temporarily disabled.
- *
- * The user should make sure that this mask is updated
- * after disabling interrupts and before perf_guest_get_msrs();
- */
- u64 host_cross_mapped_mask;
-
- /*
* The gate to release perf_events not marked in
* pmc_in_use only once in a vcpu time slice.
*/
@@ -1644,6 +1635,7 @@ struct kvm_x86_ops {
/* Update external page tables for page table about to be freed. */
void (*free_external_spt)(struct kvm *kvm, struct kvm_mmu_page *sp);
+ int (*topup_external_cache)(struct kvm_vcpu *vcpu, int min_nr_spts);
bool (*has_wbinvd_exit)(void);
diff --git a/arch/x86/include/asm/local.h b/arch/x86/include/asm/local.h
index 4957018fef3e..68af7b74450a 100644
--- a/arch/x86/include/asm/local.h
+++ b/arch/x86/include/asm/local.h
@@ -32,14 +32,14 @@ static inline void local_add(long i, local_t *l)
{
asm volatile(_ASM_ADD "%1,%0"
: "+m" (l->a.counter)
- : "ir" (i));
+ : "er" (i));
}
static inline void local_sub(long i, local_t *l)
{
asm volatile(_ASM_SUB "%1,%0"
: "+m" (l->a.counter)
- : "ir" (i));
+ : "er" (i));
}
/**
diff --git a/arch/x86/include/asm/math_emu.h b/arch/x86/include/asm/math_emu.h
deleted file mode 100644
index 3c42743083ed..000000000000
--- a/arch/x86/include/asm/math_emu.h
+++ /dev/null
@@ -1,15 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 */
-#ifndef _ASM_X86_MATH_EMU_H
-#define _ASM_X86_MATH_EMU_H
-
-#include <asm/ptrace.h>
-
-/* This structure matches the layout of the data saved to the stack
- following a device-not-present interrupt, part of it saved
- automatically by the 80386/80486.
- */
-struct math_emu_info {
- long ___orig_eip;
- struct pt_regs *regs;
-};
-#endif /* _ASM_X86_MATH_EMU_H */
diff --git a/arch/x86/include/asm/msr-index.h b/arch/x86/include/asm/msr-index.h
index 3a8e51a0c9e8..82ca6356dc62 100644
--- a/arch/x86/include/asm/msr-index.h
+++ b/arch/x86/include/asm/msr-index.h
@@ -350,6 +350,13 @@
#define ARCH_PEBS_LBR_SHIFT 40
#define ARCH_PEBS_LBR (0x3ull << ARCH_PEBS_LBR_SHIFT)
#define ARCH_PEBS_VECR_XMM BIT_ULL(49)
+#define ARCH_PEBS_VECR_YMMH BIT_ULL(50)
+#define ARCH_PEBS_VECR_EGPRS BIT_ULL(51)
+#define ARCH_PEBS_VECR_OPMASK BIT_ULL(53)
+#define ARCH_PEBS_VECR_ZMMH BIT_ULL(54)
+#define ARCH_PEBS_VECR_H16ZMM BIT_ULL(55)
+#define ARCH_PEBS_VECR_EXT_SHIFT 49
+#define ARCH_PEBS_VECR_EXT (0x7full << ARCH_PEBS_VECR_EXT_SHIFT)
#define ARCH_PEBS_GPR BIT_ULL(61)
#define ARCH_PEBS_AUX BIT_ULL(62)
#define ARCH_PEBS_EN BIT_ULL(63)
@@ -761,6 +768,9 @@
#define MSR_AMD64_SEG_RMP_ENABLED_BIT 0
#define MSR_AMD64_SEG_RMP_ENABLED BIT_ULL(MSR_AMD64_SEG_RMP_ENABLED_BIT)
#define MSR_AMD64_RMP_SEGMENT_SHIFT(x) (((x) & GENMASK_ULL(13, 8)) >> 8)
+#define MSR_AMD64_RMPOPT_BASE 0xc0010139
+#define MSR_AMD64_RMPOPT_ENABLE_BIT 0
+#define MSR_AMD64_RMPOPT_ENABLE BIT_ULL(MSR_AMD64_RMPOPT_ENABLE_BIT)
#define MSR_SVSM_CAA 0xc001f000
diff --git a/arch/x86/include/asm/perf_event.h b/arch/x86/include/asm/perf_event.h
index 1eb13673e889..5c92d43bbef8 100644
--- a/arch/x86/include/asm/perf_event.h
+++ b/arch/x86/include/asm/perf_event.h
@@ -150,6 +150,11 @@
#define PEBS_DATACFG_LBRS BIT_ULL(3)
#define PEBS_DATACFG_CNTR BIT_ULL(4)
#define PEBS_DATACFG_METRICS BIT_ULL(5)
+#define PEBS_DATACFG_YMMHS BIT_ULL(6)
+#define PEBS_DATACFG_OPMASKS BIT_ULL(7)
+#define PEBS_DATACFG_ZMMHS BIT_ULL(8)
+#define PEBS_DATACFG_H16ZMMS BIT_ULL(9)
+#define PEBS_DATACFG_EGPRS BIT_ULL(10)
#define PEBS_DATACFG_LBR_SHIFT 24
#define PEBS_DATACFG_CNTR_SHIFT 32
#define PEBS_DATACFG_CNTR_MASK GENMASK_ULL(15, 0)
@@ -547,7 +552,8 @@ struct arch_pebs_header {
rsvd3:7,
xmm:1,
ymmh:1,
- rsvd4:2,
+ egpr:1,
+ rsvd4:1,
opmask:1,
zmmh:1,
h16zmm:1,
@@ -728,7 +734,32 @@ extern void perf_events_lapic_init(void);
struct pt_regs;
struct x86_perf_regs {
struct pt_regs regs;
- u64 *xmm_regs;
+ u64 abi;
+ union {
+ u64 *xmm_regs;
+ u32 *xmm_space; /* for xsaves */
+ };
+ union {
+ u64 *ymmh_regs;
+ struct ymmh_struct *ymmh;
+ };
+ union {
+ u64 *zmmh_regs;
+ struct avx_512_zmm_uppers_state *zmmh;
+ };
+ union {
+ u64 *h16zmm_regs;
+ struct avx_512_hi16_state *h16zmm;
+ };
+ union {
+ u64 *opmask_regs;
+ struct avx_512_opmask_state *opmask;
+ };
+ union {
+ u64 *egpr_regs;
+ struct apx_state *egpr;
+ };
+ u64 *ssp;
};
extern unsigned long perf_arch_instruction_pointer(struct pt_regs *regs);
@@ -788,11 +819,18 @@ extern void perf_load_guest_lvtpc(u32 guest_lvtpc);
extern void perf_put_guest_lvtpc(void);
#endif
+struct x86_guest_pebs {
+ u64 enable;
+ u64 ds_area;
+ u64 data_cfg;
+};
#if defined(CONFIG_PERF_EVENTS) && defined(CONFIG_CPU_SUP_INTEL)
-extern struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data);
+extern struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr,
+ struct x86_guest_pebs *guest_pebs);
extern void x86_perf_get_lbr(struct x86_pmu_lbr *lbr);
#else
-struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr, void *data);
+struct perf_guest_switch_msr *perf_guest_get_msrs(int *nr,
+ struct x86_guest_pebs *guest_pebs);
static inline void x86_perf_get_lbr(struct x86_pmu_lbr *lbr)
{
memset(lbr, 0, sizeof(*lbr));
diff --git a/arch/x86/include/asm/preempt.h b/arch/x86/include/asm/preempt.h
index fafb6f8cdac3..d16d9c2f7b08 100644
--- a/arch/x86/include/asm/preempt.h
+++ b/arch/x86/include/asm/preempt.h
@@ -137,43 +137,15 @@ static __always_inline bool should_resched(int preempt_offset)
extern asmlinkage void preempt_schedule(void);
extern asmlinkage void preempt_schedule_thunk(void);
-#define preempt_schedule_dynamic_enabled preempt_schedule_thunk
-#define preempt_schedule_dynamic_disabled NULL
-
extern asmlinkage void preempt_schedule_notrace(void);
extern asmlinkage void preempt_schedule_notrace_thunk(void);
-#define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace_thunk
-#define preempt_schedule_notrace_dynamic_disabled NULL
-
-#ifdef CONFIG_PREEMPT_DYNAMIC
-
-DECLARE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
-
-#define __preempt_schedule() \
-do { \
- __STATIC_CALL_MOD_ADDRESSABLE(preempt_schedule); \
- asm volatile ("call " STATIC_CALL_TRAMP_STR(preempt_schedule) : ASM_CALL_CONSTRAINT); \
-} while (0)
-
-DECLARE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
-
-#define __preempt_schedule_notrace() \
-do { \
- __STATIC_CALL_MOD_ADDRESSABLE(preempt_schedule_notrace); \
- asm volatile ("call " STATIC_CALL_TRAMP_STR(preempt_schedule_notrace) : ASM_CALL_CONSTRAINT); \
-} while (0)
-
-#else /* PREEMPT_DYNAMIC */
-
#define __preempt_schedule() \
asm volatile ("call preempt_schedule_thunk" : ASM_CALL_CONSTRAINT);
#define __preempt_schedule_notrace() \
asm volatile ("call preempt_schedule_notrace_thunk" : ASM_CALL_CONSTRAINT);
-#endif /* PREEMPT_DYNAMIC */
-
#endif /* PREEMPTION */
#undef __pc_op
diff --git a/arch/x86/include/asm/processor.h b/arch/x86/include/asm/processor.h
index ec9db0dfa0df..77488ff5eb8c 100644
--- a/arch/x86/include/asm/processor.h
+++ b/arch/x86/include/asm/processor.h
@@ -10,7 +10,7 @@ struct mm_struct;
struct io_bitmap;
struct vm86;
-#include <asm/math_emu.h>
+#include <asm/ptrace.h>
#include <asm/segment.h>
#include <asm/types.h>
#include <uapi/asm/sigcontext.h>
diff --git a/arch/x86/include/asm/sev.h b/arch/x86/include/asm/sev.h
index 9e7a077c445d..c7ea9845351e 100644
--- a/arch/x86/include/asm/sev.h
+++ b/arch/x86/include/asm/sev.h
@@ -517,6 +517,7 @@ void snp_accept_memory(phys_addr_t start, phys_addr_t end);
u64 snp_get_unsupported_features(u64 status);
u64 sev_get_status(void);
void sev_show_status(void);
+bool early_is_sevsnp_guest(void);
int prepare_pte_enc(struct pte_enc_desc *d);
void set_pte_enc_mask(pte_t *kpte, unsigned long pfn, pgprot_t new_prot);
void snp_kexec_finish(void);
@@ -626,6 +627,7 @@ static inline void snp_accept_memory(phys_addr_t start, phys_addr_t end) { }
static inline u64 snp_get_unsupported_features(u64 status) { return 0; }
static inline u64 sev_get_status(void) { return 0; }
static inline void sev_show_status(void) { }
+static inline bool early_is_sevsnp_guest(void) { return false; }
static inline int prepare_pte_enc(struct pte_enc_desc *d) { return 0; }
static inline void set_pte_enc_mask(pte_t *kpte, unsigned long pfn, pgprot_t new_prot) { }
static inline void snp_kexec_finish(void) { }
@@ -662,6 +664,7 @@ static inline void snp_leak_pages(u64 pfn, unsigned int pages)
__snp_leak_pages(pfn, pages, true);
}
int snp_prepare(void);
+void snp_enable_rmpopt(void);
void snp_shutdown(void);
#else
static inline bool snp_probe_rmptable_info(void) { return false; }
@@ -680,6 +683,7 @@ static inline void snp_leak_pages(u64 pfn, unsigned int npages) {}
static inline void kdump_sev_callback(void) { }
static inline void snp_fixup_e820_tables(void) {}
static inline int snp_prepare(void) { return -ENODEV; }
+static inline void snp_enable_rmpopt(void) {}
static inline void snp_shutdown(void) {}
#endif
diff --git a/arch/x86/include/asm/shared/string.h b/arch/x86/include/asm/shared/string.h
new file mode 100644
index 000000000000..6bef90d62a21
--- /dev/null
+++ b/arch/x86/include/asm/shared/string.h
@@ -0,0 +1,52 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef _ASM_X86_SHARED_STRING_H
+#define _ASM_X86_SHARED_STRING_H
+
+static __always_inline void *__inline_memcpy(void *to, const void *from, size_t len)
+{
+ void *ret = to;
+
+ asm volatile("rep movsb"
+ : "+D" (to), "+S" (from), "+c" (len)
+ : : "memory");
+ return ret;
+}
+
+static __always_inline void *__inline_memset(void *s, int v, size_t n)
+{
+ void *ret = s;
+
+ asm volatile("rep stosb"
+ : "+D" (s), "+c" (n)
+ : "a" ((uint8_t)v)
+ : "memory");
+ return ret;
+}
+
+/*
+ * Returns: 0 (equal)
+ * 1 (not equal)
+ *
+ * In contrast, the regular memcmp() follows glibc return value semantics.
+ */
+static __always_inline int __inline_memcmp(const void *s1, const void *s2, size_t len)
+{
+ bool diff;
+
+ /*
+ * Make sure ZF is properly set in the len==0 case because in it,
+ * RCX==0 and the REPE; CMPSB won't get executed.
+ *
+ * The "cc" clobber has no meaning anymore, just source compatibility.
+ * On x86 the flag status bits are automatically added to the clobber
+ * set when there are no =@ccXY constraints. Keep it as documentation.
+ */
+ asm volatile("test %3, %3\n\t"
+ "repe cmpsb"
+ : "=@ccnz" (diff), "+D" (s1), "+S" (s2), "+c" (len)
+ : : /* "cc", */ "memory");
+
+ return diff;
+}
+
+#endif /* _ASM_X86_SHARED_STRING_H */
diff --git a/arch/x86/include/asm/shared/tdx.h b/arch/x86/include/asm/shared/tdx.h
index f20e91d7ac35..bf000f1fb42e 100644
--- a/arch/x86/include/asm/shared/tdx.h
+++ b/arch/x86/include/asm/shared/tdx.h
@@ -143,6 +143,11 @@ struct tdx_module_args {
u64 rbx;
u64 rdi;
u64 rsi;
+ /*
+ * Leaf ABI version. Note that it gets encoded into RAX along with the
+ * leaf number.
+ */
+ u8 version;
};
/* Used to communicate with the TDX module */
@@ -171,6 +176,7 @@ static inline u64 _tdx_hypercall(u64 fn, u64 r12, u64 r13, u64 r14, u64 r15)
return __tdx_hypercall(&args);
}
+void __noreturn tdx_panic(const char *msg);
/* Called from __tdx_hypercall() for unrecoverable failure */
void __noreturn __tdx_hypercall_failed(void);
diff --git a/arch/x86/include/asm/string.h b/arch/x86/include/asm/string.h
index 9cb5aae7fba9..dbf59f0d4cca 100644
--- a/arch/x86/include/asm/string.h
+++ b/arch/x86/include/asm/string.h
@@ -8,25 +8,6 @@
# include <asm/string_64.h>
#endif
-static __always_inline void *__inline_memcpy(void *to, const void *from, size_t len)
-{
- void *ret = to;
-
- asm volatile("rep movsb"
- : "+D" (to), "+S" (from), "+c" (len)
- : : "memory");
- return ret;
-}
-
-static __always_inline void *__inline_memset(void *s, int v, size_t n)
-{
- void *ret = s;
-
- asm volatile("rep stosb"
- : "+D" (s), "+c" (n)
- : "a" ((uint8_t)v)
- : "memory");
- return ret;
-}
+#include <asm/shared/string.h>
#endif /* _ASM_X86_STRING_H */
diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h
index 831d3dda3b38..2dae70c2b014 100644
--- a/arch/x86/include/asm/string_64.h
+++ b/arch/x86/include/asm/string_64.h
@@ -3,7 +3,6 @@
#define _ASM_X86_STRING_64_H
#ifdef __KERNEL__
-#include <linux/jump_label.h>
/* Written 2002 by Andi Kleen */
diff --git a/arch/x86/include/asm/tdx.h b/arch/x86/include/asm/tdx.h
index 89e97d5761d8..e186dfe5bf88 100644
--- a/arch/x86/include/asm/tdx.h
+++ b/arch/x86/include/asm/tdx.h
@@ -36,6 +36,7 @@
/* Bit definitions of TDX_FEATURES0 metadata field */
#define TDX_FEATURES0_TD_PRESERVING BIT_ULL(1)
#define TDX_FEATURES0_NO_RBP_MOD BIT_ULL(18)
+#define TDX_FEATURES0_DYNAMIC_PAMT BIT_ULL(36)
#ifndef __ASSEMBLER__
@@ -118,12 +119,34 @@ static inline bool tdx_supports_runtime_update(const struct tdx_sys_info *sysinf
return sysinfo->features.tdx_features0 & TDX_FEATURES0_TD_PRESERVING;
}
+bool tdx_supports_dynamic_pamt(const struct tdx_sys_info *sysinfo);
+
+/* Simple structure for pre-allocating DPAMT pages outside of spinlocks. */
+struct tdx_pamt_cache {
+ struct list_head page_list;
+ int cnt;
+};
+
+static inline void tdx_init_pamt_cache(struct tdx_pamt_cache *cache)
+{
+ INIT_LIST_HEAD(&cache->page_list);
+ cache->cnt = 0;
+}
+
+void tdx_free_pamt_cache(struct tdx_pamt_cache *cache);
+int tdx_topup_pamt_cache(struct tdx_pamt_cache *cache, unsigned long npages);
+int tdx_pamt_get(kvm_pfn_t pfn, struct tdx_pamt_cache *cache);
+void tdx_pamt_put(kvm_pfn_t pfn);
+
int tdx_guest_keyid_alloc(void);
u32 tdx_get_nr_guest_keyids(void);
void tdx_guest_keyid_free(unsigned int keyid);
void tdx_quirk_reset_paddr(unsigned long base, unsigned long size);
+struct page *tdx_alloc_control_page(void);
+void tdx_free_control_page(struct page *page);
+
struct tdx_td {
/* TD root structure: */
struct page *tdr_page;
diff --git a/arch/x86/include/asm/tdx_global_metadata.h b/arch/x86/include/asm/tdx_global_metadata.h
index 41150d546589..8a3cc1a2a41e 100644
--- a/arch/x86/include/asm/tdx_global_metadata.h
+++ b/arch/x86/include/asm/tdx_global_metadata.h
@@ -1,7 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
-/* Automatically generated TDX global metadata structures. */
-#ifndef _X86_VIRT_TDX_AUTO_GENERATED_TDX_GLOBAL_METADATA_H
-#define _X86_VIRT_TDX_AUTO_GENERATED_TDX_GLOBAL_METADATA_H
+/* TDX global metadata structures. */
+#ifndef _X86_VIRT_TDX_TDX_GLOBAL_METADATA_H
+#define _X86_VIRT_TDX_TDX_GLOBAL_METADATA_H
#include <linux/types.h>
@@ -21,6 +21,9 @@ struct tdx_sys_info_tdmr {
u16 pamt_4k_entry_size;
u16 pamt_2m_entry_size;
u16 pamt_1g_entry_size;
+
+ /* Optional metadata, if DPAMT is supported */
+ u8 pamt_page_bitmap_entry_bits;
};
struct tdx_sys_info_td_ctrl {
diff --git a/arch/x86/include/asm/traps.h b/arch/x86/include/asm/traps.h
index 3f24cc472ce9..e13f1025c62e 100644
--- a/arch/x86/include/asm/traps.h
+++ b/arch/x86/include/asm/traps.h
@@ -37,8 +37,6 @@ static inline int get_si_code(unsigned long condition)
return TRAP_BRKPT;
}
-void math_emulate(struct math_emu_info *);
-
bool fault_in_kernel_space(unsigned long address);
#ifdef CONFIG_VMAP_STACK
diff --git a/arch/x86/include/uapi/asm/perf_regs.h b/arch/x86/include/uapi/asm/perf_regs.h
index 7c9d2bb3833b..faaa82df688d 100644
--- a/arch/x86/include/uapi/asm/perf_regs.h
+++ b/arch/x86/include/uapi/asm/perf_regs.h
@@ -2,6 +2,8 @@
#ifndef _ASM_X86_PERF_REGS_H
#define _ASM_X86_PERF_REGS_H
+#include <linux/bits.h>
+
enum perf_event_x86_regs {
PERF_REG_X86_AX,
PERF_REG_X86_BX,
@@ -27,9 +29,35 @@ enum perf_event_x86_regs {
PERF_REG_X86_R13,
PERF_REG_X86_R14,
PERF_REG_X86_R15,
+ /*
+ * The eGPRs/SSP and XMM have overlaps. Only one can be used
+ * at a time. The ABI PERF_SAMPLE_REGS_ABI_SIMD is used to
+ * distinguish which one is used. If PERF_SAMPLE_REGS_ABI_SIMD
+ * is set, then eGPRs/SSP is used, otherwise, XMM is used.
+ *
+ * Extended GPRs (eGPRs)
+ */
+ PERF_REG_X86_R16,
+ PERF_REG_X86_R17,
+ PERF_REG_X86_R18,
+ PERF_REG_X86_R19,
+ PERF_REG_X86_R20,
+ PERF_REG_X86_R21,
+ PERF_REG_X86_R22,
+ PERF_REG_X86_R23,
+ PERF_REG_X86_R24,
+ PERF_REG_X86_R25,
+ PERF_REG_X86_R26,
+ PERF_REG_X86_R27,
+ PERF_REG_X86_R28,
+ PERF_REG_X86_R29,
+ PERF_REG_X86_R30,
+ PERF_REG_X86_R31,
+ PERF_REG_X86_SSP,
/* These are the limits for the GPRs. */
PERF_REG_X86_32_MAX = PERF_REG_X86_GS + 1,
PERF_REG_X86_64_MAX = PERF_REG_X86_R15 + 1,
+ PERF_REG_MISC_MAX = PERF_REG_X86_SSP + 1,
/* These all need two bits set because they are 128bit */
PERF_REG_X86_XMM0 = 32,
@@ -54,5 +82,30 @@ enum perf_event_x86_regs {
};
#define PERF_REG_EXTENDED_MASK (~((1ULL << PERF_REG_X86_XMM0) - 1))
+#define PERF_X86_EGPRS_MASK __GENMASK_ULL(PERF_REG_X86_R31, PERF_REG_X86_R16)
+
+enum {
+ PERF_X86_SIMD_XMM_REGS = 16,
+ PERF_X86_SIMD_YMM_REGS = 16,
+ PERF_X86_SIMD_ZMM_REGS = 32,
+ PERF_X86_SIMD_VEC_REGS_MAX = PERF_X86_SIMD_ZMM_REGS,
+
+ PERF_X86_SIMD_OPMASK_REGS = 8,
+ PERF_X86_SIMD_PRED_REGS_MAX = PERF_X86_SIMD_OPMASK_REGS,
+};
+
+#define PERF_X86_SIMD_PRED_MASK __GENMASK(PERF_X86_SIMD_PRED_REGS_MAX - 1, 0)
+#define PERF_X86_SIMD_VEC_MASK __GENMASK_ULL(PERF_X86_SIMD_VEC_REGS_MAX - 1, 0)
+
+#define PERF_X86_H16ZMM_BASE 16
+
+enum {
+ /* 1 qword = 8 bytes */
+ PERF_X86_OPMASK_QWORDS = 1,
+ PERF_X86_XMM_QWORDS = 2,
+ PERF_X86_YMM_QWORDS = 4,
+ PERF_X86_ZMM_QWORDS = 8,
+ PERF_X86_SIMD_QWORDS_MAX = PERF_X86_ZMM_QWORDS,
+};
#endif /* _ASM_X86_PERF_REGS_H */
diff --git a/arch/x86/include/uapi/asm/sigcontext.h b/arch/x86/include/uapi/asm/sigcontext.h
index d0d9b331d3a1..cff01406c0f4 100644
--- a/arch/x86/include/uapi/asm/sigcontext.h
+++ b/arch/x86/include/uapi/asm/sigcontext.h
@@ -34,6 +34,21 @@
* fpstate+extended_size-FP_XSTATE_MAGIC2_SIZE address) is set to
* FP_XSTATE_MAGIC2 so that you can sanity check your size calculations.)
*
+ * The xstate_size field indicates the actual size of the xstate context
+ * (including the 512-byte FXSAVE area and the 64-byte XSAVE header struct
+ * _header). This size is used in conjunction with the pointer to the xstate
+ * context to locate FP_XSTATE_MAGIC2.
+ *
+ * In 64-bit signal frames, the fpstate pointer points directly to the xstate
+ * context. In 32-bit signal frames (including 32-bit compat tasks on 64-bit
+ * kernels), the fpstate pointer points to struct _fpstate_32, which contains
+ * the 112-byte legacy FPU state followed by the 512-byte FXSR state (and any
+ * extended xstate), so the xstate context starts at fpstate + 112.
+ *
+ * This makes the signal frame self-describing and portable across machines
+ * with different xstate features. See Documentation/arch/x86/xstate.rst
+ * for details on signal frame portability and its architectural constraints.
+ *
* This extended area typically grows with newer CPUs that have larger and
* larger XSAVE areas.
*/
diff --git a/arch/x86/kernel/asm-offsets.c b/arch/x86/kernel/asm-offsets.c
index 081816888f7a..b3c00ff4d819 100644
--- a/arch/x86/kernel/asm-offsets.c
+++ b/arch/x86/kernel/asm-offsets.c
@@ -95,6 +95,7 @@ static void __used common(void)
OFFSET(TDX_MODULE_rbx, tdx_module_args, rbx);
OFFSET(TDX_MODULE_rdi, tdx_module_args, rdi);
OFFSET(TDX_MODULE_rsi, tdx_module_args, rsi);
+ OFFSET(TDX_MODULE_version, tdx_module_args, version);
BLANK();
OFFSET(BP_scratch, boot_params, scratch);
diff --git a/arch/x86/kernel/cpu/amd.c b/arch/x86/kernel/cpu/amd.c
index 54e14ed276b5..e5279bc648d3 100644
--- a/arch/x86/kernel/cpu/amd.c
+++ b/arch/x86/kernel/cpu/amd.c
@@ -1192,7 +1192,7 @@ static unsigned int amd_size_cache(struct cpuinfo_x86 *c, unsigned int size)
static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
{
- u32 ebx, eax, ecx, edx;
+ u32 ebx, eax, ecx, edx, shift, tmp;
u16 mask = 0xfff;
if (c->x86 < 0xf)
@@ -1201,10 +1201,12 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
if (c->extended_cpuid_level < 0x80000006)
return;
+ shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5;
+
cpuid(0x80000006, &eax, &ebx, &ecx, &edx);
- tlb_lld_4k = (ebx >> 16) & mask;
- tlb_lli_4k = ebx & mask;
+ tlb_lld_4k = ((ebx >> 16) & mask) << shift;
+ tlb_lli_4k = (ebx & mask) << shift;
/*
* K8 doesn't have 2M/4M entries in the L2 TLB so read out the L1 TLB
@@ -1216,16 +1218,18 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
}
/* Handle DTLB 2M and 4M sizes, fall back to L1 if L2 is disabled */
- if (!((eax >> 16) & mask))
+ tmp = ((eax >> 16) & mask) << shift;
+ if (!tmp)
tlb_lld_2m = (cpuid_eax(0x80000005) >> 16) & 0xff;
else
- tlb_lld_2m = (eax >> 16) & mask;
+ tlb_lld_2m = tmp;
/* a 4M entry uses two 2M entries */
tlb_lld_4m = tlb_lld_2m >> 1;
/* Handle ITLB 2M and 4M sizes, fall back to L1 if L2 is disabled */
- if (!(eax & mask)) {
+ tmp = (eax & mask) << shift;
+ if (!tmp) {
/* Erratum 658 */
if (c->x86 == 0x15 && c->x86_model <= 0x1f) {
tlb_lli_2m = 1024;
@@ -1233,8 +1237,9 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
cpuid(0x80000005, &eax, &ebx, &ecx, &edx);
tlb_lli_2m = eax & 0xff;
}
- } else
- tlb_lli_2m = eax & mask;
+ } else {
+ tlb_lli_2m = tmp;
+ }
tlb_lli_4m = tlb_lli_2m >> 1;
diff --git a/arch/x86/kernel/cpu/bugs.c b/arch/x86/kernel/cpu/bugs.c
index 56eac5611c31..1b2381da4d83 100644
--- a/arch/x86/kernel/cpu/bugs.c
+++ b/arch/x86/kernel/cpu/bugs.c
@@ -1175,6 +1175,7 @@ enum srso_mitigation {
SRSO_MITIGATION_IBPB,
SRSO_MITIGATION_IBPB_ON_VMEXIT,
SRSO_MITIGATION_BP_SPEC_REDUCE,
+ SRSO_MITIGATION_USER_IBPB,
};
static enum srso_mitigation srso_mitigation __ro_after_init = SRSO_MITIGATION_AUTO;
@@ -2908,7 +2909,8 @@ static const char * const srso_strings[] = {
[SRSO_MITIGATION_SAFE_RET] = "Mitigation: Safe RET",
[SRSO_MITIGATION_IBPB] = "Mitigation: IBPB",
[SRSO_MITIGATION_IBPB_ON_VMEXIT] = "Mitigation: IBPB on VMEXIT only",
- [SRSO_MITIGATION_BP_SPEC_REDUCE] = "Mitigation: Reduced Speculation"
+ [SRSO_MITIGATION_BP_SPEC_REDUCE] = "Mitigation: Reduced Speculation",
+ [SRSO_MITIGATION_USER_IBPB] = "Mitigation: IBPB on context switch",
};
static int __init srso_parse_cmdline(char *str)
@@ -2948,7 +2950,9 @@ static void __init srso_select_mitigation(void)
* required. Otherwise the 'microcode' mitigation is sufficient
* to protect the user->user and guest->guest vectors.
*/
- if (cpu_attack_vector_mitigated(CPU_MITIGATE_GUEST_HOST) ||
+ if ((cpu_attack_vector_mitigated(CPU_MITIGATE_GUEST_HOST) &&
+ !boot_cpu_has(X86_FEATURE_BTB_CTX_ISOLATION))
+ ||
(cpu_attack_vector_mitigated(CPU_MITIGATE_USER_KERNEL) &&
!boot_cpu_has(X86_FEATURE_SRSO_USER_KERNEL_NO))) {
srso_mitigation = SRSO_MITIGATION_SAFE_RET;
@@ -3024,6 +3028,16 @@ static void __init srso_update_mitigation(void)
boot_cpu_has(X86_FEATURE_IBPB_BRTYPE))
srso_mitigation = SRSO_MITIGATION_IBPB;
+ /*
+ * See if IBPB on context switch is the only thing needed to address
+ * GUEST/GUEST and USER/USER vectors.
+ */
+ if (srso_mitigation == SRSO_MITIGATION_MICROCODE &&
+ boot_cpu_has(X86_FEATURE_SRSO_USER_KERNEL_NO) &&
+ boot_cpu_has(X86_FEATURE_BTB_CTX_ISOLATION) &&
+ spectre_v2_user_ibpb != SPECTRE_V2_USER_NONE)
+ srso_mitigation = SRSO_MITIGATION_USER_IBPB;
+
pr_info("%s\n", srso_strings[srso_mitigation]);
}
diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c
index c7352827f491..7d0b9bdc64cc 100644
--- a/arch/x86/kernel/cpu/common.c
+++ b/arch/x86/kernel/cpu/common.c
@@ -857,7 +857,7 @@ static void get_model_name(struct cpuinfo_x86 *c)
void cpu_detect_cache_sizes(struct cpuinfo_x86 *c)
{
- unsigned int n, dummy, ebx, ecx, edx, l2size;
+ unsigned int n, dummy, ebx, ecx, edx, l2size, shift __maybe_unused;
n = c->extended_cpuid_level;
@@ -877,7 +877,9 @@ void cpu_detect_cache_sizes(struct cpuinfo_x86 *c)
l2size = ecx >> 16;
#ifdef CONFIG_X86_64
+ shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5;
c->x86_tlbsize += ((ebx >> 16) & 0xfff) + (ebx & 0xfff);
+ c->x86_tlbsize <<= shift;
#else
/* do processor-specific cache resizing */
if (this_cpu->legacy_cache_size)
@@ -1427,7 +1429,7 @@ static bool __init vulnerable_to_its(u64 x86_arch_cap_msr)
return false;
}
-static struct x86_cpu_id cpu_latest_microcode[] = {
+static const struct x86_cpu_id cpu_latest_microcode[] __initconst = {
#include "microcode/intel-ucode-defs.h"
{}
};
@@ -1782,25 +1784,50 @@ static void __init cpu_parse_early_param(void)
}
}
+static void init_cpu_info(struct cpuinfo_x86 *c)
+{
+ c->x86_cache_size = 0;
+ c->x86_vendor = X86_VENDOR_UNKNOWN;
+ c->x86_model = c->x86_stepping = 0; /* So far unknown... */
+ c->x86_vendor_id[0] = '\0'; /* Unset */
+ c->x86_model_id[0] = '\0'; /* Unset */
+#ifdef CONFIG_X86_64
+ c->x86_clflush_size = 64;
+ c->x86_phys_bits = 36;
+ c->x86_virt_bits = 48;
+#else
+ c->cpuid_level = -1; /* CPUID not detected */
+ c->x86_clflush_size = 32;
+ c->x86_phys_bits = 32;
+ c->x86_virt_bits = 32;
+#endif
+ c->x86_cache_alignment = c->x86_clflush_size;
+ memset(&c->x86_capability, 0, sizeof(c->x86_capability));
+ memset(&c->cpuid, 0, sizeof(c->cpuid));
+#ifdef CONFIG_X86_VMX_FEATURE_NAMES
+ memset(&c->vmx_capability, 0, sizeof(c->vmx_capability));
+#endif
+ c->extended_cpuid_level = 0;
+}
+
/*
* Do minimum CPU detection early.
* Fields really needed: vendor, cpuid_level, family, model, mask,
* cache alignment.
- * The others are not touched to avoid unwanted side effects.
+ * The others are reset to their defaults here and only filled in later,
+ * by identify_cpu().
*
* WARNING: this function is only called on the boot CPU. Don't add code
* here that is supposed to run on all CPUs.
*/
static void __init early_identify_cpu(struct cpuinfo_x86 *c)
{
- memset(&c->x86_capability, 0, sizeof(c->x86_capability));
- memset(&c->cpuid, 0, sizeof(c->cpuid));
- c->extended_cpuid_level = 0;
+ init_cpu_info(c);
if (!cpuid_feature())
identify_cpu_without_cpuid(c);
- /* cyrix could have cpuid enabled via c_identify()*/
+ /* Cyrix could have CPUID enabled via c_identify(). */
if (cpuid_feature()) {
cpuid_scan_cpu(c);
cpu_detect(c);
@@ -1964,16 +1991,21 @@ void check_null_seg_clears_base(struct cpuinfo_x86 *c)
set_cpu_bug(c, X86_BUG_NULL_SEG);
}
-static void generic_identify(struct cpuinfo_x86 *c)
+/*
+ * This does the hard work of actually picking apart the CPU stuff...
+ */
+static void identify_cpu(struct cpuinfo_x86 *c)
{
- c->extended_cpuid_level = 0;
+ int i;
+
+ c->loops_per_jiffy = loops_per_jiffy;
if (!cpuid_feature())
identify_cpu_without_cpuid(c);
- /* cyrix could have cpuid enabled via c_identify()*/
+ /* Cyrix could have CPUID enabled via c_identify(). */
if (!cpuid_feature())
- return;
+ goto no_cpuid;
cpuid_scan_cpu(c);
cpu_detect(c);
@@ -2001,40 +2033,8 @@ static void generic_identify(struct cpuinfo_x86 *c)
#ifdef CONFIG_X86_32
set_cpu_bug(c, X86_BUG_ESPFIX);
#endif
-}
-
-/*
- * This does the hard work of actually picking apart the CPU stuff...
- */
-static void identify_cpu(struct cpuinfo_x86 *c)
-{
- int i;
-
- c->loops_per_jiffy = loops_per_jiffy;
- c->x86_cache_size = 0;
- c->x86_vendor = X86_VENDOR_UNKNOWN;
- c->x86_model = c->x86_stepping = 0; /* So far unknown... */
- c->x86_vendor_id[0] = '\0'; /* Unset */
- c->x86_model_id[0] = '\0'; /* Unset */
-#ifdef CONFIG_X86_64
- c->x86_clflush_size = 64;
- c->x86_phys_bits = 36;
- c->x86_virt_bits = 48;
-#else
- c->cpuid_level = -1; /* CPUID not detected */
- c->x86_clflush_size = 32;
- c->x86_phys_bits = 32;
- c->x86_virt_bits = 32;
-#endif
- c->x86_cache_alignment = c->x86_clflush_size;
- memset(&c->x86_capability, 0, sizeof(c->x86_capability));
- memset(&c->cpuid, 0, sizeof(c->cpuid));
-#ifdef CONFIG_X86_VMX_FEATURE_NAMES
- memset(&c->vmx_capability, 0, sizeof(c->vmx_capability));
-#endif
-
- generic_identify(c);
+no_cpuid:
cpu_parse_topology(c);
if (this_cpu->c_identify)
@@ -2126,6 +2126,9 @@ static void identify_cpu(struct cpuinfo_x86 *c)
mcheck_cpu_init(c);
numa_add_cpu(smp_processor_id());
+
+ if (IS_ENABLED(CONFIG_X86_32))
+ enable_sep_cpu();
}
/*
@@ -2163,9 +2166,6 @@ static __init void identify_boot_cpu(void)
identify_cpu(&boot_cpu_data);
if (HAS_KERNEL_IBT && cpu_feature_enabled(X86_FEATURE_IBT))
pr_info("CET detected: Indirect Branch Tracking enabled\n");
-#ifdef CONFIG_X86_32
- enable_sep_cpu();
-#endif
cpu_detect_tlb(&boot_cpu_data);
setup_cr_pinning();
@@ -2184,10 +2184,8 @@ void identify_secondary_cpu(unsigned int cpu)
*c = boot_cpu_data;
c->cpu_index = cpu;
+ init_cpu_info(c);
identify_cpu(c);
-#ifdef CONFIG_X86_32
- enable_sep_cpu();
-#endif
x86_spec_ctrl_setup_ap();
update_srbds_msr();
if (boot_cpu_has_bug(X86_BUG_GDS))
diff --git a/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h b/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h
index af8b1d8b26f6..7f463c85c5f8 100644
--- a/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h
+++ b/arch/x86/kernel/cpu/microcode/intel-ucode-defs.h
@@ -136,35 +136,35 @@
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x5f, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x3e },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x66, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x2a },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0xc0002f0 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0xd000410 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6c, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x10002e0 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6a, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0xd000421 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x6c, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x10002f1 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7a, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x42 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7a, .steppings = 0x0100, .platform_mask = 0x01, .driver_data = 0x26 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7e, .steppings = 0x0020, .platform_mask = 0x80, .driver_data = 0xca },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x7e, .steppings = 0x0020, .platform_mask = 0x80, .driver_data = 0xcc },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8a, .steppings = 0x0002, .platform_mask = 0x10, .driver_data = 0x33 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0xbc },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0004, .platform_mask = 0xc2, .driver_data = 0x3c },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8d, .steppings = 0x0002, .platform_mask = 0xc2, .driver_data = 0x56 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0xbe },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8c, .steppings = 0x0004, .platform_mask = 0xc2, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8d, .steppings = 0x0002, .platform_mask = 0xc2, .driver_data = 0x58 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0200, .platform_mask = 0x10, .driver_data = 0xf6 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0200, .platform_mask = 0xc0, .driver_data = 0xf6 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0400, .platform_mask = 0xc0, .driver_data = 0xf6 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x0800, .platform_mask = 0xd0, .driver_data = 0xf6 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8e, .steppings = 0x1000, .platform_mask = 0x94, .driver_data = 0x100 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x10, .driver_data = 0x2c000410 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x87, .driver_data = 0x2b000650 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x10, .driver_data = 0x2c000410 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0x2b000650 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x10, .driver_data = 0x2c000410 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0x2b000650 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0080, .platform_mask = 0x87, .driver_data = 0x2b000650 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x10, .driver_data = 0x2c000410 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x87, .driver_data = 0x2b000650 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x10, .driver_data = 0x2c000421 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0010, .platform_mask = 0x87, .driver_data = 0x2b000670 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x10, .driver_data = 0x2c000421 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0020, .platform_mask = 0x87, .driver_data = 0x2b000670 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x10, .driver_data = 0x2c000421 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0040, .platform_mask = 0x87, .driver_data = 0x2b000670 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0080, .platform_mask = 0x87, .driver_data = 0x2b000670 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x10, .driver_data = 0x2c000421 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x8f, .steppings = 0x0100, .platform_mask = 0x87, .driver_data = 0x2b000670 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x96, .steppings = 0x0002, .platform_mask = 0x01, .driver_data = 0x1a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x43a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x40, .driver_data = 0xb },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x80, .driver_data = 0x43a },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x97, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0008, .platform_mask = 0x80, .driver_data = 0x43b },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x40, .driver_data = 0xc },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9a, .steppings = 0x0010, .platform_mask = 0x80, .driver_data = 0x43b },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9c, .steppings = 0x0001, .platform_mask = 0x01, .driver_data = 0x24000026 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9e, .steppings = 0x0200, .platform_mask = 0x2a, .driver_data = 0xf8 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0x9e, .steppings = 0x0400, .platform_mask = 0x22, .driver_data = 0xfa },
@@ -176,30 +176,33 @@
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa5, .steppings = 0x0020, .platform_mask = 0x22, .driver_data = 0x100 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa6, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0x102 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa6, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x100 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa7, .steppings = 0x0002, .platform_mask = 0x02, .driver_data = 0x64 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaa, .steppings = 0x0010, .platform_mask = 0xe6, .driver_data = 0x25 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x20, .driver_data = 0xa000124 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x95, .driver_data = 0x10003f0 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xae, .steppings = 0x0002, .platform_mask = 0x97, .driver_data = 0x1000273 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaf, .steppings = 0x0008, .platform_mask = 0x01, .driver_data = 0x3000382 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb5, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0xa },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0002, .platform_mask = 0x32, .driver_data = 0x132 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0010, .platform_mask = 0x32, .driver_data = 0x132 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0004, .platform_mask = 0xe0, .driver_data = 0x6133 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0008, .platform_mask = 0xe0, .driver_data = 0x6133 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0100, .platform_mask = 0xe0, .driver_data = 0x6133 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbd, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x125 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbe, .steppings = 0x0001, .platform_mask = 0x19, .driver_data = 0x1e },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0040, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0080, .platform_mask = 0x07, .driver_data = 0x3d },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc5, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0010, .platform_mask = 0x82, .driver_data = 0x11a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xca, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x11a },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0002, .platform_mask = 0x87, .driver_data = 0x210002c0 },
-{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0004, .platform_mask = 0x87, .driver_data = 0x210002c0 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xa7, .steppings = 0x0002, .platform_mask = 0x02, .driver_data = 0x65 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaa, .steppings = 0x0010, .platform_mask = 0xe6, .driver_data = 0x28 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x20, .driver_data = 0xa000142 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xad, .steppings = 0x0002, .platform_mask = 0x95, .driver_data = 0x1000423 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xae, .steppings = 0x0002, .platform_mask = 0x97, .driver_data = 0x1000307 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xaf, .steppings = 0x0008, .platform_mask = 0x01, .driver_data = 0x30003a3 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb5, .steppings = 0x0001, .platform_mask = 0x80, .driver_data = 0xd },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0002, .platform_mask = 0x32, .driver_data = 0x133 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xb7, .steppings = 0x0010, .platform_mask = 0x32, .driver_data = 0x133 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0004, .platform_mask = 0xe0, .driver_data = 0x6134 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0008, .platform_mask = 0xe0, .driver_data = 0x6134 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xba, .steppings = 0x0100, .platform_mask = 0xe0, .driver_data = 0x6134 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbd, .steppings = 0x0002, .platform_mask = 0x80, .driver_data = 0x126 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbe, .steppings = 0x0001, .platform_mask = 0x19, .driver_data = 0x21 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0004, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0020, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0040, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xbf, .steppings = 0x0080, .platform_mask = 0x07, .driver_data = 0x3e },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc5, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xc6, .steppings = 0x0010, .platform_mask = 0x82, .driver_data = 0x121 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xca, .steppings = 0x0004, .platform_mask = 0x82, .driver_data = 0x121 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0002, .platform_mask = 0x90, .driver_data = 0x11b },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0004, .platform_mask = 0x90, .driver_data = 0x11b },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcc, .steppings = 0x0008, .platform_mask = 0x90, .driver_data = 0x11b },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0002, .platform_mask = 0x87, .driver_data = 0x210002e0 },
+{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0x6, .model = 0xcf, .steppings = 0x0004, .platform_mask = 0x87, .driver_data = 0x210002e0 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0080, .platform_mask = 0x01, .driver_data = 0x12 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0080, .platform_mask = 0x02, .driver_data = 0x8 },
{ .flags = X86_CPU_ID_FLAG_ENTRY_VALID, .vendor = X86_VENDOR_INTEL, .family = 0xf, .model = 0x00, .steppings = 0x0400, .platform_mask = 0x01, .driver_data = 0x13 },
diff --git a/arch/x86/kernel/cpu/mtrr/amd.c b/arch/x86/kernel/cpu/mtrr/amd.c
index a73715d6f05c..9e440e30c179 100644
--- a/arch/x86/kernel/cpu/mtrr/amd.c
+++ b/arch/x86/kernel/cpu/mtrr/amd.c
@@ -51,11 +51,10 @@ amd_get_mtrr(unsigned int reg, unsigned long *base,
/**
* amd_set_mtrr - Set variable MTRR register on the local CPU.
- *
- * @reg The register to set.
- * @base The base address of the region.
- * @size The size of the region. If this is 0 the region is disabled.
- * @type The type of the region.
+ * @reg: The register to set.
+ * @base: The base address of the region.
+ * @size: The size of the region. If this is 0 the region is disabled.
+ * @type: The type of the region.
*
* Returns nothing.
*/
diff --git a/arch/x86/kernel/cpu/resctrl/ctrlmondata.c b/arch/x86/kernel/cpu/resctrl/ctrlmondata.c
index e74f1ed54b86..62044489b052 100644
--- a/arch/x86/kernel/cpu/resctrl/ctrlmondata.c
+++ b/arch/x86/kernel/cpu/resctrl/ctrlmondata.c
@@ -16,9 +16,15 @@
#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
#include <linux/cpu.h>
+#include <linux/math.h>
#include "internal.h"
+u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val)
+{
+ return roundup(val, (unsigned long)r->membw.bw_gran);
+}
+
int resctrl_arch_update_one(struct rdt_resource *r, struct rdt_ctrl_domain *d,
u32 closid, enum resctrl_conf_type t, u32 cfg_val)
{
diff --git a/arch/x86/kernel/cpu/scattered.c b/arch/x86/kernel/cpu/scattered.c
index 8665a6474806..580b252e07cc 100644
--- a/arch/x86/kernel/cpu/scattered.c
+++ b/arch/x86/kernel/cpu/scattered.c
@@ -64,9 +64,11 @@ static const struct cpuid_bit cpuid_bits[] = {
{ X86_FEATURE_AMD_WORKLOAD_CLASS, CPUID_EAX, 22, 0x80000021, 0 },
{ X86_FEATURE_TSA_SQ_NO, CPUID_ECX, 1, 0x80000021, 0 },
{ X86_FEATURE_TSA_L1_NO, CPUID_ECX, 2, 0x80000021, 0 },
+ { X86_FEATURE_BTB_CTX_ISOLATION, CPUID_ECX, 8, 0x80000021, 0 },
{ X86_FEATURE_PERFMON_V2, CPUID_EAX, 0, 0x80000022, 0 },
{ X86_FEATURE_AMD_LBR_V2, CPUID_EAX, 1, 0x80000022, 0 },
{ X86_FEATURE_AMD_LBR_PMC_FREEZE, CPUID_EAX, 2, 0x80000022, 0 },
+ { X86_FEATURE_RMPOPT, CPUID_EDX, 0, 0x80000025, 0 },
{ X86_FEATURE_AMD_HTR_CORES, CPUID_EAX, 30, 0x80000026, 0 },
{ 0, 0, 0, 0, 0 }
};
diff --git a/arch/x86/kernel/cpu/sgx/main.c b/arch/x86/kernel/cpu/sgx/main.c
index 4505f808af5e..a5f2aabb2da1 100644
--- a/arch/x86/kernel/cpu/sgx/main.c
+++ b/arch/x86/kernel/cpu/sgx/main.c
@@ -106,7 +106,13 @@ static unsigned long __sgx_sanitize_pages(struct list_head *dirty_page_list)
left_dirty++;
}
- cond_resched();
+ /*
+ * cond_resched() only schedules when TIF_NEED_RESCHED is set.
+ * During this boot-time loop that condition may not happen for a
+ * long time, so report an RCU-Tasks quiescent state explicitly.
+ * Therefore, change cond_resched() to cond_resched_tasks_rcu_qs().
+ */
+ cond_resched_tasks_rcu_qs();
}
list_splice(&dirty, dirty_page_list);
diff --git a/arch/x86/kernel/cpu/vmware.c b/arch/x86/kernel/cpu/vmware.c
index 34b73573b108..f7ab9e7902cf 100644
--- a/arch/x86/kernel/cpu/vmware.c
+++ b/arch/x86/kernel/cpu/vmware.c
@@ -328,9 +328,9 @@ static int vmware_cpu_down_prepare(unsigned int cpu)
static __init int activate_jump_labels(void)
{
if (has_steal_clock) {
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
}
return 0;
diff --git a/arch/x86/kernel/crash.c b/arch/x86/kernel/crash.c
index e681ec9cf1dc..e6f23933a6df 100644
--- a/arch/x86/kernel/crash.c
+++ b/arch/x86/kernel/crash.c
@@ -369,9 +369,9 @@ int crash_load_segments(struct kimage *image)
* maximum CPUs and maximum memory ranges.
*/
if (IS_ENABLED(CONFIG_MEMORY_HOTPLUG))
- pnum = 2 + CONFIG_NR_CPUS_DEFAULT + CONFIG_CRASH_MAX_MEMORY_RANGES;
+ pnum = 2 + CONFIG_NR_CPUS + CONFIG_CRASH_MAX_MEMORY_RANGES;
else
- pnum += 2 + CONFIG_NR_CPUS_DEFAULT;
+ pnum += 2 + CONFIG_NR_CPUS;
if (pnum < (unsigned long)PN_XNUM) {
kbuf.memsz = pnum * sizeof(Elf64_Phdr);
@@ -430,7 +430,7 @@ unsigned int arch_crash_get_elfcorehdr_size(void)
unsigned int sz;
/* kernel_map, VMCOREINFO and maximum CPUs */
- sz = 2 + CONFIG_NR_CPUS_DEFAULT;
+ sz = 2 + CONFIG_NR_CPUS;
if (IS_ENABLED(CONFIG_MEMORY_HOTPLUG))
sz += CONFIG_CRASH_MAX_MEMORY_RANGES;
sz *= sizeof(Elf64_Phdr);
diff --git a/arch/x86/kernel/fpu/bugs.c b/arch/x86/kernel/fpu/bugs.c
index edbafc5940e3..4b84cd0a9d24 100644
--- a/arch/x86/kernel/fpu/bugs.c
+++ b/arch/x86/kernel/fpu/bugs.c
@@ -29,10 +29,6 @@ void __init fpu__init_check_bugs(void)
{
s32 fdiv_bug;
- /* kernel_fpu_begin/end() relies on patched alternative instructions. */
- if (!boot_cpu_has(X86_FEATURE_FPU))
- return;
-
kernel_fpu_begin();
/*
diff --git a/arch/x86/kernel/fpu/core.c b/arch/x86/kernel/fpu/core.c
index d1aeecd57f5e..80a7f41cf3ad 100644
--- a/arch/x86/kernel/fpu/core.c
+++ b/arch/x86/kernel/fpu/core.c
@@ -213,6 +213,19 @@ void restore_fpregs_from_fpstate(struct fpstate *fpstate, u64 mask)
}
}
+/*
+ * Save the FPU register state in fpu->fpstate->regs and set
+ * TIF_NEED_FPU_LOAD subsequently.
+ *
+ * Must be called with fpregs_lock() held, ensuring flag
+ * TIF_NEED_FPU_LOAD is set last.
+ */
+void update_fpu_state_and_flag(struct fpu *fpu, struct task_struct *task)
+{
+ save_fpregs_to_fpstate(fpu);
+ set_tsk_thread_flag(task, TIF_NEED_FPU_LOAD);
+}
+
void fpu_reset_from_exception_fixup(void)
{
restore_fpregs_from_fpstate(&init_fpstate, XFEATURE_MASK_FPSTATE);
@@ -383,13 +396,13 @@ int fpu_swap_kvm_fpstate(struct fpu_guest *guest_fpu, bool enter_guest)
/* Swap fpstate */
if (enter_guest) {
- fpu->__task_fpstate = cur_fps;
+ WRITE_ONCE(fpu->__task_fpstate, cur_fps);
+ barrier();
fpu->fpstate = guest_fps;
guest_fps->in_use = true;
} else {
guest_fps->in_use = false;
fpu->fpstate = fpu->__task_fpstate;
- fpu->__task_fpstate = NULL;
}
cur_fps = fpu->fpstate;
@@ -406,6 +419,16 @@ int fpu_swap_kvm_fpstate(struct fpu_guest *guest_fpu, bool enter_guest)
xfd_update_state(cur_fps);
}
+ /*
+ * Clear fpu->__task_fpstate after switching back to host state.
+ * A non-NULL __task_fpstate means guest state is still resident in
+ * hardware; reset it only once host state has been restored.
+ */
+ if (!enter_guest) {
+ barrier();
+ WRITE_ONCE(fpu->__task_fpstate, NULL);
+ }
+
fpregs_mark_activate();
fpregs_unlock();
return 0;
@@ -481,17 +504,15 @@ void kernel_fpu_begin_mask(unsigned int kfpu_mask)
this_cpu_write(kernel_fpu_allowed, false);
if (!(current->flags & (PF_KTHREAD | PF_USER_WORKER)) &&
- !test_thread_flag(TIF_NEED_FPU_LOAD)) {
- set_thread_flag(TIF_NEED_FPU_LOAD);
- save_fpregs_to_fpstate(x86_task_fpu(current));
- }
+ !test_thread_flag(TIF_NEED_FPU_LOAD))
+ update_fpu_state_and_flag(x86_task_fpu(current), current);
__cpu_invalidate_fpregs_state();
/* Put sane initial values into the control registers. */
if (likely(kfpu_mask & KFPU_MXCSR) && boot_cpu_has(X86_FEATURE_XMM))
ldmxcsr(MXCSR_DEFAULT);
- if (unlikely(kfpu_mask & KFPU_387) && boot_cpu_has(X86_FEATURE_FPU))
+ if (unlikely(kfpu_mask & KFPU_387))
asm volatile ("fninit");
}
EXPORT_SYMBOL_GPL(kernel_fpu_begin_mask);
@@ -672,9 +693,6 @@ int fpu_clone(struct task_struct *dst, u64 clone_flags, bool minimal,
fpstate_reset(dst_fpu);
- if (!cpu_feature_enabled(X86_FEATURE_FPU))
- return 0;
-
/*
* Enforce reload for user space tasks and prevent kernel threads
* from trying to save the FPU registers on context switch.
@@ -833,11 +851,6 @@ void fpu__clear_user_states(struct fpu *fpu)
WARN_ON_FPU(fpu != x86_task_fpu(current));
fpregs_lock();
- if (!cpu_feature_enabled(X86_FEATURE_FPU)) {
- fpu_reset_fpstate_regs();
- fpregs_unlock();
- return;
- }
/*
* Ensure that current's supervisor states are loaded into their
@@ -874,9 +887,6 @@ void fpu_flush_thread(void)
*/
void switch_fpu_return(void)
{
- if (!cpu_feature_enabled(X86_FEATURE_FPU))
- return;
-
fpregs_restore_userregs();
}
EXPORT_SYMBOL_FOR_KVM(switch_fpu_return);
diff --git a/arch/x86/kernel/fpu/init.c b/arch/x86/kernel/fpu/init.c
index 0d33c217b71c..1a92ec433f9d 100644
--- a/arch/x86/kernel/fpu/init.c
+++ b/arch/x86/kernel/fpu/init.c
@@ -31,8 +31,6 @@ static void fpu__init_cpu_generic(void)
cr0 = read_cr0();
cr0 &= ~(X86_CR0_TS|X86_CR0_EM); /* clear TS and EM */
- if (!boot_cpu_has(X86_FEATURE_FPU))
- cr0 |= X86_CR0_EM;
write_cr0(cr0);
/* Flush out any pending x87 state: */
@@ -184,9 +182,7 @@ static void __init fpu__init_system_xstate_size_legacy(void)
* Note that the size configuration might be overwritten later
* during fpu__init_system_xstate().
*/
- if (!cpu_feature_enabled(X86_FEATURE_FPU)) {
- size = sizeof(struct swregs_state);
- } else if (cpu_feature_enabled(X86_FEATURE_FXSR)) {
+ if (cpu_feature_enabled(X86_FEATURE_FXSR)) {
size = sizeof(struct fxregs_state);
fpu_user_cfg.legacy_features = XFEATURE_MASK_FPSSE;
} else {
diff --git a/arch/x86/kernel/fpu/regset.c b/arch/x86/kernel/fpu/regset.c
index 0986c2200adc..f96affd834a1 100644
--- a/arch/x86/kernel/fpu/regset.c
+++ b/arch/x86/kernel/fpu/regset.c
@@ -407,9 +407,6 @@ int fpregs_get(struct task_struct *target, const struct user_regset *regset,
sync_fpstate(fpu);
- if (!cpu_feature_enabled(X86_FEATURE_FPU))
- return fpregs_soft_get(target, regset, to);
-
if (!cpu_feature_enabled(X86_FEATURE_FXSR)) {
return membuf_write(&to, &fpu->fpstate->regs.fsave,
sizeof(struct fregs_state));
@@ -441,9 +438,6 @@ int fpregs_set(struct task_struct *target, const struct user_regset *regset,
if (pos != 0 || count != sizeof(struct user_i387_ia32_struct))
return -EINVAL;
- if (!cpu_feature_enabled(X86_FEATURE_FPU))
- return fpregs_soft_set(target, regset, pos, count, kbuf, ubuf);
-
ret = user_regset_copyin(&pos, &count, &kbuf, &ubuf, &env, 0, -1);
if (ret)
return ret;
diff --git a/arch/x86/kernel/fpu/signal.c b/arch/x86/kernel/fpu/signal.c
index 33e1284bf3e4..1f721ac84283 100644
--- a/arch/x86/kernel/fpu/signal.c
+++ b/arch/x86/kernel/fpu/signal.c
@@ -24,23 +24,24 @@
* Check for the presence of extended state information in the
* user fpstate pointer in the sigcontext.
*/
-static inline bool check_xstate_in_sigframe(struct fxregs_state __user *fxbuf,
+static inline bool check_xstate_in_sigframe(struct fxregs_state __user *buf_fx,
struct _fpx_sw_bytes *fx_sw)
{
+ struct fpstate *fpstate = x86_task_fpu(current)->fpstate;
int min_xstate_size = sizeof(struct fxregs_state) +
sizeof(struct xstate_header);
- void __user *fpstate = fxbuf;
+ void __user *buf = buf_fx;
unsigned int magic2;
- if (__copy_from_user(fx_sw, &fxbuf->sw_reserved[0], sizeof(*fx_sw)))
+ if (__copy_from_user(fx_sw, &buf_fx->sw_reserved[0], sizeof(*fx_sw)))
return false;
/* Check for the first magic field and other error scenarios. */
if (fx_sw->magic1 != FP_XSTATE_MAGIC1 ||
fx_sw->xstate_size < min_xstate_size ||
- fx_sw->xstate_size > x86_task_fpu(current)->fpstate->user_size ||
- fx_sw->xstate_size > fx_sw->extended_size)
- goto setfx;
+ fx_sw->xstate_size > fpstate->user_size ||
+ fx_sw->extended_size < fx_sw->xstate_size + FP_XSTATE_MAGIC2_SIZE)
+ goto err_setfx;
/*
* Check for the presence of second magic word at the end of memory
@@ -48,12 +49,36 @@ static inline bool check_xstate_in_sigframe(struct fxregs_state __user *fxbuf,
* fpstate layout with out copying the extended state information
* in the memory layout.
*/
- if (__get_user(magic2, (__u32 __user *)(fpstate + fx_sw->xstate_size)))
+ if (__get_user(magic2, (__u32 __user *)(buf + fx_sw->xstate_size)))
return false;
+ if (unlikely(magic2 != FP_XSTATE_MAGIC2))
+ goto err_setfx;
- if (likely(magic2 == FP_XSTATE_MAGIC2))
- return true;
-setfx:
+ if (fx_sw->xstate_size != fpstate->user_size ||
+ fx_sw->xfeatures != fpstate->user_xfeatures) {
+ unsigned int xsize;
+ u64 xfeatures;
+
+ /* Calculate size of enabled features only. */
+ xfeatures = fx_sw->xfeatures & fpstate->user_xfeatures;
+
+ xsize = xstate_calculate_size(xfeatures, false);
+ if (fx_sw->xstate_size < xsize)
+ return false;
+
+ fx_sw->xstate_size = xsize;
+ }
+
+ return true;
+err_setfx:
+ /*
+ * The fallback to FX-only state is used to preserve backward
+ * compatibility with user-space processes that are not aware of xsave
+ * states.
+ *
+ * In all other cases, returning false (to trigger SIGSEGV) is
+ * preferred to avoid silent user-space state corruption.
+ */
trace_x86_fpu_xstate_check_failed(x86_task_fpu(current));
/* Set the parameters for fx only state */
@@ -187,14 +212,6 @@ bool copy_fpstate_to_sigframe(void __user *buf, void __user *buf_fx, int size, u
ia32_fxstate &= (IS_ENABLED(CONFIG_X86_32) ||
IS_ENABLED(CONFIG_IA32_EMULATION));
- if (!cpu_feature_enabled(X86_FEATURE_FPU)) {
- struct user_i387_ia32_struct fp;
-
- fpregs_soft_get(current, NULL, (struct membuf){.p = &fp,
- .left = sizeof(fp)});
- return !copy_to_user(buf, &fp, sizeof(fp));
- }
-
if (!access_ok(buf, size))
return false;
@@ -240,15 +257,18 @@ retry:
return true;
}
-static int __restore_fpregs_from_user(void __user *buf, u64 ufeatures,
- u64 xrestore, bool fx_only)
+static int __restore_fpregs_from_user(void __user *buf, u64 task_xfeatures,
+ u64 xrestore_mask, bool fx_only)
{
if (use_xsave()) {
- u64 init_bv = ufeatures & ~xrestore;
+ u64 init_bv;
int ret;
+ /* Restore enabled features only. */
+ xrestore_mask &= task_xfeatures;
+ init_bv = task_xfeatures & ~xrestore_mask;
if (likely(!fx_only))
- ret = xrstor_from_user_sigframe(buf, xrestore);
+ ret = xrstor_from_user_sigframe(buf, xrestore_mask);
else
ret = fxrstor_from_user_sigframe(buf);
@@ -266,20 +286,19 @@ static int __restore_fpregs_from_user(void __user *buf, u64 ufeatures,
* Attempt to restore the FPU registers directly from user memory.
* Pagefaults are handled and any errors returned are fatal.
*/
-static bool restore_fpregs_from_user(void __user *buf, u64 xrestore, bool fx_only)
+static bool restore_fpregs_from_user(void __user *buf, u64 xrestore_mask,
+ bool fx_only, size_t xstate_size)
{
struct fpu *fpu = x86_task_fpu(current);
int ret;
- /* Restore enabled features only. */
- xrestore &= fpu->fpstate->user_xfeatures;
retry:
fpregs_lock();
/* Ensure that XFD is up to date */
xfd_update_state(fpu->fpstate);
pagefault_disable();
ret = __restore_fpregs_from_user(buf, fpu->fpstate->user_xfeatures,
- xrestore, fx_only);
+ xrestore_mask, fx_only);
pagefault_enable();
if (unlikely(ret)) {
@@ -302,7 +321,7 @@ retry:
if (ret != X86_TRAP_PF)
return false;
- if (!fault_in_readable(buf, fpu->fpstate->user_size))
+ if (!fault_in_readable(buf, xstate_size))
goto retry;
return false;
}
@@ -324,39 +343,33 @@ retry:
return true;
}
-static bool __fpu_restore_sig(void __user *buf, void __user *buf_fx,
- bool ia32_fxstate)
+/*
+ * Restore FPU state from a signal frame when a legacy 32-bit FP frame
+ * (buf_f) is present.
+ *
+ * The legacy FP frame duplicates the FP state portion of the FX/XSAVE
+ * frame (buf_fx). For backward compatibility, the legacy FP frame is
+ * treated as the source of truth, and its state is folded into the
+ * FX/XSAVE state before restoring the registers.
+ */
+static bool restore_from_ia32_fxstate(void __user *buf_f, void __user *buf_fx,
+ u64 xrestore_mask, bool fx_only)
{
struct task_struct *tsk = current;
struct fpu *fpu = x86_task_fpu(tsk);
struct user_i387_ia32_struct env;
- bool success, fx_only = false;
union fpregs_state *fpregs;
- u64 user_xfeatures = 0;
-
- if (use_xsave()) {
- struct _fpx_sw_bytes fx_sw_user;
-
- if (!check_xstate_in_sigframe(buf_fx, &fx_sw_user))
- return false;
-
- fx_only = !fx_sw_user.magic1;
- user_xfeatures = fx_sw_user.xfeatures;
- } else {
- user_xfeatures = XFEATURE_MASK_FPSSE;
- }
+ bool success;
- if (likely(!ia32_fxstate)) {
- /* Restore the FPU registers directly from user memory. */
- return restore_fpregs_from_user(buf_fx, user_xfeatures, fx_only);
- }
+ if (!IS_ENABLED(CONFIG_X86_32) && !IS_ENABLED(CONFIG_IA32_EMULATION))
+ return false;
/*
* Copy the legacy state because the FP portion of the FX frame has
* to be ignored for histerical raisins. The legacy state is folded
* in once the larger state has been copied.
*/
- if (__copy_from_user(&env, buf, sizeof(env)))
+ if (__copy_from_user(&env, buf_f, sizeof(env)))
return false;
/*
@@ -420,7 +433,7 @@ static bool __fpu_restore_sig(void __user *buf, void __user *buf_fx,
*
* Preserve supervisor states!
*/
- u64 mask = user_xfeatures | xfeatures_mask_supervisor();
+ u64 mask = xrestore_mask | xfeatures_mask_supervisor();
fpregs->xsave.header.xfeatures &= mask;
success = !os_xrstor_safe(fpu->fpstate,
@@ -449,10 +462,11 @@ static inline unsigned int xstate_sigframe_size(struct fpstate *fpstate)
bool fpu__restore_sig(void __user *buf, int ia32_frame)
{
struct fpu *fpu = x86_task_fpu(current);
- void __user *buf_fx = buf;
+ bool success = false, fx_only = false;
bool ia32_fxstate = false;
- bool success = false;
+ void __user *buf_fx = buf;
unsigned int size;
+ u64 xrestore_mask;
if (unlikely(!buf)) {
fpu__clear_user_states(fpu);
@@ -477,14 +491,24 @@ bool fpu__restore_sig(void __user *buf, int ia32_frame)
if (!access_ok(buf, size))
goto out;
- if (!IS_ENABLED(CONFIG_X86_64) && !cpu_feature_enabled(X86_FEATURE_FPU)) {
- success = !fpregs_soft_set(current, NULL, 0,
- sizeof(struct user_i387_ia32_struct),
- NULL, buf);
+ if (use_xsave()) {
+ struct _fpx_sw_bytes fx_sw_user;
+
+ if (!check_xstate_in_sigframe(buf_fx, &fx_sw_user))
+ goto out;
+
+ fx_only = !fx_sw_user.magic1;
+ xrestore_mask = fx_sw_user.xfeatures;
+ size = fx_sw_user.xstate_size;
} else {
- success = __fpu_restore_sig(buf, buf_fx, ia32_fxstate);
+ xrestore_mask = XFEATURE_MASK_FPSSE;
+ size = fpu->fpstate->user_size;
}
+ if (ia32_fxstate)
+ success = restore_from_ia32_fxstate(buf, buf_fx, xrestore_mask, fx_only);
+ else
+ success = restore_fpregs_from_user(buf_fx, xrestore_mask, fx_only, size);
out:
if (unlikely(!success))
fpu__clear_user_states(fpu);
diff --git a/arch/x86/kernel/fpu/xstate.c b/arch/x86/kernel/fpu/xstate.c
index a7b6524a9dea..da303714379c 100644
--- a/arch/x86/kernel/fpu/xstate.c
+++ b/arch/x86/kernel/fpu/xstate.c
@@ -587,14 +587,15 @@ static bool __init check_xstate_against_struct(int nr)
return true;
}
-static unsigned int xstate_calculate_size(u64 xfeatures, bool compacted)
+unsigned int xstate_calculate_size(u64 xfeatures, bool compacted)
{
- unsigned int topmost = fls64(xfeatures) - 1;
- unsigned int offset, i;
+ unsigned int topmost, offset, i;
- if (topmost <= XFEATURE_SSE)
+ if (!(xfeatures & ~XFEATURE_MASK_FPSSE))
return sizeof(struct xregs_state);
+ topmost = fls64(xfeatures) - 1;
+
if (compacted) {
offset = xfeature_get_offset(xfeatures, topmost);
} else {
@@ -806,18 +807,15 @@ static u64 __init guest_default_mask(void)
void __init fpu__init_system_xstate(unsigned int legacy_size)
{
unsigned int eax, ebx, ecx, edx;
- u64 xfeatures;
+ u64 xfeatures, mask;
int err;
int i;
- if (!boot_cpu_has(X86_FEATURE_FPU)) {
- pr_info("x86/fpu: No FPU detected\n");
- return;
- }
-
if (!boot_cpu_has(X86_FEATURE_XSAVE)) {
pr_info("x86/fpu: x87 FPU will use %s\n",
boot_cpu_has(X86_FEATURE_FXSR) ? "FXSAVE" : "FSAVE");
+ /* Disable all dependent flags too */
+ setup_clear_cpu_cap(X86_FEATURE_XSAVE);
return;
}
@@ -833,7 +831,8 @@ void __init fpu__init_system_xstate(unsigned int legacy_size)
cpuid_count(CPUID_LEAF_XSTATE, 1, &eax, &ebx, &ecx, &edx);
fpu_kernel_cfg.max_features |= ecx + ((u64)edx << 32);
- if ((fpu_kernel_cfg.max_features & XFEATURE_MASK_FPSSE) != XFEATURE_MASK_FPSSE) {
+ mask = XFEATURE_MASK_FPSSE;
+ if ((fpu_kernel_cfg.max_features & mask) != mask) {
/*
* This indicates that something really unexpected happened
* with the enumeration. Disable XSAVE and try to continue
@@ -844,6 +843,24 @@ void __init fpu__init_system_xstate(unsigned int legacy_size)
goto out_disable;
}
+ mask |= XFEATURE_MASK_YMM;
+ if (boot_cpu_has(X86_FEATURE_AVX)) {
+ if ((fpu_kernel_cfg.max_features & mask) != mask) {
+ pr_err(FW_BUG
+ "x86/fpu: Disabling AVX support due to missing xstate features\n");
+ setup_clear_cpu_cap(X86_FEATURE_AVX);
+ }
+ }
+
+ mask |= XFEATURE_MASK_AVX512;
+ if (boot_cpu_has(X86_FEATURE_AVX512F)) {
+ if ((fpu_kernel_cfg.max_features & mask) != mask) {
+ pr_err(FW_BUG
+ "x86/fpu: Disabling AVX-512 support due to missing xstate features\n");
+ setup_clear_cpu_cap(X86_FEATURE_AVX512F);
+ }
+ }
+
if (fpu_kernel_cfg.max_features & XFEATURE_MASK_APX &&
fpu_kernel_cfg.max_features & (XFEATURE_MASK_BNDREGS | XFEATURE_MASK_BNDCSR)) {
/*
@@ -1474,6 +1491,29 @@ void xrstors(struct xregs_state *xstate, u64 mask)
WARN_ON_ONCE(err);
}
+/**
+ * xsaves_nmi - Save selected components to a kernel xstate buffer in NMI
+ * @xstate: Pointer to the buffer
+ * @mask: Feature mask to select the components to save
+ *
+ * This function is similar to xsaves(), but should only be called within
+ * the NMI handler. This function returns the actual register contents at
+ * the moment the NMI occurs.
+ *
+ * Currently, the perf subsystem is the sole user of this helper. It uses
+ * the function to snapshot SIMD (XMM/YMM/ZMM) and APX eGPRs registers.
+ */
+void xsaves_nmi(struct xregs_state *xstate, u64 mask)
+{
+ int err;
+
+ if (!in_nmi())
+ return;
+
+ XSTATE_OP(XSAVES, xstate, (u32)mask, (u32)(mask >> 32), err);
+ WARN_ON_ONCE(err);
+}
+
#if IS_ENABLED(CONFIG_KVM)
void fpstate_clear_xstate_component(struct fpstate *fpstate, unsigned int xfeature)
{
diff --git a/arch/x86/kernel/kvm.c b/arch/x86/kernel/kvm.c
index 6b0a5861ccb8..2ce01fdb8b4b 100644
--- a/arch/x86/kernel/kvm.c
+++ b/arch/x86/kernel/kvm.c
@@ -1054,9 +1054,9 @@ const __initconst struct hypervisor_x86 x86_hyper_kvm = {
static __init int activate_jump_labels(void)
{
if (has_steal_clock) {
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (steal_acc)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
}
return 0;
diff --git a/arch/x86/kernel/perf_regs.c b/arch/x86/kernel/perf_regs.c
index 624703af80a1..f8952f7c36cc 100644
--- a/arch/x86/kernel/perf_regs.c
+++ b/arch/x86/kernel/perf_regs.c
@@ -61,11 +61,29 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
{
struct x86_perf_regs *perf_regs;
- if (idx >= PERF_REG_X86_XMM0 && idx < PERF_REG_X86_XMM_MAX) {
+ if (idx > PERF_REG_X86_R15) {
perf_regs = container_of(regs, struct x86_perf_regs, regs);
- if (!perf_regs->xmm_regs)
+ if (perf_regs->abi == PERF_SAMPLE_REGS_ABI_NONE)
return 0;
- return perf_regs->xmm_regs[idx - PERF_REG_X86_XMM0];
+
+ if (perf_regs->abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ if (idx <= PERF_REG_X86_R31) {
+ if (!perf_regs->egpr_regs)
+ return 0;
+ return perf_regs->egpr_regs[idx - PERF_REG_X86_R16];
+ }
+ if (idx == PERF_REG_X86_SSP) {
+ if (!perf_regs->ssp)
+ return 0;
+ return *perf_regs->ssp;
+ }
+ } else {
+ if (idx >= PERF_REG_X86_XMM0 && idx < PERF_REG_X86_XMM_MAX) {
+ if (!perf_regs->xmm_regs)
+ return 0;
+ return perf_regs->xmm_regs[idx - PERF_REG_X86_XMM0];
+ }
+ }
}
if (WARN_ON_ONCE(idx >= ARRAY_SIZE(pt_regs_offset)))
@@ -74,22 +92,130 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
return regs_get_register(regs, pt_regs_offset[idx]);
}
-#define PERF_REG_X86_RESERVED (((1ULL << PERF_REG_X86_XMM0) - 1) & \
- ~((1ULL << PERF_REG_X86_MAX) - 1))
+#define PERF_X86_YMMH_QWORDS (PERF_X86_YMM_QWORDS / 2)
+#define PERF_X86_ZMMH_QWORDS (PERF_X86_ZMM_QWORDS / 2)
+
+u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
+ u16 qwords_idx, bool pred)
+{
+ struct x86_perf_regs *perf_regs =
+ container_of(regs, struct x86_perf_regs, regs);
+
+ if (!(perf_regs->abi & PERF_SAMPLE_REGS_ABI_SIMD))
+ return 0;
+
+ if (pred) {
+ if (WARN_ON_ONCE(idx >= PERF_X86_SIMD_PRED_REGS_MAX ||
+ qwords_idx >= PERF_X86_OPMASK_QWORDS))
+ return 0;
+ if (!perf_regs->opmask_regs)
+ return 0;
+ return perf_regs->opmask_regs[idx];
+ }
+
+ if (WARN_ON_ONCE(idx >= PERF_X86_SIMD_VEC_REGS_MAX ||
+ qwords_idx >= PERF_X86_SIMD_QWORDS_MAX))
+ return 0;
+
+ if (idx >= PERF_X86_H16ZMM_BASE) {
+ if (!perf_regs->h16zmm_regs)
+ return 0;
+ return perf_regs->h16zmm_regs[(idx - PERF_X86_H16ZMM_BASE) *
+ PERF_X86_ZMM_QWORDS + qwords_idx];
+ }
+
+ if (qwords_idx < PERF_X86_XMM_QWORDS) {
+ if (!perf_regs->xmm_regs)
+ return 0;
+ return perf_regs->xmm_regs[idx * PERF_X86_XMM_QWORDS +
+ qwords_idx];
+ } else if (qwords_idx < PERF_X86_YMM_QWORDS) {
+ if (!perf_regs->ymmh_regs)
+ return 0;
+ return perf_regs->ymmh_regs[idx * PERF_X86_YMMH_QWORDS +
+ qwords_idx - PERF_X86_XMM_QWORDS];
+ } else if (qwords_idx < PERF_X86_ZMM_QWORDS) {
+ if (!perf_regs->zmmh_regs)
+ return 0;
+ return perf_regs->zmmh_regs[idx * PERF_X86_ZMMH_QWORDS +
+ qwords_idx - PERF_X86_YMM_QWORDS];
+ }
+
+ return 0;
+}
+
+int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
+ u16 pred_qwords, u32 pred_mask)
+{
+ unsigned long mask;
+ u64 size;
+
+ if (!vec_qwords && !pred_qwords) {
+ if (vec_mask || pred_mask)
+ return -EINVAL;
+ }
+
+ if (vec_qwords) {
+ if (vec_qwords != PERF_X86_XMM_QWORDS &&
+ vec_qwords != PERF_X86_YMM_QWORDS &&
+ vec_qwords != PERF_X86_ZMM_QWORDS)
+ return -EINVAL;
+ if (vec_mask & ~PERF_X86_SIMD_VEC_MASK)
+ return -EINVAL;
+ /* Only full-register sampling is allowed. */
+ mask = vec_mask;
+ if (vec_qwords == PERF_X86_XMM_QWORDS && mask &&
+ !bitmap_full(&mask, PERF_X86_SIMD_XMM_REGS))
+ return -EINVAL;
+ if (vec_qwords == PERF_X86_YMM_QWORDS && mask &&
+ !bitmap_full(&mask, PERF_X86_SIMD_YMM_REGS))
+ return -EINVAL;
+ if (vec_qwords == PERF_X86_ZMM_QWORDS && mask &&
+ !bitmap_full(&mask, PERF_X86_SIMD_ZMM_REGS))
+ return -EINVAL;
+ }
+
+ if (pred_qwords) {
+ if (pred_qwords != PERF_X86_OPMASK_QWORDS)
+ return -EINVAL;
+ if (pred_mask & ~PERF_X86_SIMD_PRED_MASK)
+ return -EINVAL;
+ /* Only full-register sampling is allowed. */
+ mask = pred_mask;
+ if (pred_qwords == PERF_X86_OPMASK_QWORDS && mask &&
+ !bitmap_full(&mask, PERF_X86_SIMD_OPMASK_REGS))
+ return -EINVAL;
+ }
+
+ size = sizeof(u64) * 4;
+ size += (hweight64(vec_mask) * vec_qwords +
+ hweight32(pred_mask) * pred_qwords) * sizeof(u64);
+ /*
+ * INTR_REGS and USR_REGS could be sampled simultaneously,
+ * so roughly restrict the size to half of U16_MAX.
+ */
+ if (size >= U16_MAX / 2)
+ return -EINVAL;
+
+ return 0;
+}
+
+#define PERF_REG_X86_RESERVED (GENMASK_ULL(PERF_REG_X86_XMM0 - 1, PERF_REG_X86_AX) & \
+ ~GENMASK_ULL(PERF_REG_X86_R15, PERF_REG_X86_AX))
+#define PERF_REG_X86_EXT_RESERVED (~GENMASK_ULL(PERF_REG_MISC_MAX - 1, PERF_REG_X86_AX))
#ifdef CONFIG_X86_32
-#define REG_NOSUPPORT ((1ULL << PERF_REG_X86_R8) | \
- (1ULL << PERF_REG_X86_R9) | \
- (1ULL << PERF_REG_X86_R10) | \
- (1ULL << PERF_REG_X86_R11) | \
- (1ULL << PERF_REG_X86_R12) | \
- (1ULL << PERF_REG_X86_R13) | \
- (1ULL << PERF_REG_X86_R14) | \
- (1ULL << PERF_REG_X86_R15))
-
-int perf_reg_validate(u64 mask)
+#define REG_NOSUPPORT GENMASK_ULL(PERF_REG_X86_R15, PERF_REG_X86_R8)
+
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
- if (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED)))
+ if (!simd_enabled &&
+ (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED))))
+ return -EINVAL;
+
+ /* The mask could be 0 if only the SIMD registers are interested */
+ if (simd_enabled &&
+ (mask & ~GENMASK_ULL(PERF_REG_X86_GS, PERF_REG_X86_AX)))
return -EINVAL;
return 0;
@@ -100,21 +226,21 @@ u64 perf_reg_abi(struct task_struct *task)
return PERF_SAMPLE_REGS_ABI_32;
}
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
#else /* CONFIG_X86_64 */
#define REG_NOSUPPORT ((1ULL << PERF_REG_X86_DS) | \
(1ULL << PERF_REG_X86_ES) | \
(1ULL << PERF_REG_X86_FS) | \
(1ULL << PERF_REG_X86_GS))
-int perf_reg_validate(u64 mask)
+int perf_reg_validate(u64 mask, bool simd_enabled)
{
- if (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED)))
+ if (!simd_enabled &&
+ (!mask || (mask & (REG_NOSUPPORT | PERF_REG_X86_RESERVED))))
+ return -EINVAL;
+
+ /* The mask could be 0 if only the SIMD registers are interested */
+ if (simd_enabled &&
+ (mask & (REG_NOSUPPORT | PERF_REG_X86_EXT_RESERVED)))
return -EINVAL;
return 0;
diff --git a/arch/x86/kernel/shstk.c b/arch/x86/kernel/shstk.c
index 0ca64900192f..eb690ba90180 100644
--- a/arch/x86/kernel/shstk.c
+++ b/arch/x86/kernel/shstk.c
@@ -490,7 +490,7 @@ static int wrss_control(bool enable)
* when disabling.
*/
if (!features_enabled(ARCH_SHSTK_SHSTK))
- return -EPERM;
+ return -EINVAL;
/* Already enabled/disabled? */
if (features_enabled(ARCH_SHSTK_WRSS) == enable)
diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
index a0d608e3fceb..46c033d87d39 100644
--- a/arch/x86/kvm/mmu/mmu.c
+++ b/arch/x86/kvm/mmu/mmu.c
@@ -617,6 +617,10 @@ static int mmu_topup_memory_caches(struct kvm_vcpu *vcpu, bool maybe_indirect)
PT64_ROOT_MAX_LEVEL);
if (r)
return r;
+
+ r = kvm_x86_call(topup_external_cache)(vcpu, PT64_ROOT_MAX_LEVEL);
+ if (r)
+ return r;
}
r = kvm_mmu_topup_memory_cache(&vcpu->arch.mmu_shadow_page_cache,
PT64_ROOT_MAX_LEVEL);
diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c
index 0c1ebb16cec6..1fe26778db9e 100644
--- a/arch/x86/kvm/svm/sev.c
+++ b/arch/x86/kvm/svm/sev.c
@@ -3048,6 +3048,8 @@ void sev_vm_destroy(struct kvm *kvm)
*/
if (snp_decommission_context(kvm))
return;
+
+ snp_enable_rmpopt();
} else {
sev_unbind_asid(kvm, sev->handle);
}
diff --git a/arch/x86/kvm/vmx/pmu_intel.c b/arch/x86/kvm/vmx/pmu_intel.c
index 70a8c4816135..fb7ad53855f1 100644
--- a/arch/x86/kvm/vmx/pmu_intel.c
+++ b/arch/x86/kvm/vmx/pmu_intel.c
@@ -750,24 +750,36 @@ static void intel_pmu_cleanup(struct kvm_vcpu *vcpu)
intel_pmu_release_guest_lbr_event(vcpu);
}
-void intel_pmu_cross_mapped_check(struct kvm_pmu *pmu)
+u64 __intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu)
{
- struct kvm_pmc *pmc = NULL;
+ u64 guest_pebs_enable = pmu->pebs_enable & pmu->global_ctrl;
+ u64 pebs_enable = 0;
+ struct kvm_pmc *pmc;
int bit, hw_idx;
- kvm_for_each_pmc(pmu, pmc, bit, (unsigned long *)&pmu->global_ctrl) {
- if (!pmc_is_locally_enabled(pmc) ||
- !pmc_is_globally_enabled(pmc) || !pmc->perf_event)
+ /*
+ * Omit counters that are locally disabled, don't have a perf event, or
+ * ended up with a perf event that is using a different counter than
+ * the guest, i.e. where the guest PMC is different than the host PMC
+ * being used on behalf of the guest. PEBS records include
+ * PERF_GLOBAL_STATUS, and so using a counter with a different index
+ * means the guest will see overflow status for the wrong counter(s).
+ */
+ kvm_for_each_pmc(pmu, pmc, bit, (unsigned long *)&guest_pebs_enable) {
+ if (!pmc_is_locally_enabled(pmc) || !pmc->perf_event)
continue;
/*
- * A negative index indicates the event isn't mapped to a
+ * Note, a negative index indicates the event isn't mapped to a
* physical counter in the host, e.g. due to contention.
*/
hw_idx = pmc->perf_event->hw.idx;
- if (hw_idx != pmc->idx && hw_idx > -1)
- pmu->host_cross_mapped_mask |= BIT_ULL(hw_idx);
+ if (hw_idx != pmc->idx)
+ continue;
+
+ pebs_enable |= BIT_ULL(pmc->idx);
}
+ return pebs_enable;
}
static bool intel_pmu_is_mediated_pmu_supported(struct x86_pmu_capability *host_pmu)
diff --git a/arch/x86/kvm/vmx/tdx.c b/arch/x86/kvm/vmx/tdx.c
index 00850a3a77d6..aac79fe08005 100644
--- a/arch/x86/kvm/vmx/tdx.c
+++ b/arch/x86/kvm/vmx/tdx.c
@@ -362,7 +362,7 @@ static void tdx_reclaim_control_page(struct page *ctrl_page)
if (tdx_reclaim_page(ctrl_page))
return;
- __free_page(ctrl_page);
+ tdx_free_control_page(ctrl_page);
}
struct tdx_flush_vp_arg {
@@ -589,7 +589,7 @@ static void tdx_reclaim_td_control_pages(struct kvm *kvm)
tdx_quirk_reset_paddr(page_to_phys(kvm_tdx->td.tdr_page), PAGE_SIZE);
- __free_page(kvm_tdx->td.tdr_page);
+ tdx_free_control_page(kvm_tdx->td.tdr_page);
kvm_tdx->td.tdr_page = NULL;
}
@@ -681,6 +681,8 @@ int tdx_vcpu_create(struct kvm_vcpu *vcpu)
if (!irqchip_split(vcpu->kvm))
return -EINVAL;
+ tdx_init_pamt_cache(&tdx->pamt_cache);
+
fpstate_set_confidential(&vcpu->arch.guest_fpu);
vcpu->arch.apic->guest_apic_protected = true;
INIT_LIST_HEAD(&tdx->vt.pi_wakeup_list);
@@ -866,6 +868,8 @@ void tdx_vcpu_free(struct kvm_vcpu *vcpu)
struct vcpu_tdx *tdx = to_tdx(vcpu);
int i;
+ tdx_free_pamt_cache(&tdx->pamt_cache);
+
if (vcpu->cpu != -1) {
KVM_BUG_ON(tdx->state == VCPU_TD_STATE_INITIALIZED, vcpu->kvm);
tdx_flush_vp_on_cpu(vcpu);
@@ -1618,6 +1622,17 @@ void tdx_load_mmu_pgd(struct kvm_vcpu *vcpu, hpa_t root_hpa, int pgd_level)
td_vmcs_write64(to_tdx(vcpu), SHARED_EPT_POINTER, root_hpa);
}
+static int tdx_topup_external_pamt_cache(struct kvm_vcpu *vcpu, int min_nr_spts)
+{
+ /*
+ * Minus one page to exclude the root SPT, but plus one page for a
+ * possible 4KB private mapping.
+ */
+ min_nr_spts += -1 + 1;
+
+ return tdx_topup_pamt_cache(&to_tdx(vcpu)->pamt_cache, min_nr_spts);
+}
+
static int tdx_mem_page_add(struct kvm *kvm, gfn_t gfn, enum pg_level level,
kvm_pfn_t pfn)
{
@@ -1676,16 +1691,28 @@ static struct page *tdx_spte_to_sept_pt(struct kvm *kvm, gfn_t gfn,
static int tdx_sept_map_nonleaf_spte(struct kvm *kvm, gfn_t gfn,
enum pg_level level, u64 new_spte)
{
+ struct kvm_vcpu *vcpu = kvm_get_running_vcpu();
gpa_t gpa = gfn_to_gpa(gfn);
u64 err, entry, level_state;
struct page *sept_pt;
+ int ret;
+
+ if (KVM_BUG_ON(!vcpu, kvm))
+ return -EIO;
sept_pt = tdx_spte_to_sept_pt(kvm, gfn, new_spte, level);
if (!sept_pt)
return -EIO;
+ ret = tdx_pamt_get(page_to_pfn(sept_pt), &to_tdx(vcpu)->pamt_cache);
+ if (KVM_BUG_ON(ret, kvm))
+ return ret;
+
err = tdh_mem_sept_add(&to_kvm_tdx(kvm)->td, gpa, level, sept_pt,
&entry, &level_state);
+ if (err)
+ tdx_pamt_put(page_to_pfn(sept_pt));
+
if (unlikely(tdx_operand_busy(err)))
return -EBUSY;
@@ -1698,8 +1725,13 @@ static int tdx_sept_map_nonleaf_spte(struct kvm *kvm, gfn_t gfn,
static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level level,
u64 new_spte)
{
+ struct kvm_vcpu *vcpu = kvm_get_running_vcpu();
struct kvm_tdx *kvm_tdx = to_kvm_tdx(kvm);
kvm_pfn_t pfn = spte_to_pfn(new_spte);
+ int ret;
+
+ if (KVM_BUG_ON(!vcpu, kvm))
+ return -EIO;
/* TODO: handle large pages. */
if (KVM_BUG_ON(level != PG_LEVEL_4K, kvm))
@@ -1707,6 +1739,10 @@ static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level leve
WARN_ON_ONCE((new_spte & VMX_EPT_RWX_MASK) != VMX_EPT_RWX_MASK);
+ ret = tdx_pamt_get(pfn, &to_tdx(vcpu)->pamt_cache);
+ if (KVM_BUG_ON(ret, kvm))
+ return ret;
+
/*
* Ensure pre_fault_allowed is read by kvm_arch_vcpu_pre_fault_memory()
* before kvm_tdx->state. Userspace must not be allowed to pre-fault
@@ -1719,10 +1755,15 @@ static int tdx_sept_map_leaf_spte(struct kvm *kvm, gfn_t gfn, enum pg_level leve
* If the TD isn't finalized/runnable, then userspace is initializing
* the VM image via KVM_TDX_INIT_MEM_REGION; ADD the page to the TD.
*/
- if (unlikely(kvm_tdx->state != TD_STATE_RUNNABLE))
- return tdx_mem_page_add(kvm, gfn, level, pfn);
+ if (likely(kvm_tdx->state == TD_STATE_RUNNABLE))
+ ret = tdx_mem_page_aug(kvm, gfn, level, pfn);
+ else
+ ret = tdx_mem_page_add(kvm, gfn, level, pfn);
+
+ if (ret)
+ tdx_pamt_put(pfn);
- return tdx_mem_page_aug(kvm, gfn, level, pfn);
+ return ret;
}
/*
@@ -1819,6 +1860,7 @@ static int tdx_sept_remove_leaf_spte(struct kvm *kvm, gfn_t gfn,
return -EIO;
tdx_quirk_reset_paddr(PFN_PHYS(pfn), PAGE_SIZE);
+ tdx_pamt_put(pfn);
return 0;
}
@@ -1862,6 +1904,8 @@ static int tdx_sept_set_private_spte(struct kvm *kvm, gfn_t gfn, u64 old_spte,
*/
static void tdx_sept_free_private_spt(struct kvm *kvm, struct kvm_mmu_page *sp)
{
+ struct page *sept_pt = virt_to_page(sp->external_spt);
+
/*
* KVM doesn't (yet) zap page table pages in mirror page table while
* TD is active, though guest pages mapped in mirror page table could be
@@ -1875,15 +1919,15 @@ static void tdx_sept_free_private_spt(struct kvm *kvm, struct kvm_mmu_page *sp)
* the page to prevent the kernel from accessing the encrypted page.
*/
if (KVM_BUG_ON(is_hkid_assigned(to_kvm_tdx(kvm)), kvm) ||
- tdx_reclaim_page(virt_to_page(sp->external_spt)))
+ tdx_reclaim_page(sept_pt))
goto out;
/*
- * Immediately free the S-EPT page because RCU-time free is unnecessary
- * after TDH.PHYMEM.PAGE.RECLAIM ensures there are no outstanding
- * readers.
+ * Immediately free the S-EPT page as the TDX subsystem doesn't support
+ * freeing pages from RCU callbacks, and more importantly because
+ * TDH.PHYMEM.PAGE.RECLAIM ensures there are no outstanding readers.
*/
- free_page((unsigned long)sp->external_spt);
+ tdx_free_control_page(sept_pt);
out:
sp->external_spt = NULL;
}
@@ -2456,7 +2500,7 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params,
ret = -ENOMEM;
- tdr_page = alloc_page(GFP_KERNEL_ACCOUNT);
+ tdr_page = tdx_alloc_control_page();
if (!tdr_page)
goto free_hkid;
@@ -2469,7 +2513,7 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params,
goto free_tdr;
for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++) {
- tdcs_pages[i] = alloc_page(GFP_KERNEL_ACCOUNT);
+ tdcs_pages[i] = tdx_alloc_control_page();
if (!tdcs_pages[i])
goto free_tdcs;
}
@@ -2587,10 +2631,8 @@ static int __tdx_td_init(struct kvm *kvm, struct td_params *td_params,
teardown:
/* Only free pages not yet added, so start at 'i' */
for (; i < kvm_tdx->td.tdcs_nr_pages; i++) {
- if (tdcs_pages[i]) {
- __free_page(tdcs_pages[i]);
- tdcs_pages[i] = NULL;
- }
+ tdx_free_control_page(tdcs_pages[i]);
+ tdcs_pages[i] = NULL;
}
if (!kvm_tdx->td.tdcs_pages)
kfree(tdcs_pages);
@@ -2605,16 +2647,13 @@ free_packages:
free_cpumask_var(packages);
free_tdcs:
- for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++) {
- if (tdcs_pages[i])
- __free_page(tdcs_pages[i]);
- }
+ for (i = 0; i < kvm_tdx->td.tdcs_nr_pages; i++)
+ tdx_free_control_page(tdcs_pages[i]);
kfree(tdcs_pages);
kvm_tdx->td.tdcs_pages = NULL;
free_tdr:
- if (tdr_page)
- __free_page(tdr_page);
+ tdx_free_control_page(tdr_page);
kvm_tdx->td.tdr_page = NULL;
free_hkid:
@@ -2943,7 +2982,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx)
int ret, i;
u64 err;
- page = alloc_page(GFP_KERNEL_ACCOUNT);
+ page = tdx_alloc_control_page();
if (!page)
return -ENOMEM;
tdx->vp.tdvpr_page = page;
@@ -2963,7 +3002,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx)
}
for (i = 0; i < kvm_tdx->td.tdcx_nr_pages; i++) {
- page = alloc_page(GFP_KERNEL_ACCOUNT);
+ page = tdx_alloc_control_page();
if (!page) {
ret = -ENOMEM;
goto free_tdcx;
@@ -2985,7 +3024,7 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx)
* method, but the rest are freed here.
*/
for (; i < kvm_tdx->td.tdcx_nr_pages; i++) {
- __free_page(tdx->vp.tdcx_pages[i]);
+ tdx_free_control_page(tdx->vp.tdcx_pages[i]);
tdx->vp.tdcx_pages[i] = NULL;
}
return -EIO;
@@ -3013,16 +3052,14 @@ static int tdx_td_vcpu_init(struct kvm_vcpu *vcpu, u64 vcpu_rcx)
free_tdcx:
for (i = 0; i < kvm_tdx->td.tdcx_nr_pages; i++) {
- if (tdx->vp.tdcx_pages[i])
- __free_page(tdx->vp.tdcx_pages[i]);
+ tdx_free_control_page(tdx->vp.tdcx_pages[i]);
tdx->vp.tdcx_pages[i] = NULL;
}
kfree(tdx->vp.tdcx_pages);
tdx->vp.tdcx_pages = NULL;
free_tdvpr:
- if (tdx->vp.tdvpr_page)
- __free_page(tdx->vp.tdvpr_page);
+ tdx_free_control_page(tdx->vp.tdvpr_page);
tdx->vp.tdvpr_page = NULL;
tdx->vp.tdvpr_pa = 0;
@@ -3482,6 +3519,10 @@ int __init tdx_hardware_setup(void)
vt_x86_ops.set_external_spte = tdx_sept_set_private_spte;
vt_x86_ops.free_external_spt = tdx_sept_free_private_spt;
+
+ if (tdx_supports_dynamic_pamt(tdx_sysinfo))
+ vt_x86_ops.topup_external_cache = tdx_topup_external_pamt_cache;
+
vt_x86_ops.protected_apic_has_interrupt = tdx_protected_apic_has_interrupt;
return 0;
diff --git a/arch/x86/kvm/vmx/tdx.h b/arch/x86/kvm/vmx/tdx.h
index ac8323a68b16..fd368e3ee060 100644
--- a/arch/x86/kvm/vmx/tdx.h
+++ b/arch/x86/kvm/vmx/tdx.h
@@ -72,6 +72,8 @@ struct vcpu_tdx {
u64 map_gpa_next;
u64 map_gpa_end;
+
+ struct tdx_pamt_cache pamt_cache;
};
void tdh_vp_rd_failed(struct vcpu_tdx *tdx, char *uclass, u32 field, u64 err);
diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c
index 612ab07d4100..2d0443562a16 100644
--- a/arch/x86/kvm/vmx/vmx.c
+++ b/arch/x86/kvm/vmx/vmx.c
@@ -7363,12 +7363,14 @@ static void atomic_switch_perf_msrs(struct vcpu_vmx *vmx)
if (kvm_vcpu_has_mediated_pmu(&vmx->vcpu))
return;
- pmu->host_cross_mapped_mask = 0;
- if (pmu->pebs_enable & pmu->global_ctrl)
- intel_pmu_cross_mapped_check(pmu);
+ struct x86_guest_pebs guest_pebs = {
+ .enable = intel_pmu_compute_pebs_enable(pmu),
+ .ds_area = pmu->ds_area,
+ .data_cfg = pmu->pebs_data_cfg,
+ };
/* Note, nr_msrs may be garbage if perf_guest_get_msrs() returns NULL. */
- msrs = perf_guest_get_msrs(&nr_msrs, (void *)pmu);
+ msrs = perf_guest_get_msrs(&nr_msrs, &guest_pebs);
if (!msrs)
return;
diff --git a/arch/x86/kvm/vmx/vmx.h b/arch/x86/kvm/vmx/vmx.h
index dc8517f15bc4..1db461060c7e 100644
--- a/arch/x86/kvm/vmx/vmx.h
+++ b/arch/x86/kvm/vmx/vmx.h
@@ -664,7 +664,20 @@ static __always_inline struct vcpu_vmx *to_vmx(struct kvm_vcpu *vcpu)
return container_of(vcpu, struct vcpu_vmx, vcpu);
}
-void intel_pmu_cross_mapped_check(struct kvm_pmu *pmu);
+u64 __intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu);
+
+static inline u64 intel_pmu_compute_pebs_enable(struct kvm_pmu *pmu)
+{
+ /*
+ * Avoid the function call overhead in the common case that the guest
+ * isn't using PEBS.
+ */
+ if (!(pmu->pebs_enable & pmu->global_ctrl))
+ return 0;
+
+ return __intel_pmu_compute_pebs_enable(pmu);
+}
+
int intel_pmu_create_guest_lbr_event(struct kvm_vcpu *vcpu);
void vmx_passthrough_lbr_msrs(struct kvm_vcpu *vcpu);
diff --git a/arch/x86/platform/Makefile b/arch/x86/platform/Makefile
index 3ed03a2552d0..727b92d0ca25 100644
--- a/arch/x86/platform/Makefile
+++ b/arch/x86/platform/Makefile
@@ -10,5 +10,4 @@ obj-y += intel-mid/
obj-y += intel-quark/
obj-y += olpc/
obj-y += scx200/
-obj-y += ts5500/
obj-y += uv/
diff --git a/arch/x86/platform/olpc/olpc-xo15-sci.c b/arch/x86/platform/olpc/olpc-xo15-sci.c
index 75ed6ceb9df3..a13dd6b8bc6e 100644
--- a/arch/x86/platform/olpc/olpc-xo15-sci.c
+++ b/arch/x86/platform/olpc/olpc-xo15-sci.c
@@ -210,8 +210,8 @@ static int xo15_sci_resume(struct device *dev)
static SIMPLE_DEV_PM_OPS(xo15_sci_pm, NULL, xo15_sci_resume);
static const struct acpi_device_id xo15_sci_device_ids[] = {
- {"XO15EC", 0},
- {"", 0},
+ { .id = "XO15EC" },
+ { }
};
static struct platform_driver xo15_sci_drv = {
diff --git a/arch/x86/platform/pvh/enlighten.c b/arch/x86/platform/pvh/enlighten.c
index f2053cbe9b0c..cb442cbd9d82 100644
--- a/arch/x86/platform/pvh/enlighten.c
+++ b/arch/x86/platform/pvh/enlighten.c
@@ -8,6 +8,7 @@
#include <asm/hypervisor.h>
#include <asm/e820/api.h>
#include <asm/x86_init.h>
+#include <asm/string.h>
#include <asm/xen/interface.h>
@@ -129,7 +130,7 @@ void __init xen_prepare_pvh(void)
* This must not compile to "call memset" because memset() may be
* instrumented.
*/
- __builtin_memset(&pvh_bootparams, 0, sizeof(pvh_bootparams));
+ __inline_memset(&pvh_bootparams, 0, sizeof(pvh_bootparams));
hypervisor_specific_init(xen_guest);
diff --git a/arch/x86/platform/ts5500/Makefile b/arch/x86/platform/ts5500/Makefile
deleted file mode 100644
index 910fe9e3ffb4..000000000000
--- a/arch/x86/platform/ts5500/Makefile
+++ /dev/null
@@ -1,2 +0,0 @@
-# SPDX-License-Identifier: GPL-2.0-only
-obj-$(CONFIG_TS5500) += ts5500.o
diff --git a/arch/x86/platform/ts5500/ts5500.c b/arch/x86/platform/ts5500/ts5500.c
deleted file mode 100644
index 0b67da056fd9..000000000000
--- a/arch/x86/platform/ts5500/ts5500.c
+++ /dev/null
@@ -1,341 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0-or-later
-/*
- * Technologic Systems TS-5500 Single Board Computer support
- *
- * Copyright (C) 2013-2014 Savoir-faire Linux Inc.
- * Vivien Didelot <vivien.didelot@savoirfairelinux.com>
- *
- * This driver registers the Technologic Systems TS-5500 Single Board Computer
- * (SBC) and its devices, and exposes information to userspace such as jumpers'
- * state or available options. For further information about sysfs entries, see
- * Documentation/ABI/testing/sysfs-platform-ts5500.
- *
- * This code may be extended to support similar x86-based platforms.
- * Actually, the TS-5500 and TS-5400 are supported.
- */
-
-#include <linux/delay.h>
-#include <linux/io.h>
-#include <linux/kernel.h>
-#include <linux/leds.h>
-#include <linux/init.h>
-#include <linux/platform_data/max197.h>
-#include <linux/platform_device.h>
-#include <linux/slab.h>
-
-/* Product code register */
-#define TS5500_PRODUCT_CODE_ADDR 0x74
-#define TS5500_PRODUCT_CODE 0x60 /* TS-5500 product code */
-#define TS5400_PRODUCT_CODE 0x40 /* TS-5400 product code */
-
-/* SRAM/RS-485/ADC options, and RS-485 RTS/Automatic RS-485 flags register */
-#define TS5500_SRAM_RS485_ADC_ADDR 0x75
-#define TS5500_SRAM BIT(0) /* SRAM option */
-#define TS5500_RS485 BIT(1) /* RS-485 option */
-#define TS5500_ADC BIT(2) /* A/D converter option */
-#define TS5500_RS485_RTS BIT(6) /* RTS for RS-485 */
-#define TS5500_RS485_AUTO BIT(7) /* Automatic RS-485 */
-
-/* External Reset/Industrial Temperature Range options register */
-#define TS5500_ERESET_ITR_ADDR 0x76
-#define TS5500_ERESET BIT(0) /* External Reset option */
-#define TS5500_ITR BIT(1) /* Indust. Temp. Range option */
-
-/* LED/Jumpers register */
-#define TS5500_LED_JP_ADDR 0x77
-#define TS5500_LED BIT(0) /* LED flag */
-#define TS5500_JP1 BIT(1) /* Automatic CMOS */
-#define TS5500_JP2 BIT(2) /* Enable Serial Console */
-#define TS5500_JP3 BIT(3) /* Write Enable Drive A */
-#define TS5500_JP4 BIT(4) /* Fast Console (115K baud) */
-#define TS5500_JP5 BIT(5) /* User Jumper */
-#define TS5500_JP6 BIT(6) /* Console on COM1 (req. JP2) */
-#define TS5500_JP7 BIT(7) /* Undocumented (Unused) */
-
-/* A/D Converter registers */
-#define TS5500_ADC_CONV_BUSY_ADDR 0x195 /* Conversion state register */
-#define TS5500_ADC_CONV_BUSY BIT(0)
-#define TS5500_ADC_CONV_INIT_LSB_ADDR 0x196 /* Start conv. / LSB register */
-#define TS5500_ADC_CONV_MSB_ADDR 0x197 /* MSB register */
-#define TS5500_ADC_CONV_DELAY 12 /* usec */
-
-/**
- * struct ts5500_sbc - TS-5500 board description
- * @name: Board model name.
- * @id: Board product ID.
- * @sram: Flag for SRAM option.
- * @rs485: Flag for RS-485 option.
- * @adc: Flag for Analog/Digital converter option.
- * @ereset: Flag for External Reset option.
- * @itr: Flag for Industrial Temperature Range option.
- * @jumpers: Bitfield for jumpers' state.
- */
-struct ts5500_sbc {
- const char *name;
- int id;
- bool sram;
- bool rs485;
- bool adc;
- bool ereset;
- bool itr;
- u8 jumpers;
-};
-
-/* Board signatures in BIOS shadow RAM */
-static const struct {
- const char * const string;
- const ssize_t offset;
-} ts5500_signatures[] __initconst = {
- { "TS-5x00 AMD Elan", 0xb14 },
-};
-
-static int __init ts5500_check_signature(void)
-{
- void __iomem *bios;
- int i, ret = -ENODEV;
-
- bios = ioremap(0xf0000, 0x10000);
- if (!bios)
- return -ENOMEM;
-
- for (i = 0; i < ARRAY_SIZE(ts5500_signatures); i++) {
- if (check_signature(bios + ts5500_signatures[i].offset,
- ts5500_signatures[i].string,
- strlen(ts5500_signatures[i].string))) {
- ret = 0;
- break;
- }
- }
-
- iounmap(bios);
- return ret;
-}
-
-static int __init ts5500_detect_config(struct ts5500_sbc *sbc)
-{
- u8 tmp;
- int ret = 0;
-
- if (!request_region(TS5500_PRODUCT_CODE_ADDR, 4, "ts5500"))
- return -EBUSY;
-
- sbc->id = inb(TS5500_PRODUCT_CODE_ADDR);
- if (sbc->id == TS5500_PRODUCT_CODE) {
- sbc->name = "TS-5500";
- } else if (sbc->id == TS5400_PRODUCT_CODE) {
- sbc->name = "TS-5400";
- } else {
- pr_err("ts5500: unknown product code 0x%x\n", sbc->id);
- ret = -ENODEV;
- goto cleanup;
- }
-
- tmp = inb(TS5500_SRAM_RS485_ADC_ADDR);
- sbc->sram = tmp & TS5500_SRAM;
- sbc->rs485 = tmp & TS5500_RS485;
- sbc->adc = tmp & TS5500_ADC;
-
- tmp = inb(TS5500_ERESET_ITR_ADDR);
- sbc->ereset = tmp & TS5500_ERESET;
- sbc->itr = tmp & TS5500_ITR;
-
- tmp = inb(TS5500_LED_JP_ADDR);
- sbc->jumpers = tmp & ~TS5500_LED;
-
-cleanup:
- release_region(TS5500_PRODUCT_CODE_ADDR, 4);
- return ret;
-}
-
-static ssize_t name_show(struct device *dev, struct device_attribute *attr,
- char *buf)
-{
- struct ts5500_sbc *sbc = dev_get_drvdata(dev);
-
- return sprintf(buf, "%s\n", sbc->name);
-}
-static DEVICE_ATTR_RO(name);
-
-static ssize_t id_show(struct device *dev, struct device_attribute *attr,
- char *buf)
-{
- struct ts5500_sbc *sbc = dev_get_drvdata(dev);
-
- return sprintf(buf, "0x%.2x\n", sbc->id);
-}
-static DEVICE_ATTR_RO(id);
-
-static ssize_t jumpers_show(struct device *dev, struct device_attribute *attr,
- char *buf)
-{
- struct ts5500_sbc *sbc = dev_get_drvdata(dev);
-
- return sprintf(buf, "0x%.2x\n", sbc->jumpers >> 1);
-}
-static DEVICE_ATTR_RO(jumpers);
-
-#define TS5500_ATTR_BOOL(_field) \
- static ssize_t _field##_show(struct device *dev, \
- struct device_attribute *attr, char *buf) \
- { \
- struct ts5500_sbc *sbc = dev_get_drvdata(dev); \
- \
- return sprintf(buf, "%d\n", sbc->_field); \
- } \
- static DEVICE_ATTR_RO(_field)
-
-TS5500_ATTR_BOOL(sram);
-TS5500_ATTR_BOOL(rs485);
-TS5500_ATTR_BOOL(adc);
-TS5500_ATTR_BOOL(ereset);
-TS5500_ATTR_BOOL(itr);
-
-static struct attribute *ts5500_attributes[] = {
- &dev_attr_id.attr,
- &dev_attr_name.attr,
- &dev_attr_jumpers.attr,
- &dev_attr_sram.attr,
- &dev_attr_rs485.attr,
- &dev_attr_adc.attr,
- &dev_attr_ereset.attr,
- &dev_attr_itr.attr,
- NULL
-};
-
-static const struct attribute_group ts5500_attr_group = {
- .attrs = ts5500_attributes,
-};
-
-static struct resource ts5500_dio1_resource[] = {
- DEFINE_RES_IRQ_NAMED(7, "DIO1 interrupt"),
-};
-
-static struct platform_device ts5500_dio1_pdev = {
- .name = "ts5500-dio1",
- .id = -1,
- .resource = ts5500_dio1_resource,
- .num_resources = 1,
-};
-
-static struct resource ts5500_dio2_resource[] = {
- DEFINE_RES_IRQ_NAMED(6, "DIO2 interrupt"),
-};
-
-static struct platform_device ts5500_dio2_pdev = {
- .name = "ts5500-dio2",
- .id = -1,
- .resource = ts5500_dio2_resource,
- .num_resources = 1,
-};
-
-static void ts5500_led_set(struct led_classdev *led_cdev,
- enum led_brightness brightness)
-{
- outb(!!brightness, TS5500_LED_JP_ADDR);
-}
-
-static enum led_brightness ts5500_led_get(struct led_classdev *led_cdev)
-{
- return (inb(TS5500_LED_JP_ADDR) & TS5500_LED) ? LED_FULL : LED_OFF;
-}
-
-static struct led_classdev ts5500_led_cdev = {
- .name = "ts5500:green:",
- .brightness_set = ts5500_led_set,
- .brightness_get = ts5500_led_get,
-};
-
-static int ts5500_adc_convert(u8 ctrl)
-{
- u8 lsb, msb;
-
- /* Start conversion (ensure the 3 MSB are set to 0) */
- outb(ctrl & 0x1f, TS5500_ADC_CONV_INIT_LSB_ADDR);
-
- /*
- * The platform has CPLD logic driving the A/D converter.
- * The conversion must complete within 11 microseconds,
- * otherwise we have to re-initiate a conversion.
- */
- udelay(TS5500_ADC_CONV_DELAY);
- if (inb(TS5500_ADC_CONV_BUSY_ADDR) & TS5500_ADC_CONV_BUSY)
- return -EBUSY;
-
- /* Read the raw data */
- lsb = inb(TS5500_ADC_CONV_INIT_LSB_ADDR);
- msb = inb(TS5500_ADC_CONV_MSB_ADDR);
-
- return (msb << 8) | lsb;
-}
-
-static struct max197_platform_data ts5500_adc_pdata = {
- .convert = ts5500_adc_convert,
-};
-
-static struct platform_device ts5500_adc_pdev = {
- .name = "max197",
- .id = -1,
- .dev = {
- .platform_data = &ts5500_adc_pdata,
- },
-};
-
-static int __init ts5500_init(void)
-{
- struct platform_device *pdev;
- struct ts5500_sbc *sbc;
- int err;
-
- /*
- * There is no DMI available or PCI bridge subvendor info,
- * only the BIOS provides a 16-bit identification call.
- * It is safer to find a signature in the BIOS shadow RAM.
- */
- err = ts5500_check_signature();
- if (err)
- return err;
-
- pdev = platform_device_register_simple("ts5500", -1, NULL, 0);
- if (IS_ERR(pdev))
- return PTR_ERR(pdev);
-
- sbc = devm_kzalloc(&pdev->dev, sizeof(struct ts5500_sbc), GFP_KERNEL);
- if (!sbc) {
- err = -ENOMEM;
- goto error;
- }
-
- err = ts5500_detect_config(sbc);
- if (err)
- goto error;
-
- platform_set_drvdata(pdev, sbc);
-
- err = sysfs_create_group(&pdev->dev.kobj, &ts5500_attr_group);
- if (err)
- goto error;
-
- if (sbc->id == TS5500_PRODUCT_CODE) {
- ts5500_dio1_pdev.dev.parent = &pdev->dev;
- if (platform_device_register(&ts5500_dio1_pdev))
- dev_warn(&pdev->dev, "DIO1 block registration failed\n");
- ts5500_dio2_pdev.dev.parent = &pdev->dev;
- if (platform_device_register(&ts5500_dio2_pdev))
- dev_warn(&pdev->dev, "DIO2 block registration failed\n");
- }
-
- if (led_classdev_register(&pdev->dev, &ts5500_led_cdev))
- dev_warn(&pdev->dev, "LED registration failed\n");
-
- if (sbc->adc) {
- ts5500_adc_pdev.dev.parent = &pdev->dev;
- if (platform_device_register(&ts5500_adc_pdev))
- dev_warn(&pdev->dev, "ADC registration failed\n");
- }
-
- return 0;
-error:
- platform_device_unregister(pdev);
- return err;
-}
-device_initcall(ts5500_init);
diff --git a/arch/x86/virt/svm/sev.c b/arch/x86/virt/svm/sev.c
index cff285d8ad8e..bd70c4d0b774 100644
--- a/arch/x86/virt/svm/sev.c
+++ b/arch/x86/virt/svm/sev.c
@@ -19,6 +19,7 @@
#include <linux/iommu.h>
#include <linux/amd-iommu.h>
#include <linux/nospec.h>
+#include <linux/workqueue.h>
#include <asm/sev.h>
#include <asm/processor.h>
@@ -124,6 +125,28 @@ static void *rmp_bookkeeping __ro_after_init;
static u64 probed_rmp_base, probed_rmp_size;
+static u64 rmpopt_pa_start, rmpopt_pa_end;
+
+enum rmpopt_op_type {
+ RMPOPT_OP_VERIFY_AND_REPORT_STATUS,
+ RMPOPT_OP_REPORT_STATUS
+};
+
+static struct workqueue_struct *rmpopt_wq;
+static struct delayed_work rmpopt_delayed_work;
+
+/* Software RMPOPT facilities initialized */
+static bool rmpopt_soft_init;
+
+/*
+ * Delay, in milliseconds, before the RMP re-optimization pass runs after an
+ * SNP guest is torn down. This coalesces a burst of teardowns into a single
+ * scan and gives each guest's pages time to be converted back to the shared,
+ * hypervisor-owned state. The 10 second value is a heuristic trading
+ * re-optimization latency against scanning too eagerly.
+ */
+#define RMPOPT_WORK_TIMEOUT (10 * MSEC_PER_SEC)
+
static LIST_HEAD(snp_leaked_pages_list);
static DEFINE_SPINLOCK(snp_leaked_pages_list_lock);
@@ -513,7 +536,6 @@ static void clear_hsave_pa(void *arg)
int snp_prepare(void)
{
- int ret;
u64 val;
/*
@@ -526,14 +548,18 @@ int snp_prepare(void)
clear_rmp();
- cpus_read_lock();
+ /*
+ * No CPU may come online without SnpEn while SNP is active; disable
+ * hotplug here and re-enable it in snp_shutdown().
+ */
+ cpu_hotplug_disable();
if (!cpumask_equal(cpu_online_mask, cpu_present_mask)) {
- ret = -EOPNOTSUPP;
+ cpu_hotplug_enable();
pr_warn("SNP init failed: not all CPUs online. (%*pbl online <-> %*pbl present masks).\n",
cpumask_pr_args(cpu_online_mask),
cpumask_pr_args(cpu_present_mask));
- goto unlock;
+ return -EOPNOTSUPP;
}
wbinvd_on_all_cpus();
@@ -548,12 +574,7 @@ int snp_prepare(void)
/* SNP_INIT requires MSR_VM_HSAVE_PA to be cleared on all CPUs. */
on_each_cpu(clear_hsave_pa, NULL, 1);
- ret = 0;
-
-unlock:
- cpus_read_unlock();
-
- return ret;
+ return 0;
}
EXPORT_SYMBOL_FOR_MODULES(snp_prepare, "ccp");
@@ -565,18 +586,122 @@ void snp_shutdown(void)
if (syscfg & MSR_AMD64_SYSCFG_SNP_EN)
return;
+ if (rmpopt_soft_init)
+ cancel_delayed_work_sync(&rmpopt_delayed_work);
+
clear_rmp();
on_each_cpu(mfd_reconfigure, NULL, 1);
+
+ /*
+ * The firmware has disabled SNP (SnpEn is clear), so re-enable CPU
+ * hotplug. A legacy SNP shutdown returns above with SnpEn still set and
+ * leaves hotplug disabled.
+ */
+ cpu_hotplug_enable();
}
EXPORT_SYMBOL_FOR_MODULES(snp_shutdown, "ccp");
/*
+ * RMPOPT optimizations skip RMP checks at 1GB granularity if this range of
+ * memory does not contain any SNP guest memory.
+ *
+ * @pa is a system physical address; RMPOPT operates on the containing 1GB.
+ */
+static void rmpopt(u64 pa)
+{
+ enum rmpopt_op_type op = RMPOPT_OP_VERIFY_AND_REPORT_STATUS;
+ u64 pa_start = ALIGN_DOWN(pa, SZ_1G);
+
+ /* Supported by binutils 2.48+ */
+ asm volatile(".byte 0xf2, 0x0f, 0x01, 0xfc"
+ :: "a" (pa_start), "c" (op)
+ : "memory", "cc");
+}
+
+static void rmpopt_scan_range(void *arg)
+{
+ u64 pa;
+
+ for (pa = rmpopt_pa_start; pa < rmpopt_pa_end; pa += SZ_1G)
+ rmpopt(pa);
+}
+
+static void do_rmpopt_work(struct work_struct *work)
+{
+ /*
+ * Warm up the RMPOPT cache on this pinned per-CPU worker with interrupts
+ * enabled, so the IRQ-disabled fan-out below only issues cache-hit RMPOPTs.
+ */
+ rmpopt_scan_range(NULL);
+
+ on_each_cpu_mask(cpu_primary_thread_mask, rmpopt_scan_range, NULL, true);
+}
+
+static int __init rmpopt_init(void)
+{
+ if (!cpu_feature_enabled(X86_FEATURE_RMPOPT))
+ return -ENODEV;
+
+ rmpopt_wq = alloc_workqueue("rmpopt_wq", WQ_PERCPU, 1);
+ if (!rmpopt_wq) {
+ pr_err("Failed to allocate RMPOPT workqueue\n");
+ return -ENOMEM;
+ }
+
+ INIT_DELAYED_WORK(&rmpopt_delayed_work, do_rmpopt_work);
+
+ /* The optimization range is fixed at boot; compute it once. */
+ rmpopt_pa_start = ALIGN_DOWN(PFN_PHYS(min_low_pfn), SZ_1G);
+ rmpopt_pa_end = ALIGN(PFN_PHYS(max_pfn), SZ_1G);
+ if ((rmpopt_pa_end - rmpopt_pa_start) > SZ_2T)
+ rmpopt_pa_end = rmpopt_pa_start + SZ_2T;
+
+ pr_info("RMPOPT optimizations enabled\n");
+
+ rmpopt_soft_init = true;
+
+ return 0;
+}
+device_initcall(rmpopt_init);
+
+void snp_enable_rmpopt(void)
+{
+ u64 base;
+ int cpu;
+
+ if (!cpu_feature_enabled(X86_FEATURE_RMPOPT))
+ return;
+
+ if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP))
+ return;
+
+ if (!rmpopt_soft_init)
+ return;
+
+ /*
+ * Per-CPU RMPOPT tables cover at most 2 TB. Program each core's
+ * RMPOPT_BASE with the start of RAM to optimize up to 2 TB.
+ */
+ rdmsrq(MSR_AMD64_RMPOPT_BASE, base);
+ if (!(base & MSR_AMD64_RMPOPT_ENABLE))
+ for_each_cpu(cpu, cpu_primary_thread_mask)
+ wrmsrq_on_cpu(cpu, MSR_AMD64_RMPOPT_BASE,
+ rmpopt_pa_start | MSR_AMD64_RMPOPT_ENABLE);
+
+ mod_delayed_work(rmpopt_wq, &rmpopt_delayed_work,
+ msecs_to_jiffies(RMPOPT_WORK_TIMEOUT));
+}
+EXPORT_SYMBOL_FOR_MODULES(snp_enable_rmpopt, "ccp,kvm-amd");
+
+/*
* Do the necessary preparations which are verified by the firmware as
* described in the SNP_INIT_EX firmware command description in the SNP
* firmware ABI spec.
*/
int __init snp_rmptable_init(void)
{
+ u64 val;
+
if (WARN_ON_ONCE(!cc_platform_has(CC_ATTR_HOST_SEV_SNP)))
return -ENOSYS;
@@ -587,6 +712,15 @@ int __init snp_rmptable_init(void)
return -ENOSYS;
/*
+ * On a kexec boot SNP may already be enabled (legacy firmware leaves
+ * SnpEn set across shutdown), in which case snp_prepare() bails without
+ * disabling CPU hotplug, so disable it here.
+ */
+ rdmsrq(MSR_AMD64_SYSCFG, val);
+ if (val & MSR_AMD64_SYSCFG_SNP_EN)
+ cpu_hotplug_disable();
+
+ /*
* Setting crash_kexec_post_notifiers to 'true' to ensure that SNP panic
* notifier is invoked to do SNP IOMMU shutdown before kdump.
*/
@@ -683,13 +817,21 @@ static bool probe_segmented_rmptable_info(void)
bool snp_probe_rmptable_info(void)
{
- if (cpu_feature_enabled(X86_FEATURE_SEGMENTED_RMP))
+ if (cpu_feature_enabled(X86_FEATURE_SEGMENTED_RMP)) {
rdmsrq(MSR_AMD64_RMP_CFG, rmp_cfg);
- if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED)
- return probe_segmented_rmptable_info();
- else
- return probe_contiguous_rmptable_info();
+ if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) {
+ if (probe_segmented_rmptable_info())
+ return true;
+
+ setup_clear_cpu_cap(X86_FEATURE_RMPOPT);
+ return false;
+ }
+ } else {
+ setup_clear_cpu_cap(X86_FEATURE_RMPOPT);
+ }
+
+ return probe_contiguous_rmptable_info();
}
/*
diff --git a/arch/x86/virt/vmx/tdx/seamcall_internal.h b/arch/x86/virt/vmx/tdx/seamcall_internal.h
index be5f446467df..051ad2d45cab 100644
--- a/arch/x86/virt/vmx/tdx/seamcall_internal.h
+++ b/arch/x86/virt/vmx/tdx/seamcall_internal.h
@@ -11,6 +11,7 @@
#ifndef _X86_VIRT_SEAMCALL_INTERNAL_H
#define _X86_VIRT_SEAMCALL_INTERNAL_H
+#include <linux/bitfield.h>
#include <linux/printk.h>
#include <linux/types.h>
#include <asm/archrandom.h>
@@ -23,9 +24,27 @@ u64 __seamcall_saved_ret(u64 fn, struct tdx_module_args *args);
typedef u64 (*sc_func_t)(u64 fn, struct tdx_module_args *args);
+/*
+ * SEAMCALL leaf:
+ *
+ * Bit 15:0 Leaf number
+ * Bit 23:16 Leaf ABI version number
+ * Bit 24 Pending interrupts detection mode
+ * Bit 63 1 for P-SEAMLDR leaf, 0 for TDX module leaf
+ */
+#define SEAMCALL_LEAF_MASK GENMASK_U64(15, 0)
+#define SEAMCALL_SEAMLDR_MASK BIT_U64(63)
+
static __always_inline u64 __seamcall_dirty_cache(sc_func_t func, u64 fn,
struct tdx_module_args *args)
{
+ /*
+ * fn contains leaf number for TDX module calls and P-SEAMLDR calls.
+ * Other fields in SEAMCALL leaf like leaf ABI version number are in
+ * struct tdx_module_args.
+ */
+ BUILD_BUG_ON(fn & ~(SEAMCALL_LEAF_MASK | SEAMCALL_SEAMLDR_MASK));
+
lockdep_assert_preemption_disabled();
/*
diff --git a/arch/x86/virt/vmx/tdx/tdx.c b/arch/x86/virt/vmx/tdx/tdx.c
index 1b9ff749dd8e..96ced0494b68 100644
--- a/arch/x86/virt/vmx/tdx/tdx.c
+++ b/arch/x86/virt/vmx/tdx/tdx.c
@@ -30,6 +30,7 @@
#include <linux/suspend.h>
#include <linux/syscore_ops.h>
#include <linux/idr.h>
+#include <linux/vmalloc.h>
#include <asm/page.h>
#include <asm/special_insns.h>
#include <asm/msr-index.h>
@@ -46,6 +47,9 @@
#include "seamcall_internal.h"
#include "tdx.h"
+/* Number of DPAMT pages to be provided to TDX module per 2MB region of PA */
+#define TDX_DPAMT_ENTRY_PAGE_CNT 2
+
struct tdx_module_state {
bool initialized;
bool sysinit_done;
@@ -63,6 +67,14 @@ static DEFINE_PER_CPU(bool, tdx_lp_initialized);
static struct tdmr_info_list tdx_tdmr_list;
+/*
+ * On a machine with DPAMT, the kernel maintains a reference counter
+ * for every 2MB range. The counter indicates how many users there are for
+ * the DPAMT at the 2MB range. The kernel allocates DPAMT refcounts at
+ * initialization.
+ */
+static atomic_t *dpamt_refcounts;
+
/* All TDX-usable memory regions. Protected by mem_hotplug_lock. */
static LIST_HEAD(tdx_memlist);
@@ -253,6 +265,42 @@ static struct syscore tdx_syscore = {
};
/*
+ * Allocate DPAMT reference counters for all physical memory.
+ *
+ * It consumes 2MB for every 1TB of physical memory.
+ */
+static __init int init_dpamt_refcounts(void)
+{
+ size_t size = DIV_ROUND_UP(max_pfn, PTRS_PER_PTE) * sizeof(*dpamt_refcounts);
+
+ if (!tdx_supports_dynamic_pamt(&tdx_sysinfo))
+ return 0;
+
+ dpamt_refcounts = vzalloc(size);
+ if (!dpamt_refcounts)
+ return -ENOMEM;
+
+ return 0;
+}
+
+static __init void free_dpamt_refcounts(void)
+{
+ if (!tdx_supports_dynamic_pamt(&tdx_sysinfo))
+ return;
+
+ vfree(dpamt_refcounts);
+ dpamt_refcounts = NULL;
+}
+
+static atomic_t *tdx_find_dpamt_refcount(unsigned long pfn)
+{
+ /* Find which PMD a PFN is in. */
+ unsigned long index = pfn >> (PMD_SHIFT - PAGE_SHIFT);
+
+ return &dpamt_refcounts[index];
+}
+
+/*
* Add a memory region as a TDX memory block. The caller must make sure
* all memory regions are added in address ascending order and don't
* overlap.
@@ -510,35 +558,37 @@ static __init int fill_out_tdmrs(struct list_head *tmb_list,
return 0;
}
+static __init unsigned long tdmr_get_pamt_bitmap_sz(struct tdmr_info *tdmr)
+{
+ unsigned long pamt_sz, nr_pamt_entries;
+ int bits_per_entry;
+
+ bits_per_entry = tdx_sysinfo.tdmr.pamt_page_bitmap_entry_bits;
+ nr_pamt_entries = tdmr->size >> PAGE_SHIFT;
+ pamt_sz = DIV_ROUND_UP(nr_pamt_entries * bits_per_entry, BITS_PER_BYTE);
+
+ return PAGE_ALIGN(pamt_sz);
+}
+
/*
* Calculate PAMT size given a TDMR and a page size. The returned
* PAMT size is always aligned up to 4K page boundary.
*/
-static __init unsigned long tdmr_get_pamt_sz(struct tdmr_info *tdmr, int pgsz,
- u16 pamt_entry_size)
+static __init unsigned long tdmr_get_pamt_sz(struct tdmr_info *tdmr, int pgsz)
{
unsigned long pamt_sz, nr_pamt_entries;
+ const int tdx_pg_size_shift[TDX_PS_NR] = { PAGE_SHIFT, PMD_SHIFT, PUD_SHIFT };
+ const u16 pamt_entry_size[TDX_PS_NR] = {
+ tdx_sysinfo.tdmr.pamt_4k_entry_size,
+ tdx_sysinfo.tdmr.pamt_2m_entry_size,
+ tdx_sysinfo.tdmr.pamt_1g_entry_size,
+ };
- switch (pgsz) {
- case TDX_PS_4K:
- nr_pamt_entries = tdmr->size >> PAGE_SHIFT;
- break;
- case TDX_PS_2M:
- nr_pamt_entries = tdmr->size >> PMD_SHIFT;
- break;
- case TDX_PS_1G:
- nr_pamt_entries = tdmr->size >> PUD_SHIFT;
- break;
- default:
- WARN_ON_ONCE(1);
- return 0;
- }
+ nr_pamt_entries = tdmr->size >> tdx_pg_size_shift[pgsz];
+ pamt_sz = nr_pamt_entries * pamt_entry_size[pgsz];
- pamt_sz = nr_pamt_entries * pamt_entry_size;
/* TDX requires PAMT size must be 4K aligned */
- pamt_sz = ALIGN(pamt_sz, PAGE_SIZE);
-
- return pamt_sz;
+ return PAGE_ALIGN(pamt_sz);
}
/*
@@ -576,15 +626,11 @@ static __init int tdmr_get_nid(struct tdmr_info *tdmr, struct list_head *tmb_lis
* within @tdmr, and set up PAMTs for @tdmr.
*/
static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr,
- struct list_head *tmb_list,
- u16 pamt_entry_size[])
+ struct list_head *tmb_list)
{
- unsigned long pamt_base[TDX_PS_NR];
- unsigned long pamt_size[TDX_PS_NR];
- unsigned long tdmr_pamt_base;
unsigned long tdmr_pamt_size;
struct page *pamt;
- int pgsz, nid;
+ int nid;
nid = tdmr_get_nid(tdmr, tmb_list);
@@ -592,13 +638,18 @@ static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr,
* Calculate the PAMT size for each TDX supported page size
* and the total PAMT size.
*/
- tdmr_pamt_size = 0;
- for (pgsz = TDX_PS_4K; pgsz < TDX_PS_NR; pgsz++) {
- pamt_size[pgsz] = tdmr_get_pamt_sz(tdmr, pgsz,
- pamt_entry_size[pgsz]);
- tdmr_pamt_size += pamt_size[pgsz];
+ tdmr->pamt_1g_size = tdmr_get_pamt_sz(tdmr, TDX_PS_1G);
+ tdmr->pamt_2m_size = tdmr_get_pamt_sz(tdmr, TDX_PS_2M);
+
+ if (tdx_supports_dynamic_pamt(&tdx_sysinfo)) {
+ /* With DPAMT, PAMT_4K is replaced with a bitmap */
+ tdmr->pamt_4k_size = tdmr_get_pamt_bitmap_sz(tdmr);
+ } else {
+ tdmr->pamt_4k_size = tdmr_get_pamt_sz(tdmr, TDX_PS_4K);
}
+ tdmr_pamt_size = tdmr->pamt_4k_size + tdmr->pamt_2m_size + tdmr->pamt_1g_size;
+
/*
* Allocate one chunk of physically contiguous memory for all
* PAMTs. This helps minimize the PAMT's use of reserved areas
@@ -606,25 +657,17 @@ static __init int tdmr_set_up_pamt(struct tdmr_info *tdmr,
*/
pamt = alloc_contig_pages(tdmr_pamt_size >> PAGE_SHIFT, GFP_KERNEL,
nid, &node_online_map);
- if (!pamt)
- return -ENOMEM;
/*
- * Break the contiguous allocation back up into the
- * individual PAMTs for each page size.
+ * tdmr->pamt_4k_base is still zero so the error
+ * path of the caller will skip freeing the PAMT.
*/
- tdmr_pamt_base = page_to_pfn(pamt) << PAGE_SHIFT;
- for (pgsz = TDX_PS_4K; pgsz < TDX_PS_NR; pgsz++) {
- pamt_base[pgsz] = tdmr_pamt_base;
- tdmr_pamt_base += pamt_size[pgsz];
- }
+ if (!pamt)
+ return -ENOMEM;
- tdmr->pamt_4k_base = pamt_base[TDX_PS_4K];
- tdmr->pamt_4k_size = pamt_size[TDX_PS_4K];
- tdmr->pamt_2m_base = pamt_base[TDX_PS_2M];
- tdmr->pamt_2m_size = pamt_size[TDX_PS_2M];
- tdmr->pamt_1g_base = pamt_base[TDX_PS_1G];
- tdmr->pamt_1g_size = pamt_size[TDX_PS_1G];
+ tdmr->pamt_4k_base = page_to_phys(pamt);
+ tdmr->pamt_2m_base = tdmr->pamt_4k_base + tdmr->pamt_4k_size;
+ tdmr->pamt_1g_base = tdmr->pamt_2m_base + tdmr->pamt_2m_size;
return 0;
}
@@ -655,10 +698,7 @@ static __init void tdmr_do_pamt_func(struct tdmr_info *tdmr,
tdmr_get_pamt(tdmr, &pamt_base, &pamt_size);
/* Do nothing if PAMT hasn't been allocated for this TDMR */
- if (!pamt_size)
- return;
-
- if (WARN_ON_ONCE(!pamt_base))
+ if (!pamt_base)
return;
pamt_func(pamt_base, pamt_size);
@@ -684,14 +724,12 @@ static __init void tdmrs_free_pamt_all(struct tdmr_info_list *tdmr_list)
/* Allocate and set up PAMTs for all TDMRs */
static __init int tdmrs_set_up_pamt_all(struct tdmr_info_list *tdmr_list,
- struct list_head *tmb_list,
- u16 pamt_entry_size[])
+ struct list_head *tmb_list)
{
int i, ret = 0;
for (i = 0; i < tdmr_list->nr_consumed_tdmrs; i++) {
- ret = tdmr_set_up_pamt(tdmr_entry(tdmr_list, i), tmb_list,
- pamt_entry_size);
+ ret = tdmr_set_up_pamt(tdmr_entry(tdmr_list, i), tmb_list);
if (ret)
goto err;
}
@@ -968,18 +1006,13 @@ static __init int construct_tdmrs(struct list_head *tmb_list,
struct tdmr_info_list *tdmr_list,
struct tdx_sys_info_tdmr *sysinfo_tdmr)
{
- u16 pamt_entry_size[TDX_PS_NR] = {
- sysinfo_tdmr->pamt_4k_entry_size,
- sysinfo_tdmr->pamt_2m_entry_size,
- sysinfo_tdmr->pamt_1g_entry_size,
- };
int ret;
ret = fill_out_tdmrs(tmb_list, tdmr_list);
if (ret)
return ret;
- ret = tdmrs_set_up_pamt_all(tdmr_list, tmb_list, pamt_entry_size);
+ ret = tdmrs_set_up_pamt_all(tdmr_list, tmb_list);
if (ret)
return ret;
@@ -998,6 +1031,8 @@ static __init int construct_tdmrs(struct list_head *tmb_list,
return ret;
}
+#define TDX_SYS_CONFIG_DYNAMIC_PAMT BIT(16)
+
static __init int config_tdx_module(struct tdmr_info_list *tdmr_list,
u64 global_keyid)
{
@@ -1026,6 +1061,12 @@ static __init int config_tdx_module(struct tdmr_info_list *tdmr_list,
args.rcx = __pa(tdmr_pa_array);
args.rdx = tdmr_list->nr_consumed_tdmrs;
args.r8 = global_keyid;
+
+ if (tdx_supports_dynamic_pamt(&tdx_sysinfo)) {
+ pr_info("Enable Dynamic PAMT\n");
+ args.r8 |= TDX_SYS_CONFIG_DYNAMIC_PAMT;
+ }
+
ret = seamcall_prerr(TDH_SYS_CONFIG, &args);
/* Free the array as it is not required anymore. */
@@ -1167,10 +1208,14 @@ static __init int init_tdx_module(void)
*/
get_online_mems();
- ret = build_tdx_memlist(&tdx_memlist);
+ ret = init_dpamt_refcounts();
if (ret)
goto out_put_tdxmem;
+ ret = build_tdx_memlist(&tdx_memlist);
+ if (ret)
+ goto err_free_dpamt_refcounts;
+
/* Allocate enough space for constructing TDMRs */
ret = alloc_tdmr_list(&tdx_tdmr_list, &tdx_sysinfo.tdmr);
if (ret)
@@ -1220,6 +1265,8 @@ err_free_tdmrs:
free_tdmr_list(&tdx_tdmr_list);
err_free_tdxmem:
free_tdx_memlist(&tdx_memlist);
+err_free_dpamt_refcounts:
+ free_dpamt_refcounts();
goto out_put_tdxmem;
}
@@ -1912,10 +1959,11 @@ u64 tdh_vp_init(struct tdx_vp *vp, u64 initial_rcx, u32 x2apicid)
.rcx = vp->tdvpr_pa,
.rdx = initial_rcx,
.r8 = x2apicid,
+ /* apicid requires version == 1. */
+ .version = 1,
};
- /* apicid requires version == 1. */
- return seamcall(TDH_VP_INIT | (1ULL << TDX_VERSION_SHIFT), &args);
+ return seamcall(TDH_VP_INIT, &args);
}
EXPORT_SYMBOL_FOR_KVM(tdh_vp_init);
@@ -2005,6 +2053,269 @@ u64 tdh_phymem_page_wbinvd_hkid(u64 hkid, kvm_pfn_t pfn)
}
EXPORT_SYMBOL_FOR_KVM(tdh_phymem_page_wbinvd_hkid);
+bool tdx_supports_dynamic_pamt(const struct tdx_sys_info *sysinfo)
+{
+ return sysinfo->features.tdx_features0 & TDX_FEATURES0_DYNAMIC_PAMT;
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_supports_dynamic_pamt);
+
+static struct page *tdx_alloc_page_pamt_cache(struct tdx_pamt_cache *cache)
+{
+ struct page *page;
+
+ page = list_first_entry_or_null(&cache->page_list, struct page, lru);
+ if (page) {
+ list_del(&page->lru);
+ cache->cnt--;
+ }
+
+ return page;
+}
+
+static struct page *alloc_dpamt_page(struct tdx_pamt_cache *cache)
+{
+ if (cache)
+ return tdx_alloc_page_pamt_cache(cache);
+
+ return alloc_page(GFP_KERNEL_ACCOUNT);
+}
+
+static int alloc_pamt_array(struct page **pamt_pages, struct tdx_pamt_cache *cache)
+{
+ int i, j;
+
+ for (i = 0; i < TDX_DPAMT_ENTRY_PAGE_CNT; i++) {
+ pamt_pages[i] = alloc_dpamt_page(cache);
+ if (!pamt_pages[i])
+ goto err;
+ }
+
+ return 0;
+
+err:
+ for (j = 0; j < i; j++)
+ __free_page(pamt_pages[j]);
+
+ return -ENOMEM;
+}
+
+static void free_pamt_array(struct page **pamt_pages)
+{
+ int i;
+
+ for (i = 0; i < TDX_DPAMT_ENTRY_PAGE_CNT; i++) {
+ /*
+ * Reset pages unconditionally to cover cases
+ * where they were passed to the TDX module.
+ */
+ tdx_quirk_reset_paddr(page_to_phys(pamt_pages[i]), PAGE_SIZE);
+
+ __free_page(pamt_pages[i]);
+ }
+}
+
+/* Helper for building DPAMT seamcall() arguments. */
+static u64 pamt_2mb_arg(kvm_pfn_t pfn)
+{
+ /* Find the 2MB-wide DPAMT region for 'pfn': */
+ unsigned long hpa_2mb = ALIGN_DOWN(pfn << PAGE_SHIFT, PMD_SIZE);
+
+ /*
+ * TDX ABI requires specifying the page level the installed DPAMT
+ * backing will cover, even though today only 2MB is supported.
+ */
+ return hpa_2mb | TDX_PS_2M;
+}
+
+/* Add DPAMT backing for the 2MB region surrounding the given pfn. */
+static u64 tdh_phymem_pamt_add(kvm_pfn_t pfn, struct page **pamt_pages)
+{
+ struct tdx_module_args args = {
+ .rcx = pamt_2mb_arg(pfn),
+ .rdx = page_to_phys(pamt_pages[0]),
+ .r8 = page_to_phys(pamt_pages[1]),
+ };
+
+ return seamcall(TDH_PHYMEM_PAMT_ADD, &args);
+}
+
+/* Remove DPAMT backing for the 2MB region surrounding the given pfn. */
+static u64 tdh_phymem_pamt_remove(kvm_pfn_t pfn, struct page **pamt_pages)
+{
+ struct tdx_module_args args = {
+ .rcx = pamt_2mb_arg(pfn),
+ };
+ u64 ret;
+
+ ret = seamcall_ret(TDH_PHYMEM_PAMT_REMOVE, &args);
+ if (ret)
+ return ret;
+
+ /* Copy PAMT pages out of the struct per the TDX ABI */
+ pamt_pages[0] = phys_to_page(args.rdx);
+ pamt_pages[1] = phys_to_page(args.r8);
+
+ return 0;
+}
+
+/* Serializes adding/removing DPAMT memory */
+static DEFINE_SPINLOCK(dpamt_lock);
+
+/* Bump DPAMT refcount for the given pfn and allocate DPAMT backing if needed. */
+int tdx_pamt_get(kvm_pfn_t pfn, struct tdx_pamt_cache *cache)
+{
+ struct page *pamt_pages[TDX_DPAMT_ENTRY_PAGE_CNT];
+ atomic_t *dpamt_refcount;
+ u64 tdx_status;
+ int ret;
+
+ if (!tdx_supports_dynamic_pamt(&tdx_sysinfo))
+ return 0;
+
+ ret = alloc_pamt_array(pamt_pages, cache);
+ if (ret)
+ return ret;
+
+ dpamt_refcount = tdx_find_dpamt_refcount(pfn);
+
+ spin_lock(&dpamt_lock);
+
+ /*
+ * If the DPAMT entry is already added (i.e. refcount >= 1),
+ * then just increment the refcount.
+ */
+ if (atomic_inc_not_zero(dpamt_refcount))
+ goto out_free;
+
+ /* Try to add the PAMT page and take the refcount 0->1. */
+ tdx_status = tdh_phymem_pamt_add(pfn, pamt_pages);
+ if (WARN_ON_ONCE(tdx_status != TDX_SUCCESS)) {
+ ret = -EIO;
+ goto out_free;
+ }
+
+ atomic_set(dpamt_refcount, 1);
+ spin_unlock(&dpamt_lock);
+ return 0;
+
+out_free:
+ spin_unlock(&dpamt_lock);
+ free_pamt_array(pamt_pages);
+
+ return ret;
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_pamt_get);
+
+/* Drop DPAMT refcount for the given pfn and free DPAMT backing if needed. */
+void tdx_pamt_put(kvm_pfn_t pfn)
+{
+ struct page *pamt_pages[TDX_DPAMT_ENTRY_PAGE_CNT] = {};
+ atomic_t *dpamt_refcount;
+ u64 tdx_status;
+
+ if (!tdx_supports_dynamic_pamt(&tdx_sysinfo))
+ return;
+
+ dpamt_refcount = tdx_find_dpamt_refcount(pfn);
+
+ spin_lock(&dpamt_lock);
+ /*
+ * If there is more than 1 reference on the DPAMT entry, don't
+ * remove it yet. Just decrement the refcount.
+ */
+ if (atomic_read(dpamt_refcount) > 1) {
+ atomic_dec(dpamt_refcount);
+ goto out_unlock;
+ }
+
+ /* Try to remove the pamt page and take the refcount 1->0. */
+ tdx_status = tdh_phymem_pamt_remove(pfn, pamt_pages);
+
+ /*
+ * Don't free pamt_pages as it could hold garbage when
+ * tdh_phymem_pamt_remove() fails. Don't panic/BUG_ON(), as
+ * there is no risk of data corruption, but do yell loudly as
+ * failure indicates a kernel bug, memory is being leaked, and
+ * the dangling DPAMT entry may cause future operations to fail.
+ */
+ if (WARN_ON_ONCE(tdx_status != TDX_SUCCESS))
+ goto out_unlock;
+
+ atomic_set(dpamt_refcount, 0);
+ spin_unlock(&dpamt_lock);
+ free_pamt_array(pamt_pages);
+ return;
+out_unlock:
+ spin_unlock(&dpamt_lock);
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_pamt_put);
+
+void tdx_free_pamt_cache(struct tdx_pamt_cache *cache)
+{
+ struct page *page;
+
+ while ((page = tdx_alloc_page_pamt_cache(cache)))
+ __free_page(page);
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_free_pamt_cache);
+
+int tdx_topup_pamt_cache(struct tdx_pamt_cache *cache, unsigned long npages)
+{
+ if (WARN_ON_ONCE(!tdx_supports_dynamic_pamt(&tdx_sysinfo)))
+ return 0;
+
+ npages *= TDX_DPAMT_ENTRY_PAGE_CNT;
+
+ while (cache->cnt < npages) {
+ struct page *page = alloc_page(GFP_KERNEL_ACCOUNT);
+
+ if (!page)
+ return -ENOMEM;
+
+ list_add(&page->lru, &cache->page_list);
+ cache->cnt++;
+ }
+
+ return 0;
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_topup_pamt_cache);
+
+/*
+ * Return a page that can be gifted to the TDX module for use as a "control"
+ * page, i.e. pages that are used for control structures for a given TDX
+ * guest, and thus obtain TDX protections, including DPAMT tracking.
+ */
+struct page *tdx_alloc_control_page(void)
+{
+ struct page *page;
+
+ page = alloc_page(GFP_KERNEL_ACCOUNT);
+ if (!page)
+ return NULL;
+
+ if (tdx_pamt_get(page_to_pfn(page), NULL)) {
+ __free_page(page);
+ return NULL;
+ }
+
+ return page;
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_alloc_control_page);
+
+/*
+ * Free a page that was gifted to the TDX module for use as a control
+ * page. After this, the page is no longer protected by TDX.
+ */
+void tdx_free_control_page(struct page *page)
+{
+ if (!page)
+ return;
+
+ tdx_pamt_put(page_to_pfn(page));
+ __free_page(page);
+}
+EXPORT_SYMBOL_FOR_KVM(tdx_free_control_page);
+
void tdx_sys_disable(void)
{
struct tdx_module_args args = {};
diff --git a/arch/x86/virt/vmx/tdx/tdx.h b/arch/x86/virt/vmx/tdx/tdx.h
index bdfd0e1e337a..db209541d3cd 100644
--- a/arch/x86/virt/vmx/tdx/tdx.h
+++ b/arch/x86/virt/vmx/tdx/tdx.h
@@ -48,16 +48,10 @@
#define TDH_SYS_CONFIG 45
#define TDH_SYS_SHUTDOWN 52
#define TDH_SYS_UPDATE 53
+#define TDH_PHYMEM_PAMT_ADD 58
+#define TDH_PHYMEM_PAMT_REMOVE 59
#define TDH_SYS_DISABLE 69
-/*
- * SEAMCALL leaf:
- *
- * Bit 15:0 Leaf number
- * Bit 23:16 Version number
- */
-#define TDX_VERSION_SHIFT 16
-
/* TDX page types */
#define PT_NDA 0x0
#define PT_RSVD 0x1
diff --git a/arch/x86/virt/vmx/tdx/tdx_global_metadata.c b/arch/x86/virt/vmx/tdx/tdx_global_metadata.c
index e49c300f23d4..98ebf17aab1c 100644
--- a/arch/x86/virt/vmx/tdx/tdx_global_metadata.c
+++ b/arch/x86/virt/vmx/tdx/tdx_global_metadata.c
@@ -1,6 +1,6 @@
// SPDX-License-Identifier: GPL-2.0
/*
- * Automatically generated functions to read TDX global metadata.
+ * Functions to read TDX global metadata.
*
* This file doesn't compile on its own as it lacks of inclusion
* of SEAMCALL wrapper primitive which reads global metadata.
@@ -33,6 +33,18 @@ static __init int get_tdx_sys_info_features(struct tdx_sys_info_features *sysinf
return ret;
}
+static __init int get_tdx_sys_info_tdmr_dpamt(struct tdx_sys_info_tdmr *sysinfo_tdmr)
+{
+ int ret;
+ u64 val;
+
+ ret = read_sys_metadata_field(0x9100000000000013, &val);
+ if (!ret)
+ sysinfo_tdmr->pamt_page_bitmap_entry_bits = val;
+
+ return ret;
+}
+
static __init int get_tdx_sys_info_tdmr(struct tdx_sys_info_tdmr *sysinfo_tdmr)
{
int ret = 0;
@@ -129,5 +141,14 @@ static __init int get_tdx_sys_info(struct tdx_sys_info *sysinfo)
ret = ret ?: get_tdx_sys_info_td_ctrl(&sysinfo->td_ctrl);
ret = ret ?: get_tdx_sys_info_td_conf(&sysinfo->td_conf);
+ /*
+ * The kernel supports using TDX without DPAMT, so
+ * avoid reporting failure if it's not supported. Don't
+ * try to support buggy TDX modules that advertise
+ * DPAMT but don't expose the metadata.
+ */
+ if (!ret && tdx_supports_dynamic_pamt(sysinfo))
+ ret = get_tdx_sys_info_tdmr_dpamt(&sysinfo->tdmr);
+
return ret;
}
diff --git a/arch/x86/virt/vmx/tdx/tdxcall.S b/arch/x86/virt/vmx/tdx/tdxcall.S
index 016a2a1ec1d6..a194e83613e7 100644
--- a/arch/x86/virt/vmx/tdx/tdxcall.S
+++ b/arch/x86/virt/vmx/tdx/tdxcall.S
@@ -45,8 +45,14 @@
.macro TDX_MODULE_CALL host:req ret=0 saved=0
FRAME_BEGIN
- /* Move Leaf ID to RAX */
- mov %rdi, %rax
+ /* Leaf ABI version -> RAX[23:16]. Zero rest of RAX. */
+ movzbl TDX_MODULE_version(%rsi), %eax
+ shl $16, %eax
+ /*
+ * Combine leaf number arg and leaf ABI version into RAX, they don't
+ * overlap.
+ */
+ or %rdi, %rax
/* Move other input regs from 'struct tdx_module_args' */
movq TDX_MODULE_rcx(%rsi), %rcx
diff --git a/arch/x86/xen/pmu.c b/arch/x86/xen/pmu.c
index 5f50a3ee08f5..3f4dd3f50f56 100644
--- a/arch/x86/xen/pmu.c
+++ b/arch/x86/xen/pmu.c
@@ -456,12 +456,14 @@ static void xen_convert_regs(const struct xen_pmu_regs *xen_regs,
}
}
+static DEFINE_PER_CPU(struct x86_perf_regs, x86_xen_intr_regs);
irqreturn_t xen_pmu_irq_handler(int irq, void *dev_id)
{
int err, ret = IRQ_NONE;
struct pt_regs regs = {0};
const struct xen_pmu_data *xenpmu_data = get_xenpmu_data();
uint8_t xenpmu_flags = get_xenpmu_flags();
+ struct x86_perf_regs *x86_regs = this_cpu_ptr(&x86_xen_intr_regs);
if (!xenpmu_data) {
pr_warn_once("%s: pmudata not initialized\n", __func__);
@@ -472,7 +474,8 @@ irqreturn_t xen_pmu_irq_handler(int irq, void *dev_id)
xenpmu_flags | XENPMU_IRQ_PROCESSING;
xen_convert_regs(&xenpmu_data->pmu.r.regs, &regs,
xenpmu_data->pmu.pmu_flags);
- if (x86_pmu.handle_irq(&regs))
+ x86_regs->regs = regs;
+ if (x86_pmu.handle_irq(&x86_regs->regs))
ret = IRQ_HANDLED;
/* Write out cached context to HW */
diff --git a/drivers/base/cpu.c b/drivers/base/cpu.c
index 69e52fed4241..747915ff974f 100644
--- a/drivers/base/cpu.c
+++ b/drivers/base/cpu.c
@@ -391,6 +391,15 @@ static int cpu_uevent(const struct device *dev, struct kobj_uevent_env *env)
}
#endif
+#ifdef CONFIG_PREFERRED_CPU
+static ssize_t preferred_show(struct device *dev,
+ struct device_attribute *attr, char *buf)
+{
+ return sysfs_emit(buf, "%*pbl\n", cpumask_pr_args(cpu_preferred_mask));
+}
+static DEVICE_ATTR_RO(preferred);
+#endif
+
const struct bus_type cpu_subsys = {
.name = "cpu",
.dev_name = "cpu",
@@ -532,6 +541,9 @@ static struct attribute *cpu_root_attrs[] = {
#ifdef CONFIG_GENERIC_CPU_AUTOPROBE
&dev_attr_modalias.attr,
#endif
+#ifdef CONFIG_PREFERRED_CPU
+ &dev_attr_preferred.attr,
+#endif
NULL
};
diff --git a/drivers/clocksource/timer-clint.c b/drivers/clocksource/timer-clint.c
index 0bdd9d7ec545..e56eee7e3781 100644
--- a/drivers/clocksource/timer-clint.c
+++ b/drivers/clocksource/timer-clint.c
@@ -243,7 +243,7 @@ static int __init clint_timer_init_dt(struct device_node *np)
}
#ifdef CONFIG_SMP
- rc = ipi_mux_create(BITS_PER_BYTE, clint_send_ipi);
+ rc = ipi_mux_create(IPI_MAX, clint_send_ipi);
if (rc <= 0) {
pr_err("unable to create muxed IPIs\n");
rc = (rc < 0) ? rc : -ENODEV;
@@ -251,7 +251,7 @@ static int __init clint_timer_init_dt(struct device_node *np)
}
irq_set_chained_handler(clint_ipi_irq, clint_ipi_interrupt);
- riscv_ipi_set_virq_range(rc, BITS_PER_BYTE);
+ riscv_ipi_set_virq_range(rc, IPI_MAX);
clint_clear_ipi();
#endif
diff --git a/drivers/crypto/ccp/sev-dev.c b/drivers/crypto/ccp/sev-dev.c
index f833cb7e4da3..5c996ab63895 100644
--- a/drivers/crypto/ccp/sev-dev.c
+++ b/drivers/crypto/ccp/sev-dev.c
@@ -1663,6 +1663,8 @@ static int __sev_snp_init_locked(int *error, unsigned int max_snp_asid)
sev_es_tmr_size = SNP_TMR_SIZE;
+ snp_enable_rmpopt();
+
return 0;
}
diff --git a/drivers/firmware/efi/libstub/x86-stub.c b/drivers/firmware/efi/libstub/x86-stub.c
index cef32e2c82d8..80556a7e7552 100644
--- a/drivers/firmware/efi/libstub/x86-stub.c
+++ b/drivers/firmware/efi/libstub/x86-stub.c
@@ -10,6 +10,7 @@
#include <linux/pci.h>
#include <linux/stddef.h>
+#include <asm/cpuid/api.h>
#include <asm/efi.h>
#include <asm/e820/types.h>
#include <asm/setup.h>
@@ -17,6 +18,7 @@
#include <asm/boot.h>
#include <asm/kaslr.h>
#include <asm/sev.h>
+#include <asm/shared/tdx.h>
#include "efistub.h"
#include "x86-stub.h"
@@ -1068,3 +1070,40 @@ void efi64_stub_entry(efi_handle_t handle, efi_system_table_t *sys_table_arg,
struct boot_params *boot_params);
#endif
#endif
+
+#ifdef CONFIG_UNACCEPTED_MEMORY
+/*
+ * process_unaccepted_memory() is called after ExitBootServices(), and so these
+ * memory acceptance routines cannot rely on EFI protocols for detecting the
+ * presence of TDX or SEV-SNP, or emit any kind of output if any error
+ * conditions are detected.
+ */
+static bool early_is_tdx_guest(void)
+{
+ static bool once;
+ static bool is_tdx;
+
+ if (!IS_ENABLED(CONFIG_INTEL_TDX_GUEST))
+ return false;
+
+ if (!once) {
+ u32 eax = TDX_CPUID_LEAF_ID, sig[3] = {};
+
+ native_cpuid(&eax, &sig[0], &sig[2], &sig[1]);
+ is_tdx = !memcmp(TDX_IDENT, sig, sizeof(sig));
+ once = true;
+ }
+
+ return is_tdx;
+}
+
+void arch_accept_memory(phys_addr_t start, phys_addr_t end)
+{
+ if (early_is_tdx_guest()) {
+ if (!tdx_accept_memory(start, end))
+ tdx_panic("Failed to accept memory");
+ } else if (early_is_sevsnp_guest()) {
+ snp_accept_memory(start, end);
+ }
+}
+#endif
diff --git a/drivers/irqchip/Kconfig b/drivers/irqchip/Kconfig
index 20b77fbc51ee..a8f8c423b795 100644
--- a/drivers/irqchip/Kconfig
+++ b/drivers/irqchip/Kconfig
@@ -522,17 +522,19 @@ config GOLDFISH_PIC
config QCOM_PDC
tristate "Qualcomm PDC"
- depends on ARCH_QCOM
+ depends on ARCH_QCOM || COMPILE_TEST
select IRQ_DOMAIN_HIERARCHY
+ default ARCH_QCOM
help
Power Domain Controller driver to manage and configure wakeup
IRQs for Qualcomm Technologies Inc (QTI) mobile chips.
config QCOM_MPM
tristate "Qualcomm MPM"
- depends on ARCH_QCOM
+ depends on ARCH_QCOM || COMPILE_TEST
depends on MAILBOX
select IRQ_DOMAIN_HIERARCHY
+ default ARCH_QCOM if ARM64
help
MSM Power Manager driver to manage and configure wakeup
IRQs for Qualcomm Technologies Inc (QTI) mobile chips.
diff --git a/drivers/irqchip/irq-aclint-sswi.c b/drivers/irqchip/irq-aclint-sswi.c
index ca06efd86fa1..5e010fc401f7 100644
--- a/drivers/irqchip/irq-aclint-sswi.c
+++ b/drivers/irqchip/irq-aclint-sswi.c
@@ -138,7 +138,7 @@ static int __init aclint_sswi_probe(struct fwnode_handle *fwnode)
}
/* Register SSWI irq and handler */
- virq = ipi_mux_create(BITS_PER_BYTE, aclint_sswi_ipi_send);
+ virq = ipi_mux_create(IPI_MAX, aclint_sswi_ipi_send);
if (virq <= 0) {
pr_err("unable to create muxed IPIs\n");
irq_dispose_mapping(sswi_ipi_virq);
@@ -152,7 +152,7 @@ static int __init aclint_sswi_probe(struct fwnode_handle *fwnode)
aclint_sswi_starting_cpu,
aclint_sswi_dying_cpu);
- riscv_ipi_set_virq_range(virq, BITS_PER_BYTE);
+ riscv_ipi_set_virq_range(virq, IPI_MAX);
return 0;
}
diff --git a/drivers/irqchip/irq-al-fic.c b/drivers/irqchip/irq-al-fic.c
index d10ac9b63c99..760317db9dba 100644
--- a/drivers/irqchip/irq-al-fic.c
+++ b/drivers/irqchip/irq-al-fic.c
@@ -180,7 +180,7 @@ err_domain_remove:
* @name: name of the fic
* @parent_irq: interrupt of parent
*
- * This API will configure the fic hardware to to work in wire mode.
+ * This API will configure the fic hardware to work in wire mode.
* In wire mode, fic hardware is generating a wire ("wired") interrupt.
* Interrupt can be generated based on positive edge or level - configuration is
* to be determined based on connected hardware to this fic.
diff --git a/drivers/irqchip/irq-gic-v3-its.c b/drivers/irqchip/irq-gic-v3-its.c
index e9807af23537..b6509b495d6d 100644
--- a/drivers/irqchip/irq-gic-v3-its.c
+++ b/drivers/irqchip/irq-gic-v3-its.c
@@ -2248,7 +2248,8 @@ out:
static void its_lpi_free(unsigned long *bitmap, u32 base, u32 nr_ids)
{
- WARN_ON(free_lpi_range(base, nr_ids));
+ if (free_lpi_range(base, nr_ids))
+ pr_err_ratelimited("ITS: failed to free LPI range %u:%u\n", base, nr_ids);
bitmap_free(bitmap);
}
@@ -4895,6 +4896,9 @@ static bool __maybe_unused its_enable_quirk_hip09_162100801(void *data)
}
static const char * const dma_32bit_impaired_platforms[] = {
+#ifdef CONFIG_ALTERA_ERRATUM_AGILEX5_2_1_23
+ "intel,socfpga-agilex5",
+#endif
#ifdef CONFIG_RENESAS_ERRATUM_GEN4GICITS1
"renesas,r8a779f0",
"renesas,r8a779g0",
diff --git a/drivers/irqchip/irq-gic-v3.c b/drivers/irqchip/irq-gic-v3.c
index 6e1fa5b247fc..19a15c2ea61b 100644
--- a/drivers/irqchip/irq-gic-v3.c
+++ b/drivers/irqchip/irq-gic-v3.c
@@ -687,7 +687,7 @@ static void gic_eoimode1_eoi_irq(struct irq_data *d)
{
/*
* No need to deactivate an LPI, or an interrupt that
- * is is getting forwarded to a vcpu.
+ * is getting forwarded to a vcpu.
*/
if (irqd_to_hwirq(d) >= 8192 || irqd_is_forwarded_to_vcpu(d))
return;
@@ -2276,7 +2276,6 @@ static struct
bool single_redist;
int enabled_rdists;
u32 maint_irq;
- int maint_irq_mode;
phys_addr_t vcpu_base;
} acpi_data __initdata;
@@ -2454,21 +2453,19 @@ static int __init gic_acpi_parse_virt_madt_gicc(union acpi_subtable_headers *hea
{
struct acpi_madt_generic_interrupt *gicc =
(struct acpi_madt_generic_interrupt *)header;
- int maint_irq_mode;
static int first_madt = true;
if (!(gicc->flags &
(ACPI_MADT_ENABLED | ACPI_MADT_GICC_ONLINE_CAPABLE)))
return 0;
- maint_irq_mode = (gicc->flags & ACPI_MADT_VGIC_IRQ_MODE) ?
- ACPI_EDGE_SENSITIVE : ACPI_LEVEL_SENSITIVE;
+ if (gicc->flags & ACPI_MADT_VGIC_IRQ_MODE)
+ pr_warn_once(FW_BUG "MI wrongly advertised as Edge-triggered\n");
if (first_madt) {
first_madt = false;
acpi_data.maint_irq = gicc->vgic_interrupt;
- acpi_data.maint_irq_mode = maint_irq_mode;
acpi_data.vcpu_base = gicc->gicv_base_address;
return 0;
@@ -2478,7 +2475,6 @@ static int __init gic_acpi_parse_virt_madt_gicc(union acpi_subtable_headers *hea
* The maintenance interrupt and GICV should be the same for every CPU
*/
if ((acpi_data.maint_irq != gicc->vgic_interrupt) ||
- (acpi_data.maint_irq_mode != maint_irq_mode) ||
(acpi_data.vcpu_base != gicc->gicv_base_address))
return -EINVAL;
@@ -2511,7 +2507,7 @@ static void __init gic_acpi_setup_kvm_info(void)
gic_v3_kvm_info.type = GIC_V3;
irq = acpi_register_gsi(NULL, acpi_data.maint_irq,
- acpi_data.maint_irq_mode,
+ ACPI_LEVEL_SENSITIVE,
ACPI_ACTIVE_HIGH);
if (irq <= 0)
return;
diff --git a/drivers/irqchip/irq-gic-v5.c b/drivers/irqchip/irq-gic-v5.c
index 5f2551cf077d..74be71be479a 100644
--- a/drivers/irqchip/irq-gic-v5.c
+++ b/drivers/irqchip/irq-gic-v5.c
@@ -1168,21 +1168,18 @@ static int __init gicv5_init_common(struct fwnode_handle *parent_domain)
if (ret)
goto out_int;
- ret = set_handle_irq(gicv5_handle_irq);
+ ret = gicv5_irs_enable();
if (ret)
goto out_int;
- ret = gicv5_irs_enable();
- if (ret)
- goto out_handle;
+ if (set_handle_irq(gicv5_handle_irq))
+ panic("GICv5: unable to install root IRQ handler\n");
gicv5_smp_init();
gicv5_irs_its_probe();
return 0;
-out_handle:
- set_handle_irq(NULL);
out_int:
gicv5_cpu_disable_interrupts();
gicv5_free_domains();
diff --git a/drivers/irqchip/irq-gic.c b/drivers/irqchip/irq-gic.c
index f6bc29f515fb..b2926a3ddaf1 100644
--- a/drivers/irqchip/irq-gic.c
+++ b/drivers/irqchip/irq-gic.c
@@ -1527,7 +1527,6 @@ static struct
{
phys_addr_t cpu_phys_base;
u32 maint_irq;
- int maint_irq_mode;
phys_addr_t vctrl_base;
phys_addr_t vcpu_base;
} acpi_data __initdata;
@@ -1553,10 +1552,11 @@ gic_acpi_parse_madt_cpu(union acpi_subtable_headers *header,
if (cpu_base_assigned && gic_cpu_base != acpi_data.cpu_phys_base)
return -EINVAL;
+ if (processor->flags & ACPI_MADT_VGIC_IRQ_MODE)
+ pr_warn_once(FW_BUG "MI wrongly advertised as Edge-triggered\n");
+
acpi_data.cpu_phys_base = gic_cpu_base;
acpi_data.maint_irq = processor->vgic_interrupt;
- acpi_data.maint_irq_mode = (processor->flags & ACPI_MADT_VGIC_IRQ_MODE) ?
- ACPI_EDGE_SENSITIVE : ACPI_LEVEL_SENSITIVE;
acpi_data.vctrl_base = processor->gich_base_address;
acpi_data.vcpu_base = processor->gicv_base_address;
@@ -1616,7 +1616,7 @@ static void __init gic_acpi_setup_kvm_info(void)
vcpu_res->end = vcpu_res->start + ACPI_GICV2_VCPU_MEM_SIZE - 1;
irq = acpi_register_gsi(NULL, acpi_data.maint_irq,
- acpi_data.maint_irq_mode,
+ ACPI_LEVEL_SENSITIVE,
ACPI_ACTIVE_HIGH);
if (irq <= 0)
return;
diff --git a/drivers/irqchip/irq-lan966x-oic.c b/drivers/irqchip/irq-lan966x-oic.c
index 8af08d0e4182..5122f3f1b353 100644
--- a/drivers/irqchip/irq-lan966x-oic.c
+++ b/drivers/irqchip/irq-lan966x-oic.c
@@ -220,7 +220,6 @@ static int lan966x_oic_probe(struct platform_device *pdev)
};
struct irq_domain_info d_info = {
.fwnode = of_fwnode_handle(pdev->dev.of_node),
- .domain_flags = IRQ_DOMAIN_FLAG_DESTROY_GC,
.size = LAN966X_OIC_NR_IRQ,
.hwirq_max = LAN966X_OIC_NR_IRQ,
.ops = &irq_generic_chip_ops,
diff --git a/drivers/irqchip/irq-mtk-cirq.c b/drivers/irqchip/irq-mtk-cirq.c
index 914d1d639fe3..d30c34ce0f56 100644
--- a/drivers/irqchip/irq-mtk-cirq.c
+++ b/drivers/irqchip/irq-mtk-cirq.c
@@ -247,7 +247,7 @@ static int mtk_cirq_suspend(void *data)
writel_relaxed(mask, reg);
}
- /* set edge_only mode, record edge-triggerd interrupts */
+ /* set edge_only mode, record edge-triggered interrupts */
/* enable cirq */
reg = mtk_cirq_reg(cirq_data, CIRQ_CONTROL);
value = readl_relaxed(reg);
diff --git a/drivers/irqchip/irq-pruss-intc.c b/drivers/irqchip/irq-pruss-intc.c
index 81078d56f38d..5a3e9e5bccbe 100644
--- a/drivers/irqchip/irq-pruss-intc.c
+++ b/drivers/irqchip/irq-pruss-intc.c
@@ -181,7 +181,7 @@ static void pruss_intc_map(struct pruss_intc *intc, unsigned long hwirq)
u8 ch, host, reg_idx;
u32 val;
- mutex_lock(&intc->lock);
+ guard(mutex)(&intc->lock);
intc->event_channel[hwirq].ref_count++;
@@ -206,8 +206,6 @@ static void pruss_intc_map(struct pruss_intc *intc, unsigned long hwirq)
dev_dbg(dev, "mapped system_event = %lu channel = %d host = %d",
hwirq, ch, host);
-
- mutex_unlock(&intc->lock);
}
/**
@@ -224,7 +222,7 @@ static void pruss_intc_unmap(struct pruss_intc *intc, unsigned long hwirq)
u8 ch, host, reg_idx;
u32 val;
- mutex_lock(&intc->lock);
+ guard(mutex)(&intc->lock);
ch = intc->event_channel[hwirq].value;
host = intc->channel_host[ch].value;
@@ -251,8 +249,6 @@ static void pruss_intc_unmap(struct pruss_intc *intc, unsigned long hwirq)
dev_dbg(intc->dev, "unmapped system_event = %lu channel = %d host = %d\n",
hwirq, ch, host);
-
- mutex_unlock(&intc->lock);
}
static void pruss_intc_init(struct pruss_intc *intc)
@@ -376,17 +372,15 @@ static int pruss_intc_validate_mapping(struct pruss_intc *intc, int event,
int channel, int host)
{
struct device *dev = intc->dev;
- int ret = 0;
- mutex_lock(&intc->lock);
+ guard(mutex)(&intc->lock);
/* check if sysevent already assigned */
if (intc->event_channel[event].ref_count > 0 &&
intc->event_channel[event].value != channel) {
dev_err(dev, "event %d (req. ch %d) already assigned to channel %d\n",
event, channel, intc->event_channel[event].value);
- ret = -EBUSY;
- goto unlock;
+ return -EBUSY;
}
/* check if channel already assigned */
@@ -394,16 +388,13 @@ static int pruss_intc_validate_mapping(struct pruss_intc *intc, int event,
intc->channel_host[channel].value != host) {
dev_err(dev, "channel %d (req. host %d) already assigned to host %d\n",
channel, host, intc->channel_host[channel].value);
- ret = -EBUSY;
- goto unlock;
+ return -EBUSY;
}
intc->event_channel[event].value = channel;
intc->channel_host[channel].value = host;
-unlock:
- mutex_unlock(&intc->lock);
- return ret;
+ return 0;
}
static int
@@ -516,24 +507,21 @@ static const char * const irq_names[MAX_NUM_HOST_IRQS] = {
static int pruss_intc_probe(struct platform_device *pdev)
{
- const struct pruss_intc_match_data *data;
struct device *dev = &pdev->dev;
struct pruss_intc *intc;
struct pruss_host_irq_data *host_data;
int i, irq, ret;
u8 max_system_events, irqs_reserved = 0;
- data = of_device_get_match_data(dev);
- if (!data)
- return -ENODEV;
-
- max_system_events = data->num_system_events;
-
intc = devm_kzalloc(dev, sizeof(*intc), GFP_KERNEL);
if (!intc)
return -ENOMEM;
- intc->soc_config = data;
+ intc->soc_config = of_device_get_match_data(dev);
+ if (!intc->soc_config)
+ return -ENODEV;
+ max_system_events = intc->soc_config->num_system_events;
+
intc->dev = dev;
platform_set_drvdata(pdev, intc);
@@ -553,7 +541,9 @@ static int pruss_intc_probe(struct platform_device *pdev)
pruss_intc_init(intc);
- mutex_init(&intc->lock);
+ ret = devm_mutex_init(dev, &intc->lock);
+ if (ret)
+ return ret;
intc->domain = irq_domain_create_linear(dev_fwnode(dev), max_system_events,
&pruss_intc_irq_domain_ops, intc);
diff --git a/drivers/irqchip/irq-riscv-imsic-early.c b/drivers/irqchip/irq-riscv-imsic-early.c
index 12efd241ce88..823f5f2ecb3d 100644
--- a/drivers/irqchip/irq-riscv-imsic-early.c
+++ b/drivers/irqchip/irq-riscv-imsic-early.c
@@ -67,12 +67,12 @@ static int __init imsic_ipi_domain_init(void)
return 0;
/* Create IMSIC IPI multiplexing */
- virq = ipi_mux_create(IMSIC_NR_IPI, imsic_ipi_send);
+ virq = ipi_mux_create(IPI_MAX, imsic_ipi_send);
if (virq <= 0)
return virq < 0 ? virq : -ENOMEM;
/* Set vIRQ range */
- riscv_ipi_set_virq_range(virq, IMSIC_NR_IPI);
+ riscv_ipi_set_virq_range(virq, IPI_MAX);
/* Announce that IMSIC is providing IPIs */
pr_info("%pfwP: providing IPIs using interrupt %d\n", imsic->fwnode, IMSIC_IPI_ID);
diff --git a/drivers/irqchip/irq-riscv-imsic-state.c b/drivers/irqchip/irq-riscv-imsic-state.c
index b8d1bbbf42f7..9505ddbd9eec 100644
--- a/drivers/irqchip/irq-riscv-imsic-state.c
+++ b/drivers/irqchip/irq-riscv-imsic-state.c
@@ -7,6 +7,7 @@
#define pr_fmt(fmt) "riscv-imsic: " fmt
#include <linux/acpi.h>
#include <linux/cpu.h>
+#include <linux/bits.h>
#include <linux/bitmap.h>
#include <linux/interrupt.h>
#include <linux/irq.h>
@@ -769,9 +770,9 @@ static int __init imsic_parse_fwnode(struct fwnode_handle *fwnode,
return -EINVAL;
}
global->base_addr = res.start;
- global->base_addr &= ~(BIT(global->guest_index_bits +
- global->hart_index_bits +
- IMSIC_MMIO_PAGE_SHIFT) - 1);
+ global->base_addr &= ~GENMASK(global->guest_index_bits +
+ global->hart_index_bits +
+ IMSIC_MMIO_PAGE_SHIFT - 1, 0);
global->base_addr &= ~((BIT(global->group_index_bits) - 1) <<
global->group_index_shift);
@@ -850,9 +851,9 @@ int __init imsic_setup_state(struct fwnode_handle *fwnode, void *opaque)
}
base_addr = mmios[i].start;
- base_addr &= ~(BIT(global->guest_index_bits +
- global->hart_index_bits +
- IMSIC_MMIO_PAGE_SHIFT) - 1);
+ base_addr &= ~GENMASK(global->guest_index_bits +
+ global->hart_index_bits +
+ IMSIC_MMIO_PAGE_SHIFT - 1, 0);
base_addr &= ~((BIT(global->group_index_bits) - 1) <<
global->group_index_shift);
if (base_addr != global->base_addr) {
diff --git a/drivers/irqchip/irq-riscv-imsic-state.h b/drivers/irqchip/irq-riscv-imsic-state.h
index c42ee180b305..878cc192ccec 100644
--- a/drivers/irqchip/irq-riscv-imsic-state.h
+++ b/drivers/irqchip/irq-riscv-imsic-state.h
@@ -13,7 +13,6 @@
#include <linux/timer.h>
#define IMSIC_IPI_ID 1
-#define IMSIC_NR_IPI 8
struct imsic_vector {
/* Fixed details of the vector */
diff --git a/drivers/irqchip/irq-sifive-plic.c b/drivers/irqchip/irq-sifive-plic.c
index 5b0dac104814..a7cddadddf40 100644
--- a/drivers/irqchip/irq-sifive-plic.c
+++ b/drivers/irqchip/irq-sifive-plic.c
@@ -452,7 +452,7 @@ static irq_hw_number_t cp100_get_hwirq(struct plic_handler *handler, void __iome
return 0;
/*
- * Interrupts delievered to hardware still become pending, but only
+ * Interrupts delivered to hardware still become pending, but only
* interrupts that are both pending and enabled can be claimed.
* Clearing the enable bit for all interrupts but the first pending
* one avoids a hardware bug that occurs during read from the claim
diff --git a/drivers/irqchip/irq-vic.c b/drivers/irqchip/irq-vic.c
index e38104c5064e..607e3284f700 100644
--- a/drivers/irqchip/irq-vic.c
+++ b/drivers/irqchip/irq-vic.c
@@ -478,7 +478,7 @@ static void __init __vic_init(void __iomem *base, int parent_irq, int irq_start,
/**
* vic_init() - initialise a vectored interrupt controller
* @base: iomem base address
- * @irq_start: starting interrupt number, must be muliple of 32
+ * @irq_start: starting interrupt number, must be multiple of 32
* @vic_sources: bitmask of interrupt sources to allow
* @resume_sources: bitmask of interrupt sources to allow for resume
*/
diff --git a/drivers/irqchip/qcom-pdc.c b/drivers/irqchip/qcom-pdc.c
index ce6d80c7f17a..29025a212ece 100644
--- a/drivers/irqchip/qcom-pdc.c
+++ b/drivers/irqchip/qcom-pdc.c
@@ -715,7 +715,10 @@ static int qcom_pdc_probe(struct platform_device *pdev, struct device_node *pare
}
pdc->x1e_quirk = true;
+ }
+ if (of_device_is_compatible(node, "qcom,x1e80100-pdc") ||
+ of_device_is_compatible(node, "qcom,x1p42100-pdc")) {
if (!qcom_scm_is_available())
return -EPROBE_DEFER;
diff --git a/drivers/resctrl/mpam_resctrl.c b/drivers/resctrl/mpam_resctrl.c
index 9d223057953a..f2c651e9f6ea 100644
--- a/drivers/resctrl/mpam_resctrl.c
+++ b/drivers/resctrl/mpam_resctrl.c
@@ -148,6 +148,11 @@ bool resctrl_arch_get_cdp_enabled(enum resctrl_res_level rid)
return mpam_resctrl_controls[rid].cdp_enabled;
}
+u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val)
+{
+ return val;
+}
+
/**
* resctrl_reset_task_closids() - Reset the PARTID/PMG values for all tasks.
*
diff --git a/drivers/soc/fsl/qe/qe_ports_ic.c b/drivers/soc/fsl/qe/qe_ports_ic.c
index 7375f92f528b..fb3b92039547 100644
--- a/drivers/soc/fsl/qe/qe_ports_ic.c
+++ b/drivers/soc/fsl/qe/qe_ports_ic.c
@@ -140,7 +140,6 @@ static int qepic_probe(struct platform_device *pdev)
};
struct irq_domain_info d_info = {
.fwnode = of_fwnode_handle(pdev->dev.of_node),
- .domain_flags = IRQ_DOMAIN_FLAG_DESTROY_GC,
.size = 32,
.hwirq_max = 32,
.ops = &irq_generic_chip_ops,
diff --git a/drivers/virt/Kconfig b/drivers/virt/Kconfig
index 52eb7e4ba71f..eeb84e578ddf 100644
--- a/drivers/virt/Kconfig
+++ b/drivers/virt/Kconfig
@@ -41,6 +41,23 @@ config FSL_HV_MANAGER
4) A kernel interface for receiving callbacks when a managed
partition shuts down.
+config STEAL_GOVERNOR
+ tristate "Dynamic vCPU management based on steal time"
+ depends on PARAVIRT && SMP
+ select PREFERRED_CPU
+ default m
+ help
+ This driver helps to reduce the steal time in paravirtualized
+ environments, thereby reducing vCPU preemption costs.
+
+ By default preferred CPUs will be same as active CPUs. Depending
+ on the steal time when steal_governor driver is enabled,
+ preferred CPUs could become subset of active CPUs.
+ More details are at: Documentation/driver-api/steal-governor.rst
+
+ It is recommended to build it as module and load the module
+ to enable it.
+
source "drivers/virt/vboxguest/Kconfig"
source "drivers/virt/nitro_enclaves/Kconfig"
diff --git a/drivers/virt/Makefile b/drivers/virt/Makefile
index f29901bd7820..05fb075ef5b8 100644
--- a/drivers/virt/Makefile
+++ b/drivers/virt/Makefile
@@ -5,6 +5,7 @@
obj-$(CONFIG_FSL_HV_MANAGER) += fsl_hypervisor.o
obj-$(CONFIG_VMGENID) += vmgenid.o
+obj-$(CONFIG_STEAL_GOVERNOR) += steal_governor.o
obj-y += vboxguest/
obj-$(CONFIG_NITRO_ENCLAVES) += nitro_enclaves/
diff --git a/drivers/virt/coco/tdx-guest/tdx-guest.c b/drivers/virt/coco/tdx-guest/tdx-guest.c
index d0303e31e816..a21bd0376b74 100644
--- a/drivers/virt/coco/tdx-guest/tdx-guest.c
+++ b/drivers/virt/coco/tdx-guest/tdx-guest.c
@@ -265,7 +265,7 @@ static int wait_for_quote_completion(struct tdx_quote_buf *quote_buf, u32 timeou
return (i == timeout) ? -ETIMEDOUT : 0;
}
-static int tdx_report_new_locked(struct tsm_report *report, void *data)
+static int tdx_report_new_locked(struct tsm_report *report)
{
u8 *buf;
struct tdx_quote_buf *quote_buf = quote_data;
@@ -333,10 +333,10 @@ static int tdx_report_new_locked(struct tsm_report *report, void *data)
return ret;
}
-static int tdx_report_new(struct tsm_report *report, void *data)
+static int tdx_report_new(struct tsm_report *report, void *unused)
{
scoped_cond_guard(mutex_intr, return -EINTR, &quote_lock)
- return tdx_report_new_locked(report, data);
+ return tdx_report_new_locked(report);
}
static bool tdx_report_attr_visible(int n)
diff --git a/drivers/virt/steal_governor.c b/drivers/virt/steal_governor.c
new file mode 100644
index 000000000000..6e31f9923dea
--- /dev/null
+++ b/drivers/virt/steal_governor.c
@@ -0,0 +1,296 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Steal time governor driver periodically computes steal time.
+ * Based on the thresholds it either reduce/increase the preferred
+ * CPUs which can be used by the workload to avoid vCPU preemption
+ * to an extent possible in paravirtualized environment.
+ *
+ * Available with CONFIG_STEAL_GOVERNOR
+ *
+ * Copyright (C) 2026 IBM
+ * Author: Shrikanth Hegde <sshegde@linux.ibm.com>
+ */
+
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/cleanup.h>
+#include <linux/cpuhplock.h>
+#include <linux/cpumask.h>
+#include <linux/init.h>
+#include <linux/kernel.h>
+#include <linux/kernel_stat.h>
+#include <linux/kconfig.h>
+#include <linux/ktime.h>
+#include <linux/math64.h>
+#include <linux/module.h>
+#include <linux/sched/isolation.h>
+#include <linux/topology.h>
+#include <linux/types.h>
+#include <linux/workqueue.h>
+#ifdef CONFIG_XEN
+#include <xen/xen.h>
+#endif
+
+#if !IS_ENABLED(CONFIG_PREFERRED_CPU)
+#error "Steal Governor requires CONFIG_PREFERRED_CPU"
+#endif
+
+struct steal_governor {
+ ktime_t time;
+ u64 steal;
+ unsigned long delay;
+ unsigned int interval_ms;
+ unsigned int high_threshold;
+ unsigned int low_threshold;
+ struct delayed_work work;
+};
+
+static struct steal_governor sg_ctx = {
+ .interval_ms = 1000, /* 1 second */
+ .high_threshold = 500, /* 5% */
+ .low_threshold = 200, /* 2% */
+};
+
+static void restore_preferred_to_active(void)
+{
+ int cpu;
+
+ guard(cpus_read_lock)();
+ for_each_cpu(cpu, cpu_active_mask)
+ set_cpu_preferred(cpu, true);
+}
+
+static int param_set_interval_ms(const char *val, const struct kernel_param *kp)
+{
+ unsigned int interval;
+ int ret;
+
+ ret = kstrtouint(val, 0, &interval);
+ if (ret)
+ return ret;
+
+ if (interval < 100 || interval > 100000) {
+ pr_err("interval_ms must be between 100 and 100000\n");
+ return -EINVAL;
+ }
+
+ return param_set_uint(val, kp);
+}
+
+static const struct kernel_param_ops interval_ms_ops = {
+ .set = param_set_interval_ms,
+ .get = param_get_uint,
+};
+
+module_param_cb(interval_ms, &interval_ms_ops, &sg_ctx.interval_ms, 0444);
+MODULE_PARM_DESC(interval_ms,
+ "Sampling frequency in milliseconds. default: 1000");
+
+static int param_set_high_threshold(const char *val, const struct kernel_param *kp)
+{
+ unsigned int threshold;
+ int ret;
+
+ ret = kstrtouint(val, 0, &threshold);
+ if (ret)
+ return ret;
+
+ if (threshold >= 100 * 100) {
+ pr_err("high_threshold (%u) can't be more than 99.99%%\n", threshold);
+ return -EINVAL;
+ }
+
+ return param_set_uint(val, kp);
+}
+
+static const struct kernel_param_ops high_threshold_ops = {
+ .set = param_set_high_threshold,
+ .get = param_get_uint,
+};
+
+module_param_cb(high_threshold, &high_threshold_ops, &sg_ctx.high_threshold, 0444);
+MODULE_PARM_DESC(high_threshold,
+ "High steal threshold. default: 500 i.e 5%. Must be > low_threshold");
+
+module_param_named(low_threshold, sg_ctx.low_threshold, uint, 0444);
+MODULE_PARM_DESC(low_threshold,
+ "Low steal threshold. default: 200 i.e 2%. Must be < high_threshold");
+
+/* Return collective steal time across system. */
+static u64 get_system_steal_time(void)
+{
+ return kcpustat_field_total(CPUTIME_STEAL, cpu_possible_mask);
+}
+
+/* Return number of CPUs to consider for steal ratio. */
+static unsigned int get_system_cpus(void)
+{
+ return num_active_cpus();
+}
+
+/*
+ * Called when the steal governor detects high physical CPU contention.
+ * It finds the last active core in the preferred mask and mark those
+ * CPUs as non-preferred.
+ *
+ * Must ensure:
+ * - at least one core is always kept as preferred
+ * - preferred is always subset of active.
+ */
+static void decrease_preferred_cpus(void)
+{
+ const struct cpumask *first_hk_core;
+ int target_cpu = nr_cpu_ids;
+ int cpu;
+
+ guard(cpus_read_lock)();
+ cpu = cpumask_first_and(housekeeping_cpumask(HK_TYPE_KERNEL_NOISE),
+ cpu_preferred_mask);
+ if (cpu >= nr_cpu_ids)
+ return;
+
+ /* Always leave first housekeeping core as preferred. */
+ first_hk_core = topology_sibling_cpumask(cpu);
+ cpu = cpumask_last(cpu_preferred_mask);
+ if (cpu >= nr_cpu_ids)
+ return;
+
+ /* Find the last CPU which doesn't belong to that first hk_core. */
+ if (!cpumask_test_cpu(cpu, first_hk_core)) {
+ target_cpu = cpu;
+ } else {
+ for_each_cpu_andnot(cpu, cpu_preferred_mask, first_hk_core)
+ target_cpu = cpu;
+ }
+
+ /* Only the first housekeeping core remains */
+ if (target_cpu >= nr_cpu_ids)
+ return;
+
+ for_each_cpu_and(cpu, topology_sibling_cpumask(target_cpu),
+ cpu_preferred_mask)
+ set_cpu_preferred(cpu, false);
+}
+
+/*
+ * Called when the steal governor detects no/low physical CPU contention.
+ * It finds the first active core outside of preferred mask and mark
+ * those CPUs as preferred.
+ *
+ * Must ensure preferred is subset of active.
+ */
+static void increase_preferred_cpus(void)
+{
+ int first_cpu, cpu;
+
+ guard(cpus_read_lock)();
+ first_cpu = cpumask_first_andnot(cpu_active_mask, cpu_preferred_mask);
+
+ /* All CPUs are preferred. Nothing to increase further */
+ if (first_cpu >= nr_cpu_ids)
+ return;
+
+ for_each_cpu_and(cpu, topology_sibling_cpumask(first_cpu),
+ cpu_active_mask)
+ set_cpu_preferred(cpu, true);
+}
+
+static bool preferred_cpus_valid(void)
+{
+ if (cpumask_empty(cpu_preferred_mask)) {
+ pr_err("empty preferred mask. stopping\n");
+ return false;
+ }
+
+ if (!cpumask_subset(cpu_preferred_mask, cpu_active_mask)) {
+ pr_err("preferred: %*pbl is not subset of active: %*pbl, stopping\n",
+ cpumask_pr_args(cpu_preferred_mask),
+ cpumask_pr_args(cpu_active_mask));
+ return false;
+ }
+
+ return true;
+}
+
+static void steal_governor_loop(struct work_struct *work)
+{
+ u64 curr_steal, delta_steal, delta_ns, steal_ratio;
+ ktime_t now;
+
+ now = ktime_get();
+ delta_ns = ktime_to_ns(ktime_sub(now, sg_ctx.time));
+
+ if (unlikely(delta_ns < NSEC_PER_MSEC)) {
+ pr_err_ratelimited("work scheduled too soon delta_ns: %llu\n", delta_ns);
+ goto requeue_work;
+ }
+
+ curr_steal = get_system_steal_time();
+ delta_steal = curr_steal > sg_ctx.steal ? curr_steal - sg_ctx.steal : 0;
+ sg_ctx.steal = curr_steal;
+ sg_ctx.time = now;
+
+ /*
+ * steal_ratio = (delta_steal * 100*100)/(delta_ns * num_cpus())
+ * To avoid possible overflow, divide the denominator early.
+ * Note minimum interval is 100ms.
+ */
+ delta_ns = max_t(u64, div_u64(delta_ns * get_system_cpus(), 10000), 1);
+ steal_ratio = div64_u64(delta_steal, delta_ns);
+
+ if (steal_ratio > sg_ctx.high_threshold)
+ decrease_preferred_cpus();
+ else if (steal_ratio <= sg_ctx.low_threshold)
+ increase_preferred_cpus();
+ /*
+ * else: steal ratio is within bounds. Still do design checks so that
+ * module restores to active if CPU hotplug breaks those assumptions.
+ */
+ if (!preferred_cpus_valid()) {
+ restore_preferred_to_active();
+ return;
+ }
+
+requeue_work:
+ schedule_delayed_work(&sg_ctx.work, sg_ctx.delay);
+}
+
+static int __init steal_governor_init(void)
+{
+#ifdef CONFIG_XEN
+ if (xen_initial_domain()) {
+ pr_err("Cannot load in Xen Dom0 (Host OS). Driver is for guests only.\n");
+ return -ENODEV;
+ }
+#endif
+
+ if (sg_ctx.low_threshold >= sg_ctx.high_threshold) {
+ pr_err("low_threshold (%u) must be less than high_threshold (%u)\n",
+ sg_ctx.low_threshold, sg_ctx.high_threshold);
+ return -EINVAL;
+ }
+
+ sg_ctx.delay = msecs_to_jiffies(sg_ctx.interval_ms);
+ INIT_DELAYED_WORK(&sg_ctx.work, steal_governor_loop);
+ sg_ctx.steal = get_system_steal_time();
+ sg_ctx.time = ktime_get();
+ schedule_delayed_work(&sg_ctx.work, sg_ctx.delay);
+ pr_info("enabled. interval: %ums, high_threshold: %u, low_threshold: %u\n",
+ sg_ctx.interval_ms, sg_ctx.high_threshold, sg_ctx.low_threshold);
+
+ return 0;
+}
+
+static void __exit steal_governor_exit(void)
+{
+ disable_delayed_work_sync(&sg_ctx.work);
+ restore_preferred_to_active();
+ pr_info("disabled\n");
+}
+
+module_init(steal_governor_init);
+module_exit(steal_governor_exit);
+
+MODULE_LICENSE("GPL");
+MODULE_AUTHOR("IBM Corporation");
+MODULE_DESCRIPTION("Virtualization Steal Time Governor");
diff --git a/drivers/xen/time.c b/drivers/xen/time.c
index a2be0a4d45b0..a02d48a2aa68 100644
--- a/drivers/xen/time.c
+++ b/drivers/xen/time.c
@@ -169,7 +169,7 @@ void __init xen_time_setup_guest(void)
static_call_update(pv_steal_clock, xen_steal_clock);
- static_key_slow_inc(&paravirt_steal_enabled);
+ static_branch_inc(&paravirt_steal_enabled);
if (xen_runstate_remote)
- static_key_slow_inc(&paravirt_steal_rq_enabled);
+ static_branch_inc(&paravirt_steal_rq_enabled);
}
diff --git a/fs/aio.c b/fs/aio.c
index 22affd3e9dbc..4f32b45cb8a0 100644
--- a/fs/aio.c
+++ b/fs/aio.c
@@ -1402,7 +1402,7 @@ static long read_events(struct kioctx *ctx, long min_nr, long nr,
w.min_nr = min_nr - ret;
ret2 = prepare_to_wait_event(&ctx->wait, &w.w, TASK_INTERRUPTIBLE);
- if (!ret2 && !t.task)
+ if (!ret2 && !hrtimer_sleeper_task_get(&t))
ret2 = -ETIME;
if (aio_read_events(ctx, min_nr, nr, event, &ret) || ret2)
diff --git a/fs/exec.c b/fs/exec.c
index aba8902d9423..8d2b14e1d455 100644
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1116,17 +1116,6 @@ static struct file *bprm_identity_file(const struct linux_binprm *bprm)
return bprm->file;
}
-static void posixtimer_exec(struct task_struct *me)
-{
-#ifdef CONFIG_POSIX_TIMERS
- spin_lock_irq(&me->sighand->siglock);
- posix_cpu_timers_exit(me);
- spin_unlock_irq(&me->sighand->siglock);
- exit_itimers(me);
- flush_itimer_signals();
-#endif
-}
-
/*
* Calling this is the point of no return. None of the failures will be
* seen by userspace since either the process is already taking a fatal
@@ -1180,7 +1169,7 @@ int begin_new_exec(struct linux_binprm * bprm)
* timer would not remove an enqueued timer because the TID lookup
* of the old TID fails.
*/
- posixtimer_exec(me);
+ posixtimer_exec();
/* see the comment in check_unsafe_exec() */
current->fs->in_exec = 0;
diff --git a/fs/proc/uptime.c b/fs/proc/uptime.c
index 433aa947cd57..53143c66cbe1 100644
--- a/fs/proc/uptime.c
+++ b/fs/proc/uptime.c
@@ -15,12 +15,8 @@ static int uptime_proc_show(struct seq_file *m, void *v)
struct timespec64 idle;
u64 idle_nsec;
u32 rem;
- int i;
-
- idle_nsec = 0;
- for_each_possible_cpu(i)
- idle_nsec += kcpustat_field(CPUTIME_IDLE, i);
+ idle_nsec = kcpustat_field_total(CPUTIME_IDLE, cpu_possible_mask);
ktime_get_boottime_ts64(&uptime);
timens_add_boottime(&uptime);
diff --git a/fs/resctrl/ctrlmondata.c b/fs/resctrl/ctrlmondata.c
index 18ec9f564b5a..cafebdff70dc 100644
--- a/fs/resctrl/ctrlmondata.c
+++ b/fs/resctrl/ctrlmondata.c
@@ -37,8 +37,8 @@ typedef int (ctrlval_parser_t)(struct rdt_parse_data *data,
/*
* Check whether MBA bandwidth percentage value is correct. The value is
* checked against the minimum and max bandwidth values specified by the
- * hardware. The allocated bandwidth percentage is rounded to the next
- * control step available on the hardware.
+ * hardware. The allocated bandwidth percentage is converted as appropriate
+ * for consumption by the specific hardware driver.
*/
static bool bw_validate(char *buf, u32 *data, struct rdt_resource *r)
{
@@ -71,7 +71,7 @@ static bool bw_validate(char *buf, u32 *data, struct rdt_resource *r)
return false;
}
- *data = roundup(bw, (unsigned long)r->membw.bw_gran);
+ *data = resctrl_arch_preconvert_bw(r, bw);
return true;
}
diff --git a/fs/resctrl/pseudo_lock.c b/fs/resctrl/pseudo_lock.c
index dea2b4bf966f..56ab63f19bad 100644
--- a/fs/resctrl/pseudo_lock.c
+++ b/fs/resctrl/pseudo_lock.c
@@ -750,17 +750,10 @@ static ssize_t pseudo_lock_measure_trigger(struct file *file,
size_t count, loff_t *ppos)
{
struct rdtgroup *rdtgrp = file->private_data;
- size_t buf_size;
- char buf[32];
int ret;
int sel;
- buf_size = min(count, (sizeof(buf) - 1));
- if (copy_from_user(buf, user_buf, buf_size))
- return -EFAULT;
-
- buf[buf_size] = '\0';
- ret = kstrtoint(buf, 10, &sel);
+ ret = kstrtoint_from_user(user_buf, count, 10, &sel);
if (ret == 0) {
if (sel != 1 && sel != 2 && sel != 3)
return -EINVAL;
diff --git a/fs/resctrl/rdtgroup.c b/fs/resctrl/rdtgroup.c
index 5dcbb0a964e8..68be9b903ac6 100644
--- a/fs/resctrl/rdtgroup.c
+++ b/fs/resctrl/rdtgroup.c
@@ -2858,7 +2858,7 @@ static int schemata_list_add(struct rdt_resource *r, enum resctrl_conf_type type
{
struct resctrl_schema *s;
const char *suffix = "";
- int ret, cl;
+ int cl;
s = kzalloc_obj(*s);
if (!s)
@@ -2882,14 +2882,12 @@ static int schemata_list_add(struct rdt_resource *r, enum resctrl_conf_type type
break;
}
- ret = snprintf(s->name, sizeof(s->name), "%s%s", r->name, suffix);
- if (ret >= sizeof(s->name)) {
+ cl = snprintf(s->name, sizeof(s->name), "%s%s", r->name, suffix);
+ if (cl >= sizeof(s->name)) {
kfree(s);
return -EINVAL;
}
- cl = strlen(s->name);
-
/*
* If CDP is supported by this resource, but not enabled,
* include the suffix. This ensures the tabular format of the
diff --git a/include/asm-generic/preempt.h b/include/asm-generic/preempt.h
index c8683c046615..1adddeab8545 100644
--- a/include/asm-generic/preempt.h
+++ b/include/asm-generic/preempt.h
@@ -96,19 +96,9 @@ static __always_inline bool should_resched(int preempt_offset)
extern asmlinkage void preempt_schedule(void);
extern asmlinkage void preempt_schedule_notrace(void);
-#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-
-void dynamic_preempt_schedule(void);
-void dynamic_preempt_schedule_notrace(void);
-#define __preempt_schedule() dynamic_preempt_schedule()
-#define __preempt_schedule_notrace() dynamic_preempt_schedule_notrace()
-
-#else /* !CONFIG_PREEMPT_DYNAMIC || !CONFIG_HAVE_PREEMPT_DYNAMIC_KEY*/
-
#define __preempt_schedule() preempt_schedule()
#define __preempt_schedule_notrace() preempt_schedule_notrace()
-#endif /* CONFIG_PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_KEY*/
#endif /* CONFIG_PREEMPTION */
#endif /* __ASM_PREEMPT_H */
diff --git a/include/linux/bitmap.h b/include/linux/bitmap.h
index 7df1573a409c..adafbcf2016b 100644
--- a/include/linux/bitmap.h
+++ b/include/linux/bitmap.h
@@ -52,6 +52,7 @@ struct device;
* bitmap_complement(dst, src, nbits) *dst = ~(*src)
* bitmap_equal(src1, src2, nbits) Are *src1 and *src2 equal?
* bitmap_intersects(src1, src2, nbits) Do *src1 and *src2 overlap?
+ * bitmap_intersects_and(src1, src2, src3, nbits) Do *src1, *src2 and *src3 overlap?
* bitmap_subset(src1, src2, nbits) Is *src1 a subset of *src2?
* bitmap_empty(src, nbits) Are all bits zero in *src?
* bitmap_full(src, nbits) Are all bits set in *src?
@@ -181,6 +182,9 @@ void __bitmap_replace(unsigned long *dst,
const unsigned long *mask, unsigned int nbits);
bool __bitmap_intersects(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
+bool __bitmap_intersects_and(const unsigned long *bitmap1,
+ const unsigned long *bitmap2,
+ const unsigned long *bitmap3, unsigned int nbits);
bool __bitmap_subset(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
unsigned int __bitmap_weight(const unsigned long *bitmap, unsigned int nbits);
@@ -446,6 +450,16 @@ bool bitmap_intersects(const unsigned long *src1, const unsigned long *src2, uns
}
static __always_inline
+bool bitmap_intersects_and(const unsigned long *src1, const unsigned long *src2,
+ const unsigned long *src3, unsigned int nbits)
+{
+ if (small_const_nbits(nbits))
+ return ((*src1 & *src2 & *src3) & BITMAP_LAST_WORD_MASK(nbits)) != 0;
+ else
+ return __bitmap_intersects_and(src1, src2, src3, nbits);
+}
+
+static __always_inline
bool bitmap_subset(const unsigned long *src1, const unsigned long *src2, unsigned int nbits)
{
if (small_const_nbits(nbits))
diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h
index 4c8bb6953107..bf89bb3f30f6 100644
--- a/include/linux/cpumask.h
+++ b/include/linux/cpumask.h
@@ -121,12 +121,20 @@ extern struct cpumask __cpu_enabled_mask;
extern struct cpumask __cpu_present_mask;
extern struct cpumask __cpu_active_mask;
extern struct cpumask __cpu_dying_mask;
+
+#ifdef CONFIG_PREFERRED_CPU
+extern struct cpumask __cpu_preferred_mask;
+#else
+#define __cpu_preferred_mask __cpu_active_mask
+#endif
+
#define cpu_possible_mask ((const struct cpumask *)&__cpu_possible_mask)
#define cpu_online_mask ((const struct cpumask *)&__cpu_online_mask)
#define cpu_enabled_mask ((const struct cpumask *)&__cpu_enabled_mask)
#define cpu_present_mask ((const struct cpumask *)&__cpu_present_mask)
#define cpu_active_mask ((const struct cpumask *)&__cpu_active_mask)
#define cpu_dying_mask ((const struct cpumask *)&__cpu_dying_mask)
+#define cpu_preferred_mask ((const struct cpumask *)&__cpu_preferred_mask)
extern atomic_t __num_online_cpus;
extern unsigned int __num_possible_cpus;
@@ -825,6 +833,24 @@ bool cpumask_intersects(const struct cpumask *src1p, const struct cpumask *src2p
}
/**
+ * cpumask_intersects_and - (*src1p & *src2p & *src3p) != 0
+ * @src1p: the first input
+ * @src2p: the second input
+ * @src3p: the third input
+ *
+ * Return: true if AND of the three cpumasks is non-empty,
+ * otherwise false
+ */
+static __always_inline
+bool cpumask_intersects_and(const struct cpumask *src1p,
+ const struct cpumask *src2p,
+ const struct cpumask *src3p)
+{
+ return bitmap_intersects_and(cpumask_bits(src1p), cpumask_bits(src2p),
+ cpumask_bits(src3p), small_cpumask_bits);
+}
+
+/**
* cpumask_subset - (*src1p & ~*src2p) == 0
* @src1p: the first input
* @src2p: the second input
@@ -1163,6 +1189,12 @@ void init_cpu_possible(const struct cpumask *src);
#define set_cpu_active(cpu, active) assign_cpu((cpu), &__cpu_active_mask, (active))
#define set_cpu_dying(cpu, dying) assign_cpu((cpu), &__cpu_dying_mask, (dying))
+#ifdef CONFIG_PREFERRED_CPU
+#define set_cpu_preferred(cpu, preferred) assign_cpu((cpu), &__cpu_preferred_mask, (preferred))
+#else
+#define set_cpu_preferred(cpu, preferred) do { } while (0)
+#endif
+
void set_cpu_online(unsigned int cpu, bool online);
void set_cpu_possible(unsigned int cpu, bool possible);
@@ -1257,6 +1289,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
return cpumask_test_cpu(cpu, cpu_dying_mask);
}
+static __always_inline bool cpu_preferred(unsigned int cpu)
+{
+ return cpumask_test_cpu(cpu, cpu_preferred_mask);
+}
+
#else
#define num_online_cpus() 1U
@@ -1295,6 +1332,11 @@ static __always_inline bool cpu_dying(unsigned int cpu)
return false;
}
+static __always_inline bool cpu_preferred(unsigned int cpu)
+{
+ return cpu == 0;
+}
+
#endif /* NR_CPUS > 1 */
#define cpu_is_offline(cpu) unlikely(!cpu_online(cpu))
diff --git a/include/linux/hrtimer.h b/include/linux/hrtimer.h
index 29072d89e5cb..cad8482337cb 100644
--- a/include/linux/hrtimer.h
+++ b/include/linux/hrtimer.h
@@ -72,7 +72,7 @@ enum hrtimer_mode {
*/
struct hrtimer_sleeper {
struct hrtimer timer;
- struct task_struct *task;
+ struct task_struct *__private task;
};
static inline void hrtimer_set_expires(struct hrtimer *timer, ktime_t time)
@@ -320,6 +320,14 @@ extern int schedule_hrtimeout_range_clock(ktime_t *expires,
const enum hrtimer_mode mode,
clockid_t clock_id);
extern int schedule_hrtimeout(ktime_t *expires, const enum hrtimer_mode mode);
+static inline struct task_struct *hrtimer_sleeper_task_get(struct hrtimer_sleeper *sl)
+{
+ return READ_ONCE(ACCESS_PRIVATE(sl, task));
+}
+static inline void hrtimer_sleeper_task_set(struct hrtimer_sleeper *sl, struct task_struct *t)
+{
+ WRITE_ONCE(ACCESS_PRIVATE(sl, task), t);
+}
/* Soft interrupt function to run the hrtimer queues: */
extern void hrtimer_run_queues(void);
diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h
index 3bf969ad8fe0..52bb684090c0 100644
--- a/include/linux/interrupt.h
+++ b/include/linux/interrupt.h
@@ -573,9 +573,11 @@ enum
* _ IRQ_POLL: irq_poll_cpu_dead() migrates the queue
*
* _ (HR)TIMER_SOFTIRQ: (hr)timers_dead_cpu() migrates the queue
+ *
+ * _ BLOCK_SOFTIRQ: blk_softirq_cpu_dead() completes the remaining requests
*/
-#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(IRQ_POLL_SOFTIRQ) |\
- BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ))
+#define SOFTIRQ_HOTPLUG_SAFE_MASK (BIT(TIMER_SOFTIRQ) | BIT(BLOCK_SOFTIRQ) |\
+ BIT(IRQ_POLL_SOFTIRQ) | BIT(HRTIMER_SOFTIRQ) | BIT(RCU_SOFTIRQ))
/* map softirq index to softirq name. update 'softirq_to_name' in
diff --git a/include/linux/irq-entry-common.h b/include/linux/irq-entry-common.h
index 0bb6c03481fa..2be273bb68f0 100644
--- a/include/linux/irq-entry-common.h
+++ b/include/linux/irq-entry-common.h
@@ -346,22 +346,7 @@ typedef struct irqentry_state {
*
* Conditional reschedule with additional sanity checks.
*/
-void raw_irqentry_exit_cond_resched(void);
-
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-#define irqentry_exit_cond_resched_dynamic_enabled raw_irqentry_exit_cond_resched
-#define irqentry_exit_cond_resched_dynamic_disabled NULL
-DECLARE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#define irqentry_exit_cond_resched() static_call(irqentry_exit_cond_resched)()
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DECLARE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void);
-#define irqentry_exit_cond_resched() dynamic_irqentry_exit_cond_resched()
-#endif
-#else /* CONFIG_PREEMPT_DYNAMIC */
-#define irqentry_exit_cond_resched() raw_irqentry_exit_cond_resched()
-#endif /* CONFIG_PREEMPT_DYNAMIC */
+void irqentry_exit_cond_resched(void);
/**
* irqentry_enter_from_kernel_mode - Establish state before invoking the irq handler
diff --git a/include/linux/kernel.h b/include/linux/kernel.h
index 24414c79e59a..4b11d1dc0a67 100644
--- a/include/linux/kernel.h
+++ b/include/linux/kernel.h
@@ -43,30 +43,10 @@ struct completion;
struct user;
#ifdef CONFIG_PREEMPT_VOLUNTARY_BUILD
-
extern int __cond_resched(void);
# define might_resched() __cond_resched()
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-
-extern int __cond_resched(void);
-
-DECLARE_STATIC_CALL(might_resched, __cond_resched);
-
-static __always_inline void might_resched(void)
-{
- static_call_mod(might_resched)();
-}
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-
-extern int dynamic_might_resched(void);
-# define might_resched() dynamic_might_resched()
-
#else
-
# define might_resched() do { } while (0)
-
#endif /* CONFIG_PREEMPT_* */
#ifdef CONFIG_DEBUG_ATOMIC_SLEEP
diff --git a/include/linux/kernel_stat.h b/include/linux/kernel_stat.h
index 9ca6c2259dfe..c1e85550bf12 100644
--- a/include/linux/kernel_stat.h
+++ b/include/linux/kernel_stat.h
@@ -196,6 +196,17 @@ static inline void kcpustat_cpu_fetch(struct kernel_cpustat *dst, int cpu)
}
#endif /* !CONFIG_VIRT_CPU_ACCOUNTING_GEN */
+static inline u64 kcpustat_field_total(enum cpu_usage_stat usage, const struct cpumask *cpus)
+{
+ u64 total = 0;
+ int cpu;
+
+ for_each_cpu(cpu, cpus)
+ total += kcpustat_field(usage, cpu);
+
+ return total;
+}
+
extern void account_user_time(struct task_struct *, u64);
extern void account_guest_time(struct task_struct *, u64);
extern void account_system_time(struct task_struct *, int, u64);
diff --git a/include/linux/list.h b/include/linux/list.h
index 77fb62f79928..e3753695e76c 100644
--- a/include/linux/list.h
+++ b/include/linux/list.h
@@ -1172,7 +1172,7 @@ static inline void hlist_move_list(struct hlist_head *old,
{
new->first = old->first;
if (new->first)
- new->first->pprev = &new->first;
+ WRITE_ONCE(new->first->pprev, &new->first);
old->first = NULL;
}
@@ -1189,10 +1189,10 @@ static inline void hlist_splice_init(struct hlist_head *from,
struct hlist_head *to)
{
if (to->first)
- to->first->pprev = &last->next;
+ WRITE_ONCE(to->first->pprev, &last->next);
last->next = to->first;
to->first = from->first;
- from->first->pprev = &to->first;
+ WRITE_ONCE(from->first->pprev, &to->first);
from->first = NULL;
}
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
index 915c6fd3f084..7797ce207555 100644
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -306,6 +306,7 @@ struct perf_event_pmu_context;
#define PERF_PMU_CAP_AUX_PAUSE 0x0200
#define PERF_PMU_CAP_AUX_PREFER_LARGE 0x0400
#define PERF_PMU_CAP_MEDIATED_VPMU 0x0800
+#define PERF_PMU_CAP_SIMD_REGS 0x1000
/**
* pmu::scope
@@ -1467,6 +1468,7 @@ static inline u32 perf_sample_data_size(struct perf_sample_data *data,
return size;
}
+extern u64 perf_update_xregs_size(struct perf_event *event, bool intr);
extern void perf_output_sample(struct perf_output_handle *handle,
struct perf_event_header *header,
struct perf_sample_data *data,
@@ -1517,6 +1519,27 @@ perf_event__output_id_sample(struct perf_event *event,
extern void
perf_log_lost_samples(struct perf_event *event, u64 lost);
+static inline bool event_has_simd_regs(struct perf_event *event)
+{
+ struct perf_event_attr *attr = &event->attr;
+
+ if (!(event->attr.sample_type &
+ (PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER)))
+ return false;
+
+ return attr->sample_simd_regs_enabled != 0;
+}
+
+static inline bool event_has_extended_regs(struct perf_event *event)
+{
+ struct perf_event_attr *attr = &event->attr;
+
+ return ((attr->sample_type & PERF_SAMPLE_REGS_USER) &&
+ (attr->sample_regs_user & PERF_REG_EXTENDED_MASK)) ||
+ ((attr->sample_type & PERF_SAMPLE_REGS_INTR) &&
+ (attr->sample_regs_intr & PERF_REG_EXTENDED_MASK));
+}
+
static inline bool event_has_any_exclude_flag(struct perf_event *event)
{
struct perf_event_attr *attr = &event->attr;
diff --git a/include/linux/perf_regs.h b/include/linux/perf_regs.h
index f632c5725f16..09dbc2fc3859 100644
--- a/include/linux/perf_regs.h
+++ b/include/linux/perf_regs.h
@@ -9,6 +9,16 @@ struct perf_regs {
struct pt_regs *regs;
};
+u64 perf_reg_value(struct pt_regs *regs, int idx);
+int perf_reg_validate(u64 mask, bool simd_enabled);
+u64 perf_reg_abi(struct task_struct *task);
+void perf_get_regs_user(struct perf_regs *regs_user,
+ struct pt_regs *regs);
+int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
+ u16 pred_qwords, u32 pred_mask);
+u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
+ u16 qwords_idx, bool pred);
+
#ifdef CONFIG_HAVE_PERF_REGS
#include <asm/perf_regs.h>
@@ -16,35 +26,9 @@ struct perf_regs {
#define PERF_REG_EXTENDED_MASK 0
#endif
-u64 perf_reg_value(struct pt_regs *regs, int idx);
-int perf_reg_validate(u64 mask);
-u64 perf_reg_abi(struct task_struct *task);
-void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs);
#else
#define PERF_REG_EXTENDED_MASK 0
-static inline u64 perf_reg_value(struct pt_regs *regs, int idx)
-{
- return 0;
-}
-
-static inline int perf_reg_validate(u64 mask)
-{
- return mask ? -ENOSYS : 0;
-}
-
-static inline u64 perf_reg_abi(struct task_struct *task)
-{
- return PERF_SAMPLE_REGS_ABI_NONE;
-}
-
-static inline void perf_get_regs_user(struct perf_regs *regs_user,
- struct pt_regs *regs)
-{
- regs_user->regs = task_pt_regs(current);
- regs_user->abi = perf_reg_abi(current);
-}
#endif /* CONFIG_HAVE_PERF_REGS */
#endif /* _LINUX_PERF_REGS_H */
diff --git a/include/linux/posix-timers.h b/include/linux/posix-timers.h
index 9a1a0c61361c..00767acbc111 100644
--- a/include/linux/posix-timers.h
+++ b/include/linux/posix-timers.h
@@ -66,38 +66,6 @@ struct cpu_timer {
struct task_struct __rcu *handling;
};
-static inline bool cpu_timer_enqueue(struct timerqueue_head *head,
- struct cpu_timer *ctmr)
-{
- ctmr->head = head;
- return timerqueue_add(head, &ctmr->node);
-}
-
-static inline bool cpu_timer_queued(struct cpu_timer *ctmr)
-{
- return !!ctmr->head;
-}
-
-static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr)
-{
- if (cpu_timer_queued(ctmr)) {
- timerqueue_del(ctmr->head, &ctmr->node);
- ctmr->head = NULL;
- return true;
- }
- return false;
-}
-
-static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr)
-{
- return ctmr->node.expires;
-}
-
-static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp)
-{
- ctmr->node.expires = exp;
-}
-
static inline void posix_cputimers_init(struct posix_cputimers *pct)
{
memset(pct, 0, sizeof(*pct));
@@ -224,14 +192,15 @@ struct k_itimer {
} ____cacheline_aligned_in_smp;
void run_posix_cpu_timers(void);
-void posix_cpu_timers_exit(struct task_struct *task);
-void posix_cpu_timers_exit_group(struct task_struct *task);
void set_process_cpu_timer(struct task_struct *task, unsigned int clock_idx,
u64 *newval, u64 *oldval);
int update_rlimit_cpu(struct task_struct *task, unsigned long rlim_new);
#ifdef CONFIG_POSIX_TIMERS
+void posixtimer_exec(void);
+void posixtimer_exit(bool group_dead);
+
static inline void posixtimer_putref(struct k_itimer *tmr)
{
if (rcuref_put(&tmr->rcuref))
@@ -259,6 +228,8 @@ static inline bool posixtimer_valid(const struct k_itimer *timer)
return !(val & 0x1UL);
}
#else /* CONFIG_POSIX_TIMERS */
+static inline void posixtimer_exec(void) { }
+static inline void posixtimer_exit(bool group_dead) { }
static inline void posixtimer_sigqueue_getref(struct sigqueue *q) { }
static inline void posixtimer_sigqueue_putref(struct sigqueue *q) { }
#endif /* !CONFIG_POSIX_TIMERS */
diff --git a/include/linux/preempt.h b/include/linux/preempt.h
index 2e689de7b29a..06ff44a4b8b6 100644
--- a/include/linux/preempt.h
+++ b/include/linux/preempt.h
@@ -498,21 +498,11 @@ DEFINE_LOCK_GUARD_0(preempt_notrace, preempt_disable_notrace(), preempt_enable_n
#ifdef CONFIG_PREEMPT_DYNAMIC
-extern bool preempt_model_none(void);
-extern bool preempt_model_voluntary(void);
extern bool preempt_model_full(void);
extern bool preempt_model_lazy(void);
#else
-static inline bool preempt_model_none(void)
-{
- return IS_ENABLED(CONFIG_PREEMPT_NONE);
-}
-static inline bool preempt_model_voluntary(void)
-{
- return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY);
-}
static inline bool preempt_model_full(void)
{
return IS_ENABLED(CONFIG_PREEMPT);
@@ -525,6 +515,16 @@ static inline bool preempt_model_lazy(void)
#endif
+static inline bool preempt_model_none(void)
+{
+ return IS_ENABLED(CONFIG_PREEMPT_NONE);
+}
+
+static inline bool preempt_model_voluntary(void)
+{
+ return IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY);
+}
+
static inline bool preempt_model_rt(void)
{
return IS_ENABLED(CONFIG_PREEMPT_RT);
diff --git a/include/linux/resctrl.h b/include/linux/resctrl.h
index dd09c2ce9a0f..10dfdca7f4bf 100644
--- a/include/linux/resctrl.h
+++ b/include/linux/resctrl.h
@@ -505,6 +505,25 @@ bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r);
*/
int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable);
+/**
+ * resctrl_arch_preconvert_bw() - Prepare bandwidth control value for arch use.
+ * @r: Resource whose schema was written.
+ * @val: Bandwidth control value written to the schemata file by userspace.
+ *
+ * Convert the user provided bandwidth control value to an appropriate form for
+ * consumption by the hardware driver for resource @r. Converted value is stored
+ * in rdt_ctrl_domain::staged_config[] for later consumption by
+ * resctrl_arch_update_domains(). Is not called when MBA software controller is
+ * enabled.
+ *
+ * Architectures for which this pre-conversion hook is not useful should supply
+ * an implementation of this function that just returns @val unmodified.
+ *
+ * Return:
+ * The converted value.
+ */
+u32 resctrl_arch_preconvert_bw(const struct rdt_resource *r, u32 val);
+
/*
* Update the ctrl_val and apply this config right now.
* Must be called on one of the domain's CPUs.
diff --git a/include/linux/sched.h b/include/linux/sched.h
index 64b7581e2012..d828f0fb1896 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -554,6 +554,7 @@ struct sched_statistics {
u64 nr_failed_migrations_running;
u64 nr_failed_migrations_hot;
u64 nr_forced_migrations;
+ u64 nr_migrations_cpu_non_preferred;
u64 nr_wakeups;
u64 nr_wakeups_sync;
@@ -2139,44 +2140,19 @@ static inline void set_need_resched_current(void)
* value indicates whether a reschedule was done in fact.
* cond_resched_lock() will drop the spinlock before scheduling,
*/
-#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
+#if !defined(CONFIG_PREEMPTION)
extern int __cond_resched(void);
-#if defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-
-DECLARE_STATIC_CALL(cond_resched, __cond_resched);
-
-static __always_inline int _cond_resched(void)
-{
- return static_call_mod(cond_resched)();
-}
-
-#elif defined(CONFIG_PREEMPT_DYNAMIC) && defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-
-extern int dynamic_cond_resched(void);
-
-static __always_inline int _cond_resched(void)
-{
- return dynamic_cond_resched();
-}
-
-#else /* !CONFIG_PREEMPTION */
-
static inline int _cond_resched(void)
{
return __cond_resched();
}
-
-#endif /* PREEMPT_DYNAMIC && CONFIG_HAVE_PREEMPT_DYNAMIC_CALL */
-
-#else /* CONFIG_PREEMPTION && !CONFIG_PREEMPT_DYNAMIC */
-
+#else
static inline int _cond_resched(void)
{
return 0;
}
-
-#endif /* !CONFIG_PREEMPTION || CONFIG_PREEMPT_DYNAMIC */
+#endif
#define cond_resched() ({ \
__might_resched(__FILE__, __LINE__, 0); \
diff --git a/include/linux/sched/cputime.h b/include/linux/sched/cputime.h
index e90efaf6d26e..694126411dfe 100644
--- a/include/linux/sched/cputime.h
+++ b/include/linux/sched/cputime.h
@@ -182,9 +182,9 @@ extern unsigned long long
task_sched_runtime(struct task_struct *task);
#ifdef CONFIG_PARAVIRT
-struct static_key;
-extern struct static_key paravirt_steal_enabled;
-extern struct static_key paravirt_steal_rq_enabled;
+#include <linux/jump_label.h>
+DECLARE_STATIC_KEY_FALSE(paravirt_steal_enabled);
+DECLARE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled);
#ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN
u64 dummy_steal_clock(int cpu);
diff --git a/include/linux/sched/task.h b/include/linux/sched/task.h
index e0c1ca8c6a18..90ed5bc3c7af 100644
--- a/include/linux/sched/task.h
+++ b/include/linux/sched/task.h
@@ -94,7 +94,6 @@ static inline void exit_thread(struct task_struct *tsk)
extern __noreturn void do_group_exit(int);
extern void exit_files(struct task_struct *);
-extern void exit_itimers(struct task_struct *);
extern pid_t kernel_clone(struct kernel_clone_args *kargs);
struct task_struct *copy_process(struct pid *pid, int trace, int node,
diff --git a/include/linux/wait.h b/include/linux/wait.h
index 7e215330199c..c2af98b0074d 100644
--- a/include/linux/wait.h
+++ b/include/linux/wait.h
@@ -556,7 +556,7 @@ do { \
} \
\
__ret = ___wait_event(wq_head, condition, state, 0, 0, \
- if (!__t.task) { \
+ if (!hrtimer_sleeper_task_get(&__t)) { \
__ret = -ETIME; \
break; \
} \
diff --git a/include/uapi/linux/perf_event.h b/include/uapi/linux/perf_event.h
index fd10aa8d697f..c49fc76292f7 100644
--- a/include/uapi/linux/perf_event.h
+++ b/include/uapi/linux/perf_event.h
@@ -314,8 +314,9 @@ enum {
*/
enum perf_sample_regs_abi {
PERF_SAMPLE_REGS_ABI_NONE = 0,
- PERF_SAMPLE_REGS_ABI_32 = 1,
- PERF_SAMPLE_REGS_ABI_64 = 2,
+ PERF_SAMPLE_REGS_ABI_32 = (1 << 0),
+ PERF_SAMPLE_REGS_ABI_64 = (1 << 1),
+ PERF_SAMPLE_REGS_ABI_SIMD = (1 << 2),
};
/*
@@ -383,6 +384,7 @@ enum perf_event_read_format {
#define PERF_ATTR_SIZE_VER7 128 /* Add: sig_data */
#define PERF_ATTR_SIZE_VER8 136 /* Add: config3 */
#define PERF_ATTR_SIZE_VER9 144 /* add: config4 */
+#define PERF_ATTR_SIZE_VER10 176 /* Add: sample_simd_{vec|pred}_reg_* */
/*
* 'struct perf_event_attr' contains various attributes that define
@@ -547,6 +549,29 @@ struct perf_event_attr {
__u64 config3; /* extension of config2 */
__u64 config4; /* extension of config3 */
+
+ /*
+ * Defines the sampling SIMD/PRED(predicate) registers bitmap and
+ * qwords (8 bytes) length.
+ *
+ * sample_simd_regs_enabled != 0 indicates there are SIMD/PRED
+ * registers to be sampled, the SIMD/PRED registers bitmap and
+ * qwords length are represented in
+ * sample_simd_{vec|pred}_reg_{intr|user} and
+ * sample_simd_{vec|pred}_reg_qwords fields separately.
+ *
+ * sample_simd_regs_enabled == 0 indicates no SIMD/PRED registers
+ * are sampled.
+ */
+ __u16 sample_simd_regs_enabled;
+ __u16 sample_simd_pred_reg_qwords;
+ __u16 sample_simd_vec_reg_qwords;
+ __u16 __reserved_4;
+
+ __u32 sample_simd_pred_reg_intr;
+ __u32 sample_simd_pred_reg_user;
+ __u64 sample_simd_vec_reg_intr;
+ __u64 sample_simd_vec_reg_user;
};
/*
@@ -1020,7 +1045,15 @@ enum perf_event_type {
* } && PERF_SAMPLE_BRANCH_STACK
*
* { u64 abi; # enum perf_sample_regs_abi
- * u64 regs[weight(mask)]; } && PERF_SAMPLE_REGS_USER
+ * u64 regs[weight(mask)];
+ * struct {
+ * u64 nr_vectors; # 0 ... weight(sample_simd_vec_reg_user)
+ * u64 vector_qwords; # 0 ... sample_simd_vec_reg_qwords
+ * u64 nr_pred; # 0 ... weight(sample_simd_pred_reg_user)
+ * u64 pred_qwords; # 0 ... sample_simd_pred_reg_qwords
+ * u64 data[nr_vectors * vector_qwords + nr_pred * pred_qwords];
+ * } && (abi & PERF_SAMPLE_REGS_ABI_SIMD)
+ * } && PERF_SAMPLE_REGS_USER
*
* { u64 size;
* char data[size];
@@ -1047,7 +1080,15 @@ enum perf_event_type {
* { u64 data_src; } && PERF_SAMPLE_DATA_SRC
* { u64 transaction; } && PERF_SAMPLE_TRANSACTION
* { u64 abi; # enum perf_sample_regs_abi
- * u64 regs[weight(mask)]; } && PERF_SAMPLE_REGS_INTR
+ * u64 regs[weight(mask)];
+ * struct {
+ * u64 nr_vectors; # 0 ... weight(sample_simd_vec_reg_intr)
+ * u64 vector_qwords; # 0 ... sample_simd_vec_reg_qwords
+ * u64 nr_pred; # 0 ... weight(sample_simd_pred_reg_intr)
+ * u64 pred_qwords; # 0 ... sample_simd_pred_reg_qwords
+ * u64 data[nr_vectors * vector_qwords + nr_pred * pred_qwords];
+ * } && (abi & PERF_SAMPLE_REGS_ABI_SIMD)
+ * } && PERF_SAMPLE_REGS_INTR
* { u64 phys_addr;} && PERF_SAMPLE_PHYS_ADDR
* { u64 cgroup;} && PERF_SAMPLE_CGROUP
* { u64 data_page_size;} && PERF_SAMPLE_DATA_PAGE_SIZE
diff --git a/include/uapi/linux/sched.h b/include/uapi/linux/sched.h
index 33a4624285cd..19ffeba89428 100644
--- a/include/uapi/linux/sched.h
+++ b/include/uapi/linux/sched.h
@@ -53,7 +53,7 @@
*/
#define UNSHARE_EMPTY_MNTNS 0x00100000 /* Unshare an empty mount namespace. */
-#ifndef __ASSEMBLY__
+#ifndef __ASSEMBLER__
/**
* struct clone_args - arguments for the clone3 syscall
* @flags: Flags for the new process as listed above.
diff --git a/include/vdso/math64.h b/include/vdso/math64.h
index 22ae212f8b28..55b45f5cf615 100644
--- a/include/vdso/math64.h
+++ b/include/vdso/math64.h
@@ -2,15 +2,36 @@
#ifndef __VDSO_MATH64_H
#define __VDSO_MATH64_H
-static __always_inline u32
-__iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder)
+static __always_inline u32 __iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder)
{
u32 ret = 0;
while (dividend >= divisor) {
- /* The following asm() prevents the compiler from
- optimising this loop into a modulo operation. */
- asm("" : "+rm"(dividend));
+ /*
+ * Prevent the compiler from optimising this loop into a
+ * modulo operation.
+ */
+ OPTIMIZER_HIDE_VAR(dividend);
+
+ dividend -= divisor;
+ ret++;
+ }
+
+ *remainder = dividend;
+
+ return ret;
+}
+
+static __always_inline u32 __iter_div64_u64_rem(u64 dividend, u64 divisor, u64 *remainder)
+{
+ u32 ret = 0;
+
+ while (dividend >= divisor) {
+ /*
+ * Prevent the compiler from optimising this loop into a
+ * modulo operation.
+ */
+ OPTIMIZER_HIDE_VAR(dividend);
dividend -= divisor;
ret++;
diff --git a/io_uring/rw.c b/io_uring/rw.c
index 0c9494fd21be..755166e90746 100644
--- a/io_uring/rw.c
+++ b/io_uring/rw.c
@@ -1296,7 +1296,7 @@ static u64 io_hybrid_iopoll_delay(struct io_ring_ctx *ctx, struct io_kiocb *req)
set_current_state(TASK_INTERRUPTIBLE);
hrtimer_sleeper_start_expires(&timer, mode);
- if (timer.task)
+ if (hrtimer_sleeper_task_get(&timer))
io_schedule();
hrtimer_cancel(&timer.timer);
diff --git a/kernel/Kconfig.kexec b/kernel/Kconfig.kexec
index 15632358bcf7..a97ed9605602 100644
--- a/kernel/Kconfig.kexec
+++ b/kernel/Kconfig.kexec
@@ -167,7 +167,7 @@ config CRASH_MAX_MEMORY_RANGES
memory regions that the elfcorehdr buffer/segment can accommodate.
These regions are obtained via walk_system_ram_res(); eg. the
'System RAM' entries in /proc/iomem.
- This value is combined with NR_CPUS_DEFAULT and multiplied by
+ This value is combined with NR_CPUS and multiplied by
sizeof(Elf64_Phdr) to determine the final elfcorehdr memory buffer/
segment size.
The value 8192, for example, covers a (sparsely populated) 1TiB system
diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt
index f294dad43bd7..edc067a0c422 100644
--- a/kernel/Kconfig.preempt
+++ b/kernel/Kconfig.preempt
@@ -132,10 +132,9 @@ config PREEMPTION
config PREEMPT_DYNAMIC
bool "Preemption behaviour defined on boot"
- depends on HAVE_PREEMPT_DYNAMIC
- select JUMP_LABEL if HAVE_PREEMPT_DYNAMIC_KEY
+ depends on ARCH_HAS_PREEMPT_LAZY
select PREEMPT_BUILD
- default y if HAVE_PREEMPT_DYNAMIC_CALL
+ default y
help
This option allows to define the preemption model on the kernel
command line parameter and thus override the default preemption
@@ -145,9 +144,7 @@ config PREEMPT_DYNAMIC
provide a pre-built kernel binary to reduce the number of kernel
flavors they offer while still offering different usecases.
- The runtime overhead is negligible with HAVE_STATIC_CALL_INLINE enabled
- but if runtime patching is not available for the specific architecture
- then the potential overhead should be considered.
+ The runtime overhead is negligible.
Interesting if you want the same pre-built kernel should be used for
both Server and Desktop workloads.
@@ -197,3 +194,7 @@ config SCHED_CLASS_EXT
For more information:
Documentation/scheduler/sched-ext.rst
https://github.com/sched-ext/scx
+
+config PREFERRED_CPU
+ bool
+ depends on SMP && PARAVIRT
diff --git a/kernel/cpu.c b/kernel/cpu.c
index b3c8553d7bd6..376d297a6292 100644
--- a/kernel/cpu.c
+++ b/kernel/cpu.c
@@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask);
atomic_t __num_online_cpus __read_mostly;
EXPORT_SYMBOL(__num_online_cpus);
+#ifdef CONFIG_PREFERRED_CPU
+struct cpumask __cpu_preferred_mask __read_mostly;
+EXPORT_SYMBOL_GPL(__cpu_preferred_mask);
+#endif
+
void init_cpu_present(const struct cpumask *src)
{
cpumask_copy(&__cpu_present_mask, src);
@@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void)
/* Mark the boot cpu "present", "online" etc for SMP and UP case */
set_cpu_online(cpu, true);
set_cpu_active(cpu, true);
+ set_cpu_preferred(cpu, true);
set_cpu_present(cpu, true);
set_cpu_possible(cpu, true);
diff --git a/kernel/crash_core.c b/kernel/crash_core.c
index 2b36aa9fade0..d0bd2d0cf899 100644
--- a/kernel/crash_core.c
+++ b/kernel/crash_core.c
@@ -648,7 +648,7 @@ int crash_check_hotplug_support(void)
* new list of CPUs and memory. To make changes to the elfcorehdr, it
* should be large enough to permit a growing number of CPU and Memory
* resources. One can estimate the elfcorehdr memory size based on
- * NR_CPUS_DEFAULT and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is
+ * NR_CPUS and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is
* excluded from SHA verification by default if the architecture
* supports crash hotplug.
*/
diff --git a/kernel/entry/common.c b/kernel/entry/common.c
index e3d381fd3d25..e234b04373fe 100644
--- a/kernel/entry/common.c
+++ b/kernel/entry/common.c
@@ -123,7 +123,7 @@ noinstr irqentry_state_t irqentry_enter(struct pt_regs *regs)
/**
* arch_irqentry_exit_need_resched - Architecture specific need resched function
*
- * Invoked from raw_irqentry_exit_cond_resched() to check if resched is needed.
+ * Invoked from irqentry_exit_cond_resched() to check if resched is needed.
* Defaults return true.
*
* The main purpose is to permit arch to avoid preemption of a task from an IRQ.
@@ -134,7 +134,7 @@ static inline bool arch_irqentry_exit_need_resched(void);
static inline bool arch_irqentry_exit_need_resched(void) { return true; }
#endif
-void raw_irqentry_exit_cond_resched(void)
+void irqentry_exit_cond_resched(void)
{
if (!preempt_count()) {
/* Sanity check RCU and thread stack */
@@ -145,19 +145,6 @@ void raw_irqentry_exit_cond_resched(void)
preempt_schedule_irq();
}
}
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-DEFINE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DEFINE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_irqentry_exit_cond_resched))
- return;
- raw_irqentry_exit_cond_resched();
-}
-#endif
-#endif
noinstr void irqentry_exit(struct pt_regs *regs, irqentry_state_t state)
{
diff --git a/kernel/events/core.c b/kernel/events/core.c
index 601e8d944c24..a34ff4cb410d 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -7828,22 +7828,82 @@ unsigned long perf_instruction_pointer(struct perf_event *event,
0 : perf_arch_instruction_pointer(regs);
}
+u64 __weak perf_reg_value(struct pt_regs *regs, int idx)
+{
+ return 0;
+}
+
+int __weak perf_reg_validate(u64 mask, bool simd_enabled)
+{
+ return mask ? -ENOSYS : 0;
+}
+
+u64 __weak perf_reg_abi(struct task_struct *task)
+{
+ return PERF_SAMPLE_REGS_ABI_NONE;
+}
+
+void __weak perf_get_regs_user(struct perf_regs *regs_user,
+ struct pt_regs *regs)
+{
+ regs_user->regs = task_pt_regs(current);
+ regs_user->abi = perf_reg_abi(current);
+}
+
+#define word_for_each_set_bit(bit, val) \
+ for (unsigned long long __v = (val); \
+ __v && ((bit = __builtin_ctzll(__v)), 1); \
+ __v &= __v - 1)
+
static void
perf_output_sample_regs(struct perf_output_handle *handle,
struct pt_regs *regs, u64 mask)
{
int bit;
- DECLARE_BITMAP(_mask, 64);
- bitmap_from_u64(_mask, mask);
- for_each_set_bit(bit, _mask, sizeof(mask) * BITS_PER_BYTE) {
- u64 val;
-
- val = perf_reg_value(regs, bit);
+ word_for_each_set_bit(bit, mask) {
+ u64 val = perf_reg_value(regs, bit);
perf_output_put(handle, val);
}
}
+static void
+perf_output_sample_simd_regs(struct perf_output_handle *handle,
+ struct perf_event *event,
+ struct pt_regs *regs,
+ u64 mask, u32 pred_mask)
+{
+ u64 pred_qwords = event->attr.sample_simd_pred_reg_qwords;
+ u64 vec_qwords = event->attr.sample_simd_vec_reg_qwords;
+ u64 nr_vectors = hweight64(mask);
+ u64 nr_pred = hweight32(pred_mask);
+ int bit;
+
+ perf_output_put(handle, nr_vectors);
+ perf_output_put(handle, vec_qwords);
+ perf_output_put(handle, nr_pred);
+ perf_output_put(handle, pred_qwords);
+
+ if (nr_vectors) {
+ word_for_each_set_bit(bit, mask) {
+ for (int i = 0; i < vec_qwords; i++) {
+ u64 val = perf_simd_reg_value(regs, bit,
+ i, false);
+ perf_output_put(handle, val);
+ }
+ }
+ }
+ if (nr_pred) {
+ word_for_each_set_bit(bit, pred_mask) {
+ for (int i = 0; i < pred_qwords; i++) {
+ u64 val = perf_simd_reg_value(regs, bit,
+ i, true);
+ perf_output_put(handle, val);
+ }
+ }
+ }
+}
+
static void perf_sample_regs_user(struct perf_regs *regs_user,
struct pt_regs *regs)
{
@@ -7877,6 +7937,17 @@ static void perf_sample_regs_intr(struct perf_regs *regs_intr,
}
}
+int __weak perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
+ u16 pred_qwords, u32 pred_mask)
+{
+ return -EINVAL;
+}
+
+u64 __weak perf_simd_reg_value(struct pt_regs *regs, int idx,
+ u16 qwords_idx, bool pred)
+{
+ return 0;
+}
/*
* Get remaining task size from user stack pointer.
@@ -8407,10 +8478,17 @@ void perf_output_sample(struct perf_output_handle *handle,
perf_output_put(handle, abi);
if (abi) {
- u64 mask = event->attr.sample_regs_user;
+ struct perf_event_attr *attr = &event->attr;
+ u64 mask = attr->sample_regs_user;
perf_output_sample_regs(handle,
data->regs_user.regs,
mask);
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ perf_output_sample_simd_regs(handle, event,
+ data->regs_user.regs,
+ attr->sample_simd_vec_reg_user,
+ attr->sample_simd_pred_reg_user);
+ }
}
}
@@ -8438,11 +8516,18 @@ void perf_output_sample(struct perf_output_handle *handle,
perf_output_put(handle, abi);
if (abi) {
- u64 mask = event->attr.sample_regs_intr;
+ struct perf_event_attr *attr = &event->attr;
+ u64 mask = attr->sample_regs_intr;
perf_output_sample_regs(handle,
data->regs_intr.regs,
mask);
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ perf_output_sample_simd_regs(handle, event,
+ data->regs_intr.regs,
+ attr->sample_simd_vec_reg_intr,
+ attr->sample_simd_pred_reg_intr);
+ }
}
}
@@ -8645,6 +8730,29 @@ static __always_inline u64 __cond_set(u64 flags, u64 s, u64 d)
return d * !!(flags & s);
}
+u64 perf_update_xregs_size(struct perf_event *event, bool intr)
+{
+ u16 pred_qwords = event->attr.sample_simd_pred_reg_qwords;
+ u16 vec_qwords = event->attr.sample_simd_vec_reg_qwords;
+ u64 pred_mask;
+ u64 mask;
+ int size;
+
+ if (intr) {
+ mask = event->attr.sample_simd_vec_reg_intr;
+ pred_mask = event->attr.sample_simd_pred_reg_intr;
+ } else {
+ mask = event->attr.sample_simd_vec_reg_user;
+ pred_mask = event->attr.sample_simd_pred_reg_user;
+ }
+
+ size = sizeof(u64) * 4;
+ size += (hweight64(mask) * vec_qwords +
+ hweight64(pred_mask) * pred_qwords) * sizeof(u64);
+
+ return size;
+}
+
void perf_prepare_sample(struct perf_sample_data *data,
struct perf_event *event,
struct pt_regs *regs)
@@ -8707,7 +8815,12 @@ void perf_prepare_sample(struct perf_sample_data *data,
if (data->regs_user.regs) {
u64 mask = event->attr.sample_regs_user;
+
size += hweight64(mask) * sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ size += perf_update_xregs_size(event, false);
+ data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
}
data->dyn_size += size;
@@ -8772,6 +8885,10 @@ void perf_prepare_sample(struct perf_sample_data *data,
u64 mask = event->attr.sample_regs_intr;
size += hweight64(mask) * sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ size += perf_update_xregs_size(event, true);
+ data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
}
data->dyn_size += size;
@@ -13116,12 +13233,6 @@ int perf_pmu_unregister(struct pmu *pmu)
}
EXPORT_SYMBOL_GPL(perf_pmu_unregister);
-static inline bool has_extended_regs(struct perf_event *event)
-{
- return (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK) ||
- (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK);
-}
-
static int perf_try_init_event(struct pmu *pmu, struct perf_event *event)
{
struct perf_event_context *ctx = NULL;
@@ -13155,8 +13266,14 @@ static int perf_try_init_event(struct pmu *pmu, struct perf_event *event)
if (ret)
goto err_pmu;
+ if (!(pmu->capabilities & PERF_PMU_CAP_SIMD_REGS) &&
+ event_has_simd_regs(event)) {
+ ret = -EOPNOTSUPP;
+ goto err_destroy;
+ }
+
if (!(pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS) &&
- has_extended_regs(event)) {
+ event_has_extended_regs(event)) {
ret = -EOPNOTSUPP;
goto err_destroy;
}
@@ -13650,7 +13767,8 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
attr->size = size;
- if (attr->__reserved_1 || attr->__reserved_2 || attr->__reserved_3)
+ if (attr->__reserved_1 || attr->__reserved_2 ||
+ attr->__reserved_3 || attr->__reserved_4)
return -EINVAL;
if (attr->sample_type & ~(PERF_SAMPLE_MAX-1))
@@ -13696,9 +13814,18 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
}
if (attr->sample_type & PERF_SAMPLE_REGS_USER) {
- ret = perf_reg_validate(attr->sample_regs_user);
+ ret = perf_reg_validate(attr->sample_regs_user,
+ attr->sample_simd_regs_enabled);
if (ret)
return ret;
+ if (attr->sample_simd_regs_enabled) {
+ ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords,
+ attr->sample_simd_vec_reg_user,
+ attr->sample_simd_pred_reg_qwords,
+ attr->sample_simd_pred_reg_user);
+ if (ret)
+ return ret;
+ }
}
if (attr->sample_type & PERF_SAMPLE_STACK_USER) {
@@ -13719,8 +13846,20 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
if (!attr->sample_max_stack)
attr->sample_max_stack = sysctl_perf_event_max_stack;
- if (attr->sample_type & PERF_SAMPLE_REGS_INTR)
- ret = perf_reg_validate(attr->sample_regs_intr);
+ if (attr->sample_type & PERF_SAMPLE_REGS_INTR) {
+ ret = perf_reg_validate(attr->sample_regs_intr,
+ attr->sample_simd_regs_enabled);
+ if (ret)
+ return ret;
+ if (attr->sample_simd_regs_enabled) {
+ ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords,
+ attr->sample_simd_vec_reg_intr,
+ attr->sample_simd_pred_reg_qwords,
+ attr->sample_simd_pred_reg_intr);
+ if (ret)
+ return ret;
+ }
+ }
#ifndef CONFIG_CGROUP_PERF
if (attr->sample_type & PERF_SAMPLE_CGROUP)
diff --git a/kernel/exit.c b/kernel/exit.c
index 29e853a36602..9ff1fa7b30ea 100644
--- a/kernel/exit.c
+++ b/kernel/exit.c
@@ -168,12 +168,6 @@ static void __exit_signal(struct release_task_post *post, struct task_struct *ts
lockdep_tasklist_lock_is_held());
spin_lock(&sighand->siglock);
-#ifdef CONFIG_POSIX_TIMERS
- posix_cpu_timers_exit(tsk);
- if (group_dead)
- posix_cpu_timers_exit_group(tsk);
-#endif
-
if (group_dead) {
tty = sig->tty;
sig->tty = NULL;
@@ -940,13 +934,12 @@ void __noreturn do_exit(long code)
panic("Attempted to kill init! exitcode=0x%08x\n",
tsk->signal->group_exit_code ?: (int)code);
-#ifdef CONFIG_POSIX_TIMERS
- hrtimer_cancel(&tsk->signal->real_timer);
- exit_itimers(tsk);
-#endif
if (tsk->mm)
setmax_mm_hiwater_rss(&tsk->signal->maxrss, tsk->mm);
}
+
+ posixtimer_exit(group_dead);
+
acct_collect(code, group_dead);
if (group_dead)
tty_audit_exit();
diff --git a/kernel/futex/requeue.c b/kernel/futex/requeue.c
index b3f4a4bccb12..842d852302dd 100644
--- a/kernel/futex/requeue.c
+++ b/kernel/futex/requeue.c
@@ -744,7 +744,7 @@ int handle_early_requeue_pi_wakeup(struct futex_hash_bucket *hb,
/* Handle spurious wakeups gracefully */
ret = -EWOULDBLOCK;
- if (timeout && !timeout->task)
+ if (timeout && !hrtimer_sleeper_task_get(timeout))
ret = -ETIMEDOUT;
else if (signal_pending(current))
ret = -ERESTARTNOINTR;
diff --git a/kernel/futex/waitwake.c b/kernel/futex/waitwake.c
index d4483d15d30a..cf18309e5770 100644
--- a/kernel/futex/waitwake.c
+++ b/kernel/futex/waitwake.c
@@ -383,7 +383,7 @@ void futex_do_wait(struct futex_q *q, struct hrtimer_sleeper *timeout)
* flagged for rescheduling. Only call schedule if there
* is no timeout, or if it has yet to expire.
*/
- if (!timeout || timeout->task)
+ if (!timeout || hrtimer_sleeper_task_get(timeout))
schedule();
}
__set_current_state(TASK_RUNNING);
@@ -539,7 +539,7 @@ retry:
static void futex_sleep_multiple(struct futex_vector *vs, unsigned int count,
struct hrtimer_sleeper *to)
{
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return;
for (; count; count--, vs++) {
@@ -590,7 +590,7 @@ int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
if (ret >= 0)
return ret;
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return -ETIMEDOUT;
else if (signal_pending(current))
return -ERESTARTSYS;
@@ -725,7 +725,7 @@ retry:
if (!futex_unqueue(&q))
return 0;
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return -ETIMEDOUT;
/*
diff --git a/kernel/irq/irqdomain.c b/kernel/irq/irqdomain.c
index 57c819da30c2..4fdcb6df5306 100644
--- a/kernel/irq/irqdomain.c
+++ b/kernel/irq/irqdomain.c
@@ -344,6 +344,7 @@ static struct irq_domain *__irq_domain_instantiate(const struct irq_domain_info
err = irq_domain_alloc_generic_chips(domain, info->dgc_info);
if (err)
goto err_domain_free;
+ domain->flags |= IRQ_DOMAIN_FLAG_DESTROY_GC;
}
if (info->init) {
diff --git a/kernel/locking/rtmutex.c b/kernel/locking/rtmutex.c
index 4728631ae719..5a9534c715b8 100644
--- a/kernel/locking/rtmutex.c
+++ b/kernel/locking/rtmutex.c
@@ -1644,7 +1644,7 @@ static int __sched rt_mutex_slowlock_block(struct rt_mutex_base *lock,
break;
}
- if (timeout && !timeout->task) {
+ if (timeout && !hrtimer_sleeper_task_get(timeout)) {
ret = -ETIMEDOUT;
break;
}
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 1fe40de6ebe3..84313c9c4ba9 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf)
/* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_rq_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled);
#endif
static void update_rq_clock_task(struct rq *rq, s64 delta)
@@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta)
}
#endif
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
- if (static_key_false((&paravirt_steal_rq_enabled))) {
+ if (static_branch_unlikely(&paravirt_steal_rq_enabled)) {
u64 prev_steal;
steal = prev_steal = paravirt_steal_clock(cpu_of(rq));
@@ -2252,7 +2252,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags)
dequeue_task(rq, p, flags);
}
-static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+static bool dequeue_block_task(struct rq *rq, struct task_struct *p,
+ unsigned long task_state)
{
int flags = DEQUEUE_NOCLOCK;
@@ -2273,9 +2274,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_
*
* Where __schedule() and ttwu() have matching control dependencies.
*
- * After this, schedule() must not care about p->state any more.
+ * Once the caller invokes __block_task(), schedule() must not care about
+ * p->state any more.
*/
- if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags))
+ return dequeue_task(rq, p, DEQUEUE_SLEEP | flags);
+}
+
+static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+{
+ if (dequeue_block_task(rq, p, task_state))
__block_task(rq, p);
}
@@ -2504,6 +2511,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq)
return rq->nr_pinned;
}
+static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu)
+{
+ /* No need to migrate from a preferred CPU */
+ if (cpu_preferred(cpu))
+ return false;
+
+ /* Only FAIR tasks honor preferred CPU state */
+ if (unlikely(p->sched_class != &fair_sched_class))
+ return false;
+
+ /* Ignore preferred state if task affinity is changing */
+ if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr)))
+ return false;
+
+ return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask,
+ task_cpu_possible_mask(p));
+}
+
/*
* Per-CPU kthreads are allowed to run on !active && online CPUs, see
* __set_cpus_allowed_ptr() and select_fallback_rq().
@@ -2519,8 +2544,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
return cpu_online(cpu);
/* Non kernel threads are not allowed during either online or offline. */
- if (!(p->flags & PF_KTHREAD))
+ if (!(p->flags & PF_KTHREAD)) {
+ /* Try to use preferred CPU if task's affinity allows */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
return cpu_active(cpu);
+ }
/* KTHREAD_IS_PER_CPU is always allowed. */
if (kthread_is_per_cpu(p))
@@ -2530,7 +2559,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
if (cpu_dying(cpu))
return false;
- /* But are allowed during online. */
+ /* Try to keep unbound kthreads on a preferred CPU if possible. */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
+
+ /* Otherwise, they are allowed to run on online CPU. */
return cpu_online(cpu);
}
@@ -3773,6 +3806,7 @@ static inline void proxy_reset_donor(struct rq *rq)
WARN_ON_ONCE(rq->donor == rq->curr);
put_prev_set_next_task(rq, rq->donor, rq->curr);
+ rq->next_class = rq->curr->sched_class;
rq_set_donor(rq, rq->curr);
zap_balance_callbacks(rq);
resched_curr(rq);
@@ -3787,6 +3821,8 @@ static inline void proxy_reset_donor(struct rq *rq)
*/
static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
{
+ bool dequeued;
+
/*
* Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu.
*
@@ -3809,12 +3845,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
/* If already current, don't need to return migrate */
if (task_current(rq, p))
return false;
-
- /* If we're return migrating the rq->donor, switch it out for idle */
- if (task_current_donor(rq, p))
- proxy_reset_donor(rq);
}
- block_task(rq, p, TASK_WAKING);
+
+ dequeued = dequeue_block_task(rq, p, TASK_WAKING);
+
+ /*
+ * Dequeue @p from its scheduling class before resetting rq->donor.
+ * In particular, sched_ext needs to end the donor's running session
+ * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by
+ * proxy_reset_donor(); otherwise it would reenqueue the blocked donor.
+ *
+ * Keep on_rq set until all donor references have been replaced.
+ */
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (dequeued)
+ __block_task(rq, p);
return true;
}
#else /* !CONFIG_SCHED_PROXY_EXEC */
@@ -3905,7 +3952,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags)
* When on_rq && !on_cpu the task is preempted, see if
* it should preempt the task that is current now.
*/
- wakeup_preempt(rq, p, wake_flags);
+ wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ);
}
ttwu_do_wakeup(p);
return 1;
@@ -5149,7 +5196,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
lockdep_assert_rq_held(rq);
while (head) {
- func = (void (*)(struct rq *))head->func;
+ func = head->func;
next = head->next;
head->next = NULL;
head = next;
@@ -5789,6 +5836,9 @@ void sched_tick(void)
unsigned long hw_pressure;
u64 resched_latency;
+ if (!cpu_preferred(cpu))
+ sched_push_current_non_preferred_cpu(rq);
+
if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
arch_scale_freq_tick();
@@ -6283,10 +6333,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
* selection. In this case, do a core-wide selection.
*/
if (rq->core->core_pick_seq == rq->core->core_task_seq &&
- rq->core->core_pick_seq != rq->core_sched_seq &&
rq->core_pick) {
- WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
-
next = rq->core_pick;
rq->dl_server = rq->core_dl_server;
rq->core_pick = NULL;
@@ -6318,11 +6365,13 @@ restart:
}
/*
- * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
+ * core->core_task_seq, core->core_pick_seq
*
* @task_seq guards the task state ({en,de}queues)
* @pick_seq is the @task_seq we did a selection on
- * @sched_seq is the @pick_seq we scheduled
+ *
+ * Once a core-wide selection is committed, a non-NULL core_pick denotes
+ * a pick which still needs to be consumed on this CPU.
*
* However, preemptions can cause multiple picks on the same task set.
* 'Fix' this by also increasing @task_seq for every pick.
@@ -6429,7 +6478,6 @@ restart:
rq->core->core_pick_seq = rq->core->core_task_seq;
next = rq->core_pick;
- rq->core_sched_seq = rq->core->core_pick_seq;
/* Something should have been selected for current CPU */
WARN_ON_ONCE(!next);
@@ -6517,7 +6565,10 @@ static bool try_steal_cookie(int this, int that)
return false;
do {
- if (p == src->core_pick || p == src->curr)
+ if (p == src->core_pick || p == src->curr || p == src->donor)
+ goto next;
+
+ if (task_is_blocked(p))
goto next;
if (!is_cpu_allowed(p, this))
@@ -6820,6 +6871,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor)
block_task(rq, donor, state);
}
+/*
+ * Remove a retained proxy donor before changing its scheduler ownership.
+ * The caller holds p->pi_lock, so p cannot wake and migrate if block_task()
+ * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task
+ * queued, switching_from_fair() completes the dequeue in the immediately
+ * following sched_change_begin().
+ */
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p)
+{
+ unsigned long state = READ_ONCE(p->__state);
+
+ lockdep_assert_held(&p->pi_lock);
+ lockdep_assert_rq_held(rq);
+
+ if (!p->is_blocked || !task_on_rq_queued(p))
+ return;
+ if (WARN_ON_ONCE(state == TASK_RUNNING))
+ return;
+
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (!p->se.sched_delayed)
+ block_task(rq, p, state);
+
+ WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed);
+}
+
static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf)
__releases(__rq_lockp(rq))
{
@@ -6865,9 +6944,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
__must_hold(__rq_lockp(rq))
{
struct rq *target_rq = cpu_rq(target_cpu);
+ LIST_HEAD(migrate_list);
lockdep_assert_rq_held(rq);
- WARN_ON(p == rq->curr);
/*
* Since we are migrating a blocked donor, it could be rq->donor,
* and we want to make sure there aren't any references from this
@@ -6880,13 +6959,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
* before we release the lock.
*/
proxy_resched_idle(rq);
-
- deactivate_task(rq, p, DEQUEUE_NOCLOCK);
- proxy_set_task_cpu(p, target_cpu);
-
+ for (; p; p = p->blocked_donor) {
+ WARN_ON(p == rq->curr);
+ deactivate_task(rq, p, DEQUEUE_NOCLOCK);
+ proxy_set_task_cpu(p, target_cpu);
+ /*
+ * We can re-use se.group_node to migrate the thing,
+ * because @p is deactivated (won't be balanced) and
+ * we hold the rq_lock.
+ */
+ list_add(&p->se.group_node, &migrate_list);
+ }
proxy_release_rq_lock(rq, rf);
- attach_one_task(target_rq, p);
+ __attach_tasks(target_rq, &migrate_list);
proxy_reacquire_rq_lock(rq, rf);
}
@@ -6979,7 +7065,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) {
/* XXX Don't handle blocked owners/delayed dequeue yet */
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
__clear_task_blocked_on(p, NULL);
goto deactivate;
}
@@ -6991,7 +7077,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* and leave that CPU to sort things out.
*/
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
goto migrate_task;
}
@@ -7004,7 +7090,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* case we should end up back in find_proxy_task(), this time
* hopefully with all relevant tasks already enqueued.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
@@ -7041,7 +7127,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* So schedule rq->idle so that ttwu_runnable() can get the rq
* lock and mark owner as running.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
* OK, now we're absolutely sure @owner is on this
@@ -7051,8 +7137,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
owner->blocked_donor = p;
}
WARN_ON_ONCE(owner && !owner->on_rq);
+
+ if (owner && !sched_cpu_cookie_match(rq, owner)) {
+ if (curr_in_chain)
+ return proxy_resched_idle(rq);
+ p = donor; /* Deactivate the donor, not the runnable owner */
+ clear_task_blocked_on(p, NULL);
+ goto deactivate;
+ }
return owner;
+resched_idle:
+ return proxy_resched_idle(rq);
deactivate:
proxy_deactivate(rq, p);
return NULL;
@@ -7184,13 +7280,12 @@ static void __sched notrace __schedule(int sched_mode)
}
} else if (!preempt && prev_state) {
/*
- * We pass task_is_blocked() as the should_block arg
- * in order to keep mutex-blocked tasks on the runqueue
- * for slection with proxy-exec (without proxy-exec
- * task_is_blocked() will always be false).
+ * Keep mutex-blocked tasks on the runqueue for proxy execution
+ * only when their scheduling class allows it. Without proxy
+ * execution, task_is_blocked() always returns false.
*/
try_to_block_task(rq, prev, &prev_state,
- !task_is_blocked(prev));
+ !task_is_blocked(prev) || !scx_allow_proxy_exec(prev));
switch_count = &prev->nvcsw;
}
@@ -7211,6 +7306,7 @@ pick_again:
}
if (next == rq->idle) {
zap_balance_callbacks(rq);
+ scx_proxy_reenqueue_retry(rq, next);
goto keep_resched;
}
}
@@ -7229,8 +7325,10 @@ pick_again:
* on_cpu.
*/
donor->sched_class->put_prev_task(rq, donor, donor);
- donor->sched_class->set_next_task(rq, donor, true);
+ donor->sched_class->set_next_task(rq, donor, SNT_PICK);
}
+ scx_proxy_donor_start(rq);
+ scx_proxy_reenqueue_retry(rq, next);
} else {
rq_set_donor(rq, next);
}
@@ -7489,27 +7587,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void)
NOKPROBE_SYMBOL(preempt_schedule);
EXPORT_SYMBOL(preempt_schedule);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# ifndef preempt_schedule_dynamic_enabled
-# define preempt_schedule_dynamic_enabled preempt_schedule
-# define preempt_schedule_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule);
-void __sched notrace dynamic_preempt_schedule(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule))
- return;
- preempt_schedule();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule);
-EXPORT_SYMBOL(dynamic_preempt_schedule);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/**
* preempt_schedule_notrace - preempt_schedule called by tracing
*
@@ -7562,27 +7639,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void)
}
EXPORT_SYMBOL_GPL(preempt_schedule_notrace);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# ifndef preempt_schedule_notrace_dynamic_enabled
-# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace
-# define preempt_schedule_notrace_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace);
-void __sched notrace dynamic_preempt_schedule_notrace(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace))
- return;
- preempt_schedule_notrace();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace);
-EXPORT_SYMBOL(dynamic_preempt_schedule_notrace);
-# endif
-#endif
-
#endif /* CONFIG_PREEMPTION */
/*
@@ -7799,7 +7855,7 @@ out_unlock:
}
#endif /* CONFIG_RT_MUTEXES */
-#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
+#if !defined(CONFIG_PREEMPTION)
int __sched __cond_resched(void)
{
if (should_resched(0) && !irqs_disabled()) {
@@ -7827,38 +7883,6 @@ int __sched __cond_resched(void)
EXPORT_SYMBOL(__cond_resched);
#endif
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# define cond_resched_dynamic_enabled __cond_resched
-# define cond_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(cond_resched);
-
-# define might_resched_dynamic_enabled __cond_resched
-# define might_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(might_resched);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched);
-int __sched dynamic_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_cond_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_cond_resched);
-
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched);
-int __sched dynamic_might_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_might_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_might_resched);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/*
* __cond_resched_lock() - if a reschedule is pending, drop the given lock,
* call schedule, and on return reacquire the lock.
@@ -7928,50 +7952,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write);
# endif
/*
- * SC:cond_resched
- * SC:might_resched
- * SC:preempt_schedule
- * SC:preempt_schedule_notrace
- * SC:irqentry_exit_cond_resched
- *
- *
* NONE:
- * cond_resched <- __cond_resched
- * might_resched <- RET0
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* VOLUNTARY:
- * cond_resched <- __cond_resched
- * might_resched <- __cond_resched
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* FULL:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- false
*
* LAZY:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- true
*/
enum {
preempt_dynamic_undefined = -1,
- preempt_dynamic_none,
- preempt_dynamic_voluntary,
preempt_dynamic_full,
preempt_dynamic_lazy,
};
@@ -7980,21 +7975,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined;
int sched_dynamic_mode(const char *str)
{
-# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY))
- if (!strcmp(str, "none"))
- return preempt_dynamic_none;
-
- if (!strcmp(str, "voluntary"))
- return preempt_dynamic_voluntary;
-# endif
-
if (!strcmp(str, "full"))
return preempt_dynamic_full;
-# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY
if (!strcmp(str, "lazy"))
return preempt_dynamic_lazy;
-# endif
return -EINVAL;
}
@@ -8002,71 +7987,18 @@ int sched_dynamic_mode(const char *str)
# define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key)
# define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key)
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled)
-# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled)
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f)
-# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f)
-# else
-# error "Unsupported PREEMPT_DYNAMIC mechanism"
-# endif
-
static DEFINE_MUTEX(sched_dynamic_mutex);
static void __sched_dynamic_update(int mode)
{
- /*
- * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in
- * the ZERO state, which is invalid.
- */
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
-
switch (mode) {
- case preempt_dynamic_none:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: none\n");
- break;
-
- case preempt_dynamic_voluntary:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: voluntary\n");
- break;
-
case preempt_dynamic_full:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_disable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: full\n");
break;
case preempt_dynamic_lazy:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_enable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: lazy\n");
@@ -8099,11 +8031,7 @@ __setup("preempt=", setup_preempt_mode);
static void __init preempt_dynamic_init(void)
{
if (preempt_dynamic_mode == preempt_dynamic_undefined) {
- if (IS_ENABLED(CONFIG_PREEMPT_NONE)) {
- sched_dynamic_update(preempt_dynamic_none);
- } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) {
- sched_dynamic_update(preempt_dynamic_voluntary);
- } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
+ if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
sched_dynamic_update(preempt_dynamic_lazy);
} else {
/* Default static call setting, nothing to do */
@@ -8123,8 +8051,6 @@ static void __init preempt_dynamic_init(void)
} \
EXPORT_SYMBOL_GPL(preempt_model_##mode)
-PREEMPT_MODEL_ACCESSOR(none);
-PREEMPT_MODEL_ACCESSOR(voluntary);
PREEMPT_MODEL_ACCESSOR(full);
PREEMPT_MODEL_ACCESSOR(lazy);
@@ -8137,7 +8063,7 @@ static inline void preempt_dynamic_init(void) { }
#endif /* CONFIG_PREEMPT_DYNAMIC */
const char *preempt_modes[] = {
- "none", "voluntary", "full", "lazy", NULL,
+ "full", "lazy", NULL,
};
const char *preempt_model_str(void)
@@ -8759,6 +8685,9 @@ int sched_cpu_activate(unsigned int cpu)
*/
sched_set_rq_online(rq, cpu);
+ /* preferred is subset of active and follows its state */
+ set_cpu_preferred(cpu, true);
+
return 0;
}
@@ -8772,6 +8701,8 @@ int sched_cpu_deactivate(unsigned int cpu)
if (ret)
return ret;
+ set_cpu_preferred(cpu, false);
+
/*
* Remove CPU from nohz.idle_cpus_mask to prevent participating in
* load balancing when not active
@@ -11349,3 +11280,88 @@ void sched_change_end(struct sched_change_ctx *ctx)
p->sched_class->prio_changed(rq, p, ctx->prio);
}
}
+
+#ifdef CONFIG_PREFERRED_CPU
+static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work);
+
+static int sched_non_preferred_cpu_push_stop(void *arg)
+{
+ struct task_struct *p = arg;
+ struct rq *rq = this_rq();
+ struct rq_flags rf;
+ int cpu;
+
+ if (cpu_preferred(rq->cpu)) {
+ scoped_guard(rq_lock_irqsave, rq)
+ rq->npc_push_work_pending = false;
+ put_task_struct(p);
+ return 0;
+ }
+
+ scoped_guard (raw_spinlock_irq, &p->pi_lock) {
+ /*
+ * select_fallback_rq() may acquire the rq lock in case of
+ * fallback. So call it before grabbing rq lock. If the task
+ * migrates to another CPU before the rq lock is acquired,
+ * subsequent validation of task's current rq will help to
+ * safely bail out.
+ */
+ cpu = select_fallback_rq(rq->cpu, p);
+ rq_lock(rq, &rf);
+ rq->npc_push_work_pending = false;
+ update_rq_clock(rq);
+ context_unsafe_alias(rq);
+
+ if (task_rq(p) == rq && task_on_rq_queued(p)) {
+ struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu);
+
+ if (rq != dest_rq)
+ schedstat_inc(p->stats.nr_migrations_cpu_non_preferred);
+ rq = dest_rq;
+ }
+ rq_unlock(rq, &rf);
+ }
+
+ put_task_struct(p);
+ return 0;
+}
+
+/*
+ * Push the current task running on non-preferred CPU(npc).
+ * Using this non preferred CPU will lead to more contention
+ * in the host. So it is better not to use this CPU.
+ *
+ * Since task is running, call a stopper to push the task out. This is
+ * similar to how task moves during hotplug. In select_fallback_rq() a
+ * preferred CPU will be chosen and henceforth task shouldn't come back to
+ * this CPU again.
+ *
+ * Works for FAIR class only.
+ *
+ * If task is affined only on non-preferred CPUs, no point in moving it out.
+ */
+void sched_push_current_non_preferred_cpu(struct rq *rq)
+{
+ struct task_struct *push_task = rq->curr;
+
+ scoped_guard(rq_lock, rq) {
+ /* Push the task if its explicit affinity allows */
+ if (!task_can_migrate_to_preferred(push_task, rq->cpu))
+ return;
+
+ /* There is already a stopper thread. Don't race with it. */
+ if (rq->npc_push_work_pending)
+ return;
+
+ if (is_migration_disabled(push_task))
+ return;
+
+ rq->npc_push_work_pending = true;
+ }
+
+ /* sched_tick runs with interrupts disabled. */
+ get_task_struct(push_task);
+ stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop,
+ push_task, this_cpu_ptr(&npc_push_task_work));
+}
+#endif
diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c
index 06bddaa738e5..f16970ca81d0 100644
--- a/kernel/sched/cputime.c
+++ b/kernel/sched/cputime.c
@@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta)
* occasion account more time than the calling functions think elapsed.
*/
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled);
#ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN
static u64 native_steal_clock(int cpu)
@@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock);
static __always_inline u64 steal_account_process_time(u64 maxtime)
{
#ifdef CONFIG_PARAVIRT
- if (static_key_false(&paravirt_steal_enabled)) {
+ if (static_branch_unlikely(&paravirt_steal_enabled)) {
u64 steal;
steal = paravirt_steal_clock(smp_processor_id());
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index 0663c00c41c0..c0ebdcde5fe5 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se)
* chosen as the deadline is too small, don't even try to
* start the timer in the past!
*/
- if (ktime_us_delta(act, now) < 0)
+ if (ktime_before(act, now))
return 0;
/*
@@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se)
* DL keeps current in tree, because ->deadline is not typically changed while
* a task is runnable.
*/
-static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_dl_entity *dl_se = &p->dl;
struct dl_rq *dl_rq = &rq->dl;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_dl_rq(&p->dl))
update_stats_wait_end_dl(dl_rq, dl_se);
@@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(dl_rq->curr);
dl_rq->curr = dl_se;
- if (!first)
+ if (type != SNT_PICK)
return;
if (rq->donor->sched_class != &dl_sched_class)
diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c
index 72236db67983..e6a3b516c703 100644
--- a/kernel/sched/debug.c
+++ b/kernel/sched/debug.c
@@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v)
#ifdef CONFIG_JUMP_LABEL
-#define jump_label_key__true STATIC_KEY_INIT_TRUE
-#define jump_label_key__false STATIC_KEY_INIT_FALSE
+#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT }
+#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT }
#define SCHED_FEAT(name, enabled) \
jump_label_key__##enabled ,
-struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
+union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = {
#include "features.h"
};
@@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
static void sched_feat_disable(int i)
{
- static_key_disable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true);
}
static void sched_feat_enable(int i)
{
- static_key_enable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false);
}
#else /* !CONFIG_JUMP_LABEL: */
static void sched_feat_disable(int i) { };
@@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf,
static int sched_dynamic_show(struct seq_file *m, void *v)
{
- int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2;
int mode = READ_ONCE(preempt_dynamic_mode);
- int j;
- /* Count entries in NULL terminated preempt_modes */
- for (j = 0; preempt_modes[j]; j++)
- ;
- j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY);
-
- for (; i < j; i++) {
+ /* Stop at NULL terminator */
+ for (int i = 0; preempt_modes[i]; i++) {
if (mode == i)
seq_puts(m, "(");
seq_puts(m, preempt_modes[i]);
@@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns,
P_SCHEDSTAT(nr_failed_migrations_running);
P_SCHEDSTAT(nr_failed_migrations_hot);
P_SCHEDSTAT(nr_forced_migrations);
+ P_SCHEDSTAT(nr_migrations_cpu_non_preferred);
P_SCHEDSTAT(nr_wakeups);
P_SCHEDSTAT(nr_wakeups_sync);
P_SCHEDSTAT(nr_wakeups_migrate);
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index e56c3c95018f..aed5286b82aa 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -24,6 +24,11 @@
DEFINE_RAW_SPINLOCK(scx_sched_lock);
+bool scx_allow_proxy_exec(const struct task_struct *p)
+{
+ return true;
+}
+
/*
* NOTE: sched_ext is in the process of growing multiple scheduler support and
* scx_root usage is in a transitional state. Naked dereferences are safe if the
@@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq)
schedule_deferred(rq);
}
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next)
+{
+}
+
void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq,
u64 reenq_flags, struct rq *locked_rq)
{
@@ -3021,10 +3030,13 @@ has_tasks:
return verdict;
}
-static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct scx_sched *sch = scx_task_sched(p);
+ if (type == SNT_REPICK)
+ return;
+
if (p->scx.flags & SCX_TASK_QUEUED) {
/*
* Core-sched might decide to execute @p before it is
@@ -3082,6 +3094,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
}
}
+void scx_proxy_donor_start(struct rq *rq)
+{
+}
+
static enum scx_cpu_preempt_reason
preempt_reason_from_class(const struct sched_class *class)
{
diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h
index 0b7fc46aee08..3cfbfeb1bf9d 100644
--- a/kernel/sched/ext/ext.h
+++ b/kernel/sched/ext/ext.h
@@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq);
int scx_check_setscheduler(struct task_struct *p, int policy);
bool task_should_scx(int policy);
bool scx_allow_ttwu_queue(const struct task_struct *p);
+bool scx_allow_proxy_exec(const struct task_struct *p);
+void scx_proxy_donor_start(struct rq *rq);
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next);
void init_sched_ext_class(void);
static inline u32 scx_cpuperf_target(s32 cpu)
@@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {}
static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; }
static inline bool task_on_scx(const struct task_struct *p) { return false; }
static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; }
+static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; }
+static inline void scx_proxy_donor_start(struct rq *rq) {}
+static inline void scx_proxy_reenqueue_retry(struct rq *rq,
+ struct task_struct *next) {}
static inline void init_sched_ext_class(void) {}
#endif /* CONFIG_SCHED_CLASS_EXT */
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 8d38c3b7d792..56f4ab6d9ada 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -23,6 +23,7 @@
#include <linux/energy_model.h>
#include <linux/mmap_lock.h>
#include <linux/jiffies.h>
+#include <linux/math.h>
#include <linux/mm_api.h>
#include <linux/highmem.h>
#include <linux/hrtimer.h>
@@ -819,12 +820,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq)
if (curr && !curr->on_rq)
curr = NULL;
- /*
- * This is called from set_next_task_fair(.first=true) /
- * set_protect_slice() so curr had better be set and on_rq.
- */
- WARN_ON_ONCE(!curr);
-
if (weight) {
s64 runtime = cfs_rq->sum_w_vruntime;
@@ -1136,10 +1131,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
/* If there are shorter slices than se's one */
if (slice != se->slice) {
+ vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
if (sched_feat(PREEMPT_SHORT))
vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
- else
- vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
}
se->vprot = vprot;
@@ -1147,10 +1141,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se)
{
- u64 slice = cfs_rq_min_slice(cfs_rq);
u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq));
+ u64 slice = normalized_sysctl_sched_base_slice;
+ u64 vprot;
- se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+ if (sched_feat(RUN_TO_PARITY))
+ slice = cfs_rq_min_slice(cfs_rq);
+
+ vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+
+ if (sched_feat(PREEMPT_SHORT) && slice != se->slice)
+ vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
+
+ se->vprot = vprot;
}
static inline bool protect_slice(struct sched_entity *se)
@@ -3712,7 +3715,7 @@ static void update_task_scan_period(struct task_struct *p,
p->mm->numa_next_scan = jiffies +
msecs_to_jiffies(p->numa_scan_period);
- return;
+ goto out;
}
/*
@@ -3756,7 +3759,10 @@ static void update_task_scan_period(struct task_struct *p,
p->numa_scan_period = clamp(p->numa_scan_period + diff,
task_scan_min(p), task_scan_max(p));
- memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality));
+
+out:
+ memset(p->numa_faults_locality, 0,
+ sizeof(p->numa_faults_locality));
}
/*
@@ -8208,7 +8214,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
struct sched_entity *se = &p->se;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight;
- bool curr;
if (task_is_throttled(p) && enqueue_throttled_task(p))
return;
@@ -8237,23 +8242,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
if (p->in_iowait)
cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
- /*
- * XXX comment on the curr thing
- */
- curr = (cfs_rq->curr == se);
- if (curr)
- place_entity(cfs_rq, se, flags);
if (se->on_rq && se->sched_delayed)
requeue_delayed_entity(cfs_rq, se);
weight = enqueue_hierarchy(p, flags);
-
- if (!curr) {
- reweight_eevdf(cfs_rq, se, weight, false);
- place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
- __enqueue_entity(cfs_rq, se);
- }
+ reweight_eevdf(cfs_rq, se, weight, false);
+ place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
+ __enqueue_entity(cfs_rq, se);
if (!rq_h_nr_queued && rq->cfs.h_nr_queued)
dl_server_start(&rq->fair_server);
@@ -8673,8 +8669,8 @@ static int
sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu)
{
unsigned long load, min_load = ULONG_MAX;
- unsigned int min_exit_latency = UINT_MAX;
- u64 latest_idle_timestamp = 0;
+ u64 min_exit_latency = U64_MAX;
+ unsigned int nr_candidates = 0;
int least_loaded_cpu = this_cpu;
int shallowest_idle_cpu = -1;
int i;
@@ -8695,24 +8691,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *
if (available_idle_cpu(i)) {
struct cpuidle_state *idle = idle_get_state(rq);
- if (idle && idle->exit_latency < min_exit_latency) {
- /*
- * We give priority to a CPU whose idle state
- * has the smallest exit latency irrespective
- * of any idle timestamp.
- */
- min_exit_latency = idle->exit_latency;
- latest_idle_timestamp = rq->idle_stamp;
- shallowest_idle_cpu = i;
- } else if ((!idle || idle->exit_latency == min_exit_latency) &&
- rq->idle_stamp > latest_idle_timestamp) {
- /*
- * If equal or no active idle state, then
- * the most recently idled CPU might have
- * a warmer cache.
- */
- latest_idle_timestamp = rq->idle_stamp;
+ u64 exit_latency = idle ? idle->exit_latency : U64_MAX;
+
+ if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) {
+ min_exit_latency = exit_latency;
shallowest_idle_cpu = i;
+ nr_candidates = 1;
+ } else if (exit_latency == min_exit_latency) {
+ nr_candidates++;
+ if (!reciprocal_scale(sched_rng(), nr_candidates))
+ shallowest_idle_cpu = i;
}
} else if (shallowest_idle_cpu == -1) {
load = cpu_load(cpu_rq(i));
@@ -10071,8 +10059,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity
static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse)
{
- if (cfs_rq->next && cfs_rq->next->slice < pse->slice)
- return false;
+ if (cfs_rq->next) {
+ if (cfs_rq->next->slice < pse->slice)
+ return false;
+
+ if (cfs_rq->next->slice == pse->slice &&
+ entity_before(cfs_rq->next, pse))
+ return false;
+ }
set_next_buddy(cfs_rq, pse);
return true;
@@ -11438,21 +11432,7 @@ next:
*/
static void attach_tasks(struct lb_env *env)
{
- struct list_head *tasks = &env->tasks;
- struct task_struct *p;
- struct rq_flags rf;
-
- rq_lock(env->dst_rq, &rf);
- update_rq_clock(env->dst_rq);
-
- while (!list_empty(tasks)) {
- p = list_first_entry(tasks, struct task_struct, se.group_node);
- list_del_init(&p->se.group_node);
-
- attach_task(env->dst_rq, p);
- }
-
- rq_unlock(env->dst_rq, &rf);
+ __attach_tasks(env->dst_rq, &env->tasks);
}
#ifdef CONFIG_NO_HZ_COMMON
@@ -13745,7 +13725,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
};
bool need_unlock = false;
- cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask);
+ cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask);
schedstat_inc(sd->lb_count[idle]);
@@ -14870,10 +14850,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
*/
this_rq->idle_stamp = rq_clock(this_rq);
- /*
- * Do not pull tasks towards !active CPUs...
- */
- if (!cpu_active(this_cpu))
+ /* Do not pull tasks towards !preferred CPUs */
+ if (!cpu_preferred(this_cpu))
return 0;
/*
@@ -15513,14 +15491,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p)
}
}
-static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_entity *se = &p->se;
- bool throttled = false;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight = NICE_0_LOAD;
+ bool first = type == SNT_PICK;
+ bool throttled = false;
bool on_rq = se->on_rq;
+ if (type == SNT_REPICK)
+ goto repick;
+
clear_buddies(cfs_rq, se);
if (on_rq)
@@ -15564,11 +15546,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(se->sched_delayed);
- if (hrtick_enabled_fair(rq))
- hrtick_start_fair(rq, p);
-
update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
+
+repick:
+ /*
+ * A same-task repick skips put_prev_task_fair(), but
+ * pick_task_fair() refreshed the entity hrtick_start_fair() reads
+ * before selecting it again. rq->cfs.curr identifies that entity,
+ * including with group scheduling.
+ */
+ if (hrtick_enabled_fair(rq))
+ hrtick_start_fair(rq, p);
}
void init_cfs_rq(struct cfs_rq *cfs_rq)
diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c
index eb73b65ce6c4..76f3c84ca684 100644
--- a/kernel/sched/idle.c
+++ b/kernel/sched/idle.c
@@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t
update_rq_avg_idle(rq);
}
-static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first)
+static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
update_idle_core(rq);
scx_update_idle(rq, true, true);
schedstat_inc(rq->sched_goidle);
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
index 85303add726d..1535046a23ff 100644
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags)
check_preempt_equal_prio(rq, p);
}
-static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first)
+static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_rt_entity *rt_se = &p->rt;
struct rt_rq *rt_rq = &rq->rt;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_rt_rq(&p->rt))
update_stats_wait_end_rt(rt_rq, rt_se);
@@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f
/* The running task is never eligible for pushing */
dequeue_pushable_task(rq, p);
- if (!first)
+ if (type != SNT_PICK)
return;
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e656c7059bf8..7d2ec527b8a2 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1326,6 +1326,9 @@ struct rq {
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
u64 prev_steal_time_rq;
#endif
+#ifdef CONFIG_PREFERRED_CPU
+ bool npc_push_work_pending;
+#endif
/* calc_load related fields */
unsigned long calc_load_update;
@@ -1371,7 +1374,6 @@ struct rq {
struct task_struct *core_pick;
struct sched_dl_entity *core_dl_server;
unsigned int core_enabled;
- unsigned int core_sched_seq;
struct rb_root core_tree;
/* shared state -- careful with sched_core_cpu_deactivate() */
@@ -2447,16 +2449,25 @@ extern __read_mostly unsigned int sysctl_sched_features;
#ifdef CONFIG_JUMP_LABEL
-#define SCHED_FEAT(name, enabled) \
-static __always_inline bool static_branch_##name(struct static_key *key) \
-{ \
- return static_key_##enabled(key); \
+union sched_feat_key {
+ struct static_key_true key_true;
+ struct static_key_false key_false;
+};
+
+#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true)
+#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false)
+
+#define SCHED_FEAT(name, enabled) \
+static __always_inline bool \
+static_branch_##name(union sched_feat_key *key) \
+{ \
+ return sched_feat_branch_##enabled(key); \
}
#include "features.h"
#undef SCHED_FEAT
-extern struct static_key sched_feat_keys[__SCHED_FEAT_NR];
+extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR];
#define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x]))
#else /* !CONFIG_JUMP_LABEL: */
@@ -2508,6 +2519,12 @@ static inline bool task_is_blocked(struct task_struct *p)
return !!p->blocked_on;
}
+#ifdef CONFIG_SCHED_PROXY_EXEC
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p);
+#else
+static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {}
+#endif
+
static inline int task_on_cpu(struct rq *rq, struct task_struct *p)
{
return p->on_cpu;
@@ -2527,11 +2544,17 @@ static inline int task_on_rq_migrating(struct task_struct *p)
#define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */
#define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */
#define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */
-
-#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */
+/*
+ * Hint that the caller expects the waker to sleep soon.
+ * Scheduler classes may use it for placement or preemption.
+ * Callers must not rely on it to prevent migration,
+ * preserve CPU locality or make the wakee run next.
+ */
+#define WF_SYNC 0x10
#define WF_MIGRATED 0x20 /* Internal use, task got migrated */
#define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */
#define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */
+#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */
static_assert(WF_EXEC == SD_BALANCE_EXEC);
static_assert(WF_FORK == SD_BALANCE_FORK);
@@ -2621,6 +2644,12 @@ struct affinity_context {
extern s64 update_curr_common(struct rq *rq);
+enum snt_e {
+ SNT_NORMAL, /* set_next_task() */
+ SNT_PICK, /* put_prev_set_next_task(): prev != next */
+ SNT_REPICK, /* put_prev_set_next_task(): prev == next */
+};
+
struct sched_class {
#ifdef CONFIG_UCLAMP_TASK
@@ -2678,7 +2707,7 @@ struct sched_class {
* __schedule: rq->lock
*/
void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next);
- void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first);
+ void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type);
/*
* select_task_rq: p->pi_lock
@@ -2781,7 +2810,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev)
static inline void set_next_task(struct rq *rq, struct task_struct *next)
{
- next->sched_class->set_next_task(rq, next, false);
+ next->sched_class->set_next_task(rq, next, SNT_NORMAL);
}
static inline void
@@ -2802,11 +2831,13 @@ static inline void put_prev_set_next_task(struct rq *rq,
__put_prev_set_next_dl_server(rq, prev, next);
- if (next == prev)
+ if (next == prev) {
+ next->sched_class->set_next_task(rq, next, SNT_REPICK);
return;
+ }
prev->sched_class->put_prev_task(rq, prev, next);
- next->sched_class->set_next_task(rq, next, true);
+ next->sched_class->set_next_task(rq, next, SNT_PICK);
}
/*
@@ -3139,6 +3170,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p)
attach_task(rq, p);
}
+/*
+ * __attach_tasks() - attaches a list of tasks (using se.group_node) to
+ * the new rq
+ */
+static inline void __attach_tasks(struct rq *rq, struct list_head *tasks)
+{
+ guard(rq_lock)(rq);
+ update_rq_clock(rq);
+
+ while (!list_empty(tasks)) {
+ struct task_struct *p;
+
+ p = list_first_entry(tasks, struct task_struct, se.group_node);
+ list_del_init(&p->se.group_node);
+
+ attach_task(rq, p);
+ }
+}
+
#ifdef CONFIG_PREEMPT_RT
# define SCHED_NR_MIGRATE_BREAK 8
#else
@@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change)
#include "ext/ext.h"
+#ifdef CONFIG_PREFERRED_CPU
+void sched_push_current_non_preferred_cpu(struct rq *rq);
+#else /* !CONFIG_PREFERRED_CPU */
+static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { }
+#endif
+
#endif /* _KERNEL_SCHED_SCHED_H */
diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c
index c909ca0d8c87..1e0109ec36b3 100644
--- a/kernel/sched/stop_task.c
+++ b/kernel/sched/stop_task.c
@@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags)
/* we're never preempted */
}
-static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first)
+static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
stop->se.exec_start = rq_clock_task(rq);
}
diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c
index d033f600f48c..477e4bf9c01e 100644
--- a/kernel/sched/wait.c
+++ b/kernel/sched/wait.c
@@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
+ * Passes WF_SYNC to waitqueue wake functions. The default wake function
+ * forwards it to the scheduler; see WF_SYNC for the hint's semantics.
*
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * If this function wakes up a task, it executes a full memory barrier
+ * before accessing the task state.
*/
void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode,
void *key)
@@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs in that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
- *
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * Same as __wake_up_sync_key(), but called with @wq_head->lock held.
*/
void __wake_up_locked_sync_key(struct wait_queue_head *wq_head,
unsigned int mode, void *key)
diff --git a/kernel/time/hrtimer.c b/kernel/time/hrtimer.c
index cbf1693c86b3..17dd38a6cee7 100644
--- a/kernel/time/hrtimer.c
+++ b/kernel/time/hrtimer.c
@@ -780,7 +780,8 @@ static void hrtimer_switch_to_hres(void)
return;
}
base->hres_active = true;
- hrtimer_resolution = HIGH_RES_NSEC;
+ if (hrtimer_resolution != HIGH_RES_NSEC)
+ hrtimer_resolution = HIGH_RES_NSEC;
tick_setup_sched_timer(true);
/* "Retrigger" the interrupt to get things going */
@@ -2003,7 +2004,7 @@ bool hrtimer_active(const struct hrtimer *timer)
base = READ_ONCE(timer->base);
seq = raw_read_seqcount_begin(&base->seq);
- if (timer->is_queued || base->running == timer)
+ if (timer->is_queued || READ_ONCE(base->running) == timer)
return true;
} while (read_seqcount_retry(&base->seq, seq) || base != READ_ONCE(timer->base));
@@ -2040,7 +2041,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc
lockdep_assert_held(&cpu_base->lock);
debug_hrtimer_deactivate(timer);
- base->running = timer;
+ WRITE_ONCE(base->running, timer);
/*
* Separate the ->running assignment from the ->is_queued assignment.
@@ -2099,7 +2100,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc
raw_write_seqcount_barrier(&base->seq);
WARN_ON_ONCE(base->running != timer);
- base->running = NULL;
+ WRITE_ONCE(base->running, NULL);
}
static void __hrtimer_run_queues(struct hrtimer_cpu_base *cpu_base, ktime_t now,
@@ -2323,9 +2324,9 @@ void hrtimer_run_queues(void)
static enum hrtimer_restart hrtimer_wakeup(struct hrtimer *timer)
{
struct hrtimer_sleeper *t = container_of(timer, struct hrtimer_sleeper, timer);
- struct task_struct *task = t->task;
+ struct task_struct *task = hrtimer_sleeper_task_get(t);
- t->task = NULL;
+ hrtimer_sleeper_task_set(t, NULL);
if (task)
wake_up_process(task);
@@ -2354,7 +2355,7 @@ void hrtimer_sleeper_start_expires(struct hrtimer_sleeper *sl, enum hrtimer_mode
/* If already expired, clear the task pointer and set current state to running */
if (!hrtimer_start_expires_user(&sl->timer, mode)) {
- sl->task = NULL;
+ hrtimer_sleeper_task_set(sl, NULL);
__set_current_state(TASK_RUNNING);
}
}
@@ -2388,7 +2389,7 @@ static void __hrtimer_setup_sleeper(struct hrtimer_sleeper *sl, clockid_t clock_
}
__hrtimer_setup(&sl->timer, hrtimer_wakeup, clock_id, mode);
- sl->task = current;
+ hrtimer_sleeper_task_set(sl, current);
}
/**
@@ -2432,17 +2433,17 @@ static int __sched do_nanosleep(struct hrtimer_sleeper *t, enum hrtimer_mode mod
set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE);
hrtimer_sleeper_start_expires(t, mode);
- if (likely(t->task))
+ if (likely(hrtimer_sleeper_task_get(t)))
schedule();
hrtimer_cancel(&t->timer);
mode = HRTIMER_MODE_ABS;
- } while (t->task && !signal_pending(current));
+ } while (hrtimer_sleeper_task_get(t) && !signal_pending(current));
__set_current_state(TASK_RUNNING);
- if (!t->task)
+ if (!hrtimer_sleeper_task_get(t))
return 0;
restart = &current->restart_block;
diff --git a/kernel/time/posix-cpu-timers.c b/kernel/time/posix-cpu-timers.c
index 0bf4fcd969c8..cd75d4bb5b64 100644
--- a/kernel/time/posix-cpu-timers.c
+++ b/kernel/time/posix-cpu-timers.c
@@ -439,6 +439,38 @@ static void trigger_base_recalc_expires(struct k_itimer *timer,
base->nextevt = 0;
}
+static inline bool cpu_timer_enqueue(struct timerqueue_head *head,
+ struct cpu_timer *ctmr)
+{
+ ctmr->head = head;
+ return timerqueue_add(head, &ctmr->node);
+}
+
+static inline bool cpu_timer_queued(struct cpu_timer *ctmr)
+{
+ return !!ctmr->head;
+}
+
+static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr)
+{
+ if (cpu_timer_queued(ctmr)) {
+ timerqueue_del(ctmr->head, &ctmr->node);
+ ctmr->head = NULL;
+ return true;
+ }
+ return false;
+}
+
+static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr)
+{
+ return ctmr->node.expires;
+}
+
+static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp)
+{
+ ctmr->node.expires = exp;
+}
+
/*
* Dequeue the timer and reset the base if it was its earliest expiration.
* It makes sure the next tick recalculates the base next expiration so we
@@ -607,6 +639,7 @@ static int posix_cpu_timer_del(struct k_itimer *timer)
}
if (!ret) {
+ WARN_ON_ONCE(cpu_timer_queued(&timer->it.cpu));
put_pid(timer->it.cpu.pid);
timer->it_status = POSIX_TIMER_DISARMED;
}
@@ -639,18 +672,50 @@ static void cleanup_timers(struct posix_cputimers *pct)
cleanup_timerqueue(&pct->bases[CPUCLOCK_SCHED].tqhead);
}
+static inline void posix_cpu_timers_exit_work(void);
+
/*
- * These are both called with the siglock held, when the current thread
- * is being reaped. When the final (leader) thread in the group is reaped,
- * posix_cpu_timers_exit_group will be called after posix_cpu_timers_exit.
+ * Invoked from posixtimer_exit_task() after PF_EXITING was set in tsk::flags or
+ * from posixtimer_exec_cleanup().
*/
-void posix_cpu_timers_exit(struct task_struct *tsk)
+void posix_cpu_timers_exit_task(void)
{
- cleanup_timers(&tsk->posix_cputimers);
+ posix_cpu_timers_exit_work();
+
+ guard(spinlock_irq)(&current->sighand->siglock);
+ cleanup_timers(&current->posix_cputimers);
}
-void posix_cpu_timers_exit_group(struct task_struct *tsk)
+
+/*
+ * Invoked from posixtimer_exit_group() after PF_EXITING was set in tsk::flags.
+ */
+void posix_cpu_timers_exit_group(void)
{
- cleanup_timers(&tsk->signal->posix_cputimers);
+ posix_cpu_timers_exit_task();
+
+ guard(spinlock_irq)(&current->sighand->siglock);
+ cleanup_timers(&current->signal->posix_cputimers);
+}
+
+/*
+ * This function validates that POSIX CPU timers can be safely enqueued on the
+ * target task.
+ *
+ * Enqueue is allowed when PF_EXITING is not set. If set then it is only allowed
+ * for process shared timers (type = PIDTYPE_TGID) as long as tsk::signal::flags
+ * does not have SIGNAL_GROUP_EXIT set. PIDTYPE_PID targets are not allowed at
+ * all when the task has PF_EXITING set.
+ *
+ * This guarantees that after the POSIX timer cleanup in posixtimer_exit() no
+ * POSIX CPU timers are queued on the task or in case of a group exit on the
+ * process.
+ */
+static inline bool task_can_enqueue_timer(struct task_struct *tsk, enum pid_type type)
+{
+ if (likely(!(tsk->flags & PF_EXITING)))
+ return true;
+
+ return type == PIDTYPE_TGID && !(tsk->signal->flags & SIGNAL_GROUP_EXIT);
}
/*
@@ -663,7 +728,13 @@ static void arm_timer(struct k_itimer *timer, struct task_struct *p)
struct cpu_timer *ctmr = &timer->it.cpu;
u64 newexp = cpu_timer_getexpires(ctmr);
+ lockdep_assert_held(&p->sighand->siglock);
+
timer->it_status = POSIX_TIMER_ARMED;
+
+ if (unlikely(!task_can_enqueue_timer(p, clock_pid_type(timer->it_clock))))
+ return;
+
if (!cpu_timer_enqueue(&base->tqhead, ctmr))
return;
@@ -1201,6 +1272,20 @@ static void posix_cpu_timers_work(struct callback_head *work)
mutex_unlock(&cw->mutex);
}
+static inline void posix_cpu_timers_exit_work(void)
+{
+ /* Canceling the work is only valid for exit() but not for exec() */
+ if (!(current->flags & PF_EXITING))
+ return;
+ /*
+ * current->flags has PF_EXITING set so this can be done lockless and
+ * with interrupts enabled as PF_EXITING prevents the interrupt from
+ * scheduling the work.
+ */
+ if (current->posix_cputimers_work.scheduled)
+ task_work_cancel(current, &current->posix_cputimers_work.work);
+}
+
/*
* Invoked from the posix-timer core when a cancel operation failed because
* the timer is marked firing. The caller holds rcu_read_lock(), which
@@ -1331,6 +1416,8 @@ static inline void __run_posix_cpu_timers(struct task_struct *tsk)
lockdep_posixtimer_exit();
}
+static inline void posix_cpu_timers_exit_work(void) { }
+
static void posix_cpu_timer_wait_running(struct k_itimer *timr)
{
cpu_relax();
@@ -1477,7 +1564,7 @@ void run_posix_cpu_timers(void)
* posix_cpu_timer_del() may fail to lock_task_sighand(tsk) and
* miss timer->it.cpu.firing != 0.
*/
- if (tsk->exit_state)
+ if (tsk->flags & PF_EXITING)
return;
/*
diff --git a/kernel/time/posix-timers.c b/kernel/time/posix-timers.c
index 436ba794cc0b..188dbedbffca 100644
--- a/kernel/time/posix-timers.c
+++ b/kernel/time/posix-timers.c
@@ -1077,13 +1077,9 @@ SYSCALL_DEFINE1(timer_delete, timer_t, timer_id)
return 0;
}
-/*
- * Invoked from do_exit() when the last thread of a thread group exits.
- * At that point no other task can access the timers of the dying
- * task anymore.
- */
-void exit_itimers(struct task_struct *tsk)
+static void posixtimer_delete_timers(void)
{
+ struct task_struct *tsk = current;
struct hlist_head timers;
struct hlist_node *next;
struct k_itimer *timer;
@@ -1120,6 +1116,24 @@ void exit_itimers(struct task_struct *tsk)
}
}
+void posixtimer_exit(bool group_dead)
+{
+ if (group_dead) {
+ hrtimer_cancel(&current->signal->real_timer);
+ posix_cpu_timers_exit_group();
+ posixtimer_delete_timers();
+ } else {
+ posix_cpu_timers_exit_task();
+ }
+}
+
+void posixtimer_exec(void)
+{
+ posix_cpu_timers_exit_task();
+ posixtimer_delete_timers();
+ flush_itimer_signals();
+}
+
SYSCALL_DEFINE2(clock_settime, const clockid_t, which_clock,
const struct __kernel_timespec __user *, tp)
{
diff --git a/kernel/time/posix-timers.h b/kernel/time/posix-timers.h
index 4ea9611dd716..79fd7ea71046 100644
--- a/kernel/time/posix-timers.h
+++ b/kernel/time/posix-timers.h
@@ -51,3 +51,6 @@ int common_timer_set(struct k_itimer *timr, int flags,
struct itimerspec64 *old_setting);
void posix_timer_set_common(struct k_itimer *timer, struct itimerspec64 *new_setting);
int common_timer_del(struct k_itimer *timer);
+
+void posix_cpu_timers_exit_task(void);
+void posix_cpu_timers_exit_group(void);
diff --git a/kernel/time/sleep_timeout.c b/kernel/time/sleep_timeout.c
index 3c90574bd904..ad8c415851ae 100644
--- a/kernel/time/sleep_timeout.c
+++ b/kernel/time/sleep_timeout.c
@@ -212,7 +212,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta,
hrtimer_set_expires_range_ns(&t.timer, *expires, delta);
hrtimer_sleeper_start_expires(&t, mode);
- if (likely(t.task))
+ if (likely(hrtimer_sleeper_task_get(&t)))
schedule();
hrtimer_cancel(&t.timer);
@@ -220,7 +220,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta,
__set_current_state(TASK_RUNNING);
- return !t.task ? 0 : -EINTR;
+ return !hrtimer_sleeper_task_get(&t) ? 0 : -EINTR;
}
EXPORT_SYMBOL_GPL(schedule_hrtimeout_range_clock);
diff --git a/kernel/time/tick-sched.c b/kernel/time/tick-sched.c
index 6c3fea386713..a7893a079a83 100644
--- a/kernel/time/tick-sched.c
+++ b/kernel/time/tick-sched.c
@@ -738,14 +738,11 @@ bool tick_nohz_tick_stopped_cpu(int cpu)
*/
static void tick_nohz_update_jiffies(ktime_t now)
{
- unsigned long flags;
+ /* Reached only from irq_enter_rcu(), i.e. hard interrupt entry. */
+ lockdep_assert_irqs_disabled();
__this_cpu_write(tick_cpu_sched.idle_waketime, now);
-
- local_irq_save(flags);
tick_do_update_jiffies64(now);
- local_irq_restore(flags);
-
touch_softlockup_watchdog_sched();
}
@@ -819,7 +816,7 @@ u64 get_jiffies_update(unsigned long *basej)
*/
static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
{
- u64 basemono, next_tick, delta, expires;
+ u64 basemono, next_tick, expires;
unsigned long basejiff;
int tick_cpu;
@@ -859,8 +856,7 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
* If the tick is due in the next period, keep it ticking or
* force prod the timer.
*/
- delta = next_tick - basemono;
- if (delta <= (u64)TICK_NSEC) {
+ if (next_tick - basemono <= (u64)TICK_NSEC) {
/*
* We've not stopped the tick yet, and there's a timer in the
* next period, so no point in stopping it either, bail.
@@ -876,17 +872,19 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
* the sleep time to the timekeeping 'max_deferment' value.
* Otherwise we can sleep as long as we want.
*/
- delta = timekeeping_max_deferment();
tick_cpu = READ_ONCE(tick_do_timer_cpu);
if (tick_cpu != cpu &&
- (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST)))
- delta = KTIME_MAX;
-
- /* Calculate the next expiry time */
- if (delta < (KTIME_MAX - basemono))
- expires = basemono + delta;
- else
+ (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) {
expires = KTIME_MAX;
+ } else {
+ expires = timekeeping_max_deferment();
+
+ /* Calculate the next expiry time */
+ if (expires < (KTIME_MAX - basemono))
+ expires += basemono;
+ else
+ expires = KTIME_MAX;
+ }
ts->timer_expires = min_t(u64, expires, next_tick);
diff --git a/kernel/time/time_test.c b/kernel/time/time_test.c
index 1b99180da288..8b718767b3ba 100644
--- a/kernel/time/time_test.c
+++ b/kernel/time/time_test.c
@@ -87,8 +87,24 @@ static void time64_to_tm_test_date_range(struct kunit *test)
}
}
+static void time64_to_tm_test_wide_day_count(struct kunit *test)
+{
+ /* 2^31 days: the first count that does not fit in a 32-bit long. */
+ time64_t timestamp = (1LL << 31) * 86400;
+ struct tm result;
+
+ time64_to_tm(timestamp, 0, &result);
+
+ KUNIT_EXPECT_EQ(test, result.tm_year, 5879680);
+ KUNIT_EXPECT_EQ(test, result.tm_mon, 6);
+ KUNIT_EXPECT_EQ(test, result.tm_mday, 12);
+ KUNIT_EXPECT_EQ(test, result.tm_yday, 193);
+ KUNIT_EXPECT_EQ(test, result.tm_wday, 6);
+}
+
static struct kunit_case time_test_cases[] = {
KUNIT_CASE_SLOW(time64_to_tm_test_date_range),
+ KUNIT_CASE(time64_to_tm_test_wide_day_count),
{}
};
diff --git a/kernel/time/timeconv.c b/kernel/time/timeconv.c
index 59b922c826e7..aed3af950fa0 100644
--- a/kernel/time/timeconv.c
+++ b/kernel/time/timeconv.c
@@ -49,8 +49,9 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result)
u32 u32tmp, day_of_century, year_of_century, day_of_year, month, day;
u64 u64tmp, udays, century, year;
bool is_Jan_or_Feb, is_leap_year;
- long days, rem;
int remainder;
+ long rem;
+ s64 days;
days = div_s64_rem(totalsecs, SECS_PER_DAY, &remainder);
rem = remainder;
@@ -70,7 +71,8 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result)
result->tm_sec = rem % 60;
/* January 1, 1970 was a Thursday. */
- result->tm_wday = (4 + days) % 7;
+ div_s64_rem(days + 4, 7, &remainder);
+ result->tm_wday = remainder;
if (result->tm_wday < 0)
result->tm_wday += 7;
diff --git a/kernel/time/timekeeping.c b/kernel/time/timekeeping.c
index ea2e6e55f37b..d54c4d303db6 100644
--- a/kernel/time/timekeeping.c
+++ b/kernel/time/timekeeping.c
@@ -861,8 +861,10 @@ static void timekeeping_update_from_shadow(struct tk_data *tkd, unsigned int act
*
* Write xtime_sec first so that even if the memcpy() tears the store
* data integrity is provided for ktime_get_real_seconds().
+ * The same goes for ktime_sec and ktime_get_seconds().
*/
WRITE_ONCE(tkd->timekeeper.xtime_sec, tk->xtime_sec);
+ WRITE_ONCE(tkd->timekeeper.ktime_sec, tk->ktime_sec);
memcpy(&tkd->timekeeper, tk, sizeof(*tk));
write_seqcount_end(&tkd->seq);
}
@@ -1169,7 +1171,7 @@ time64_t ktime_get_seconds(void)
struct timekeeper *tk = &tk_core.timekeeper;
WARN_ON(timekeeping_suspended);
- return tk->ktime_sec;
+ return READ_ONCE(tk->ktime_sec);
}
EXPORT_SYMBOL_GPL(ktime_get_seconds);
diff --git a/kernel/time/timer.c b/kernel/time/timer.c
index ae9abf14688e..42afdcb229d8 100644
--- a/kernel/time/timer.c
+++ b/kernel/time/timer.c
@@ -890,7 +890,7 @@ static inline void detach_timer(struct timer_list *timer, bool clear_pending)
__hlist_del(entry);
if (clear_pending)
- entry->pprev = NULL;
+ WRITE_ONCE(entry->pprev, NULL);
entry->next = LIST_POISON2;
}
diff --git a/kernel/time/timer_migration.c b/kernel/time/timer_migration.c
index 059d43355e65..f920e73fff51 100644
--- a/kernel/time/timer_migration.c
+++ b/kernel/time/timer_migration.c
@@ -715,7 +715,7 @@ static void __tmigr_cpu_activate(struct tmigr_cpu *tmc)
trace_tmigr_cpu_active(tmc);
- tmc->cpuevt.ignore = true;
+ WRITE_ONCE(tmc->cpuevt.ignore, true);
WRITE_ONCE(tmc->wakeup, KTIME_MAX);
walk_groups(&tmigr_active_up, &data, tmc);
@@ -1258,7 +1258,7 @@ u64 tmigr_cpu_new_timer(u64 nextexp)
ret = READ_ONCE(tmc->wakeup);
if (nextexp != KTIME_MAX) {
if (nextexp != tmc->cpuevt.nextevt.expires ||
- tmc->cpuevt.ignore) {
+ READ_ONCE(tmc->cpuevt.ignore)) {
ret = tmigr_new_timer(tmc, nextexp);
/*
* Make sure the reevaluation of timers in idle path
@@ -1362,7 +1362,7 @@ static u64 __tmigr_cpu_deactivate(struct tmigr_cpu *tmc, u64 nextexp)
* or CPU goes offline.
*/
if (nextexp != KTIME_MAX)
- tmc->cpuevt.ignore = false;
+ WRITE_ONCE(tmc->cpuevt.ignore, false);
walk_groups(&tmigr_inactive_up, &data, tmc);
return data.firstexp;
diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c
index aa59919b8f2c..0e4b499328c0 100644
--- a/kernel/time/vsyscall.c
+++ b/kernel/time/vsyscall.c
@@ -41,14 +41,12 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti
nsec = tk->tkr_mono.xtime_nsec;
nsec += ((u64)tk->wall_to_monotonic.tv_nsec << tk->tkr_mono.shift);
- while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) {
- nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift);
- vdso_ts->sec++;
- }
- vdso_ts->nsec = nsec;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
/* Copy MONOTONIC time for BOOTTIME */
sec = vdso_ts->sec;
+ nsec = vdso_ts->nsec;
/* Add the boot offset */
sec += tk->monotonic_to_boot.tv_sec;
nsec += (u64)tk->monotonic_to_boot.tv_nsec << tk->tkr_mono.shift;
@@ -56,12 +54,8 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti
/* CLOCK_BOOTTIME */
vdso_ts = &vc[CS_HRES_COARSE].basetime[CLOCK_BOOTTIME];
vdso_ts->sec = sec;
-
- while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) {
- nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift);
- vdso_ts->sec++;
- }
- vdso_ts->nsec = nsec;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
/* CLOCK_MONOTONIC_RAW */
vdso_ts = &vc[CS_RAW].basetime[CLOCK_MONOTONIC_RAW];
@@ -161,11 +155,11 @@ void vdso_time_update_aux(struct timekeeper *tk)
vdso_ts->sec = tk->xtime_sec + tk->monotonic_to_aux.tv_sec;
- nsec = tk->tkr_mono.xtime_nsec >> tk->tkr_mono.shift;
- nsec += tk->monotonic_to_aux.tv_nsec;
- vdso_ts->sec += __iter_div_u64_rem(nsec, NSEC_PER_SEC, &nsec);
- nsec = nsec << tk->tkr_mono.shift;
- vdso_ts->nsec = nsec;
+ nsec = tk->tkr_mono.xtime_nsec;
+ nsec += (u64)tk->monotonic_to_aux.tv_nsec << tk->tkr_mono.shift;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec,
+ (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
}
__arch_update_vdso_clock(vc);
diff --git a/lib/bitmap.c b/lib/bitmap.c
index ed685127a107..d1cb8a507c60 100644
--- a/lib/bitmap.c
+++ b/lib/bitmap.c
@@ -308,6 +308,23 @@ bool __bitmap_intersects(const unsigned long *bitmap1,
}
EXPORT_SYMBOL(__bitmap_intersects);
+bool __bitmap_intersects_and(const unsigned long *bitmap1,
+ const unsigned long *bitmap2,
+ const unsigned long *bitmap3, unsigned int bits)
+{
+ unsigned int k, lim = bits / BITS_PER_LONG;
+
+ for (k = 0; k < lim; ++k)
+ if (bitmap1[k] & bitmap2[k] & bitmap3[k])
+ return true;
+
+ if (bits % BITS_PER_LONG)
+ if ((bitmap1[k] & bitmap2[k] & bitmap3[k]) & BITMAP_LAST_WORD_MASK(bits))
+ return true;
+ return false;
+}
+EXPORT_SYMBOL(__bitmap_intersects_and);
+
bool __bitmap_subset(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int bits)
{
diff --git a/lib/crc/x86/crc-pclmul-template.h b/lib/crc/x86/crc-pclmul-template.h
index 02744831c6fa..893119bb7c07 100644
--- a/lib/crc/x86/crc-pclmul-template.h
+++ b/lib/crc/x86/crc-pclmul-template.h
@@ -27,16 +27,14 @@ DEFINE_STATIC_CALL(prefix##_pclmul, prefix##_pclmul_sse)
static inline bool have_vpclmul(void)
{
return boot_cpu_has(X86_FEATURE_VPCLMULQDQ) &&
- boot_cpu_has(X86_FEATURE_AVX2) &&
- cpu_has_xfeatures(XFEATURE_MASK_YMM, NULL);
+ boot_cpu_has(X86_FEATURE_AVX2);
}
static inline bool have_avx512(void)
{
return boot_cpu_has(X86_FEATURE_AVX512BW) &&
boot_cpu_has(X86_FEATURE_AVX512VL) &&
- !boot_cpu_has(X86_FEATURE_PREFER_YMM) &&
- cpu_has_xfeatures(XFEATURE_MASK_AVX512, NULL);
+ !boot_cpu_has(X86_FEATURE_PREFER_YMM);
}
/*
diff --git a/lib/crypto/x86/blake2s.h b/lib/crypto/x86/blake2s.h
index f8eed6cb042e..0f7c51f055c8 100644
--- a/lib/crypto/x86/blake2s.h
+++ b/lib/crypto/x86/blake2s.h
@@ -55,8 +55,6 @@ static void blake2s_mod_init_arch(void)
if (boot_cpu_has(X86_FEATURE_AVX) &&
boot_cpu_has(X86_FEATURE_AVX2) &&
boot_cpu_has(X86_FEATURE_AVX512F) &&
- boot_cpu_has(X86_FEATURE_AVX512VL) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM |
- XFEATURE_MASK_AVX512, NULL))
+ boot_cpu_has(X86_FEATURE_AVX512VL))
static_branch_enable(&blake2s_use_avx512);
}
diff --git a/lib/crypto/x86/chacha.h b/lib/crypto/x86/chacha.h
index 10cf8f1c569d..c79562aac56b 100644
--- a/lib/crypto/x86/chacha.h
+++ b/lib/crypto/x86/chacha.h
@@ -165,8 +165,7 @@ static void chacha_mod_init_arch(void)
static_branch_enable(&chacha_use_simd);
if (boot_cpu_has(X86_FEATURE_AVX) &&
- boot_cpu_has(X86_FEATURE_AVX2) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL)) {
+ boot_cpu_has(X86_FEATURE_AVX2)) {
static_branch_enable(&chacha_use_avx2);
if (boot_cpu_has(X86_FEATURE_AVX512VL) &&
diff --git a/lib/crypto/x86/nh.h b/lib/crypto/x86/nh.h
index 83361c2e9783..342636dcb750 100644
--- a/lib/crypto/x86/nh.h
+++ b/lib/crypto/x86/nh.h
@@ -37,9 +37,7 @@ static void nh_mod_init_arch(void)
{
if (boot_cpu_has(X86_FEATURE_XMM2)) {
static_branch_enable(&have_sse2);
- if (boot_cpu_has(X86_FEATURE_AVX2) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- NULL))
+ if (boot_cpu_has(X86_FEATURE_AVX2))
static_branch_enable(&have_avx2);
}
}
diff --git a/lib/crypto/x86/poly1305.h b/lib/crypto/x86/poly1305.h
index ee92e3740a78..b061b9926fa5 100644
--- a/lib/crypto/x86/poly1305.h
+++ b/lib/crypto/x86/poly1305.h
@@ -143,15 +143,12 @@ static void poly1305_emit(const struct poly1305_state *ctx,
#define poly1305_mod_init_arch poly1305_mod_init_arch
static void poly1305_mod_init_arch(void)
{
- if (boot_cpu_has(X86_FEATURE_AVX) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL))
+ if (boot_cpu_has(X86_FEATURE_AVX))
static_branch_enable(&poly1305_use_avx);
- if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL))
+ if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2))
static_branch_enable(&poly1305_use_avx2);
if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_AVX2) &&
boot_cpu_has(X86_FEATURE_AVX512F) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM | XFEATURE_MASK_AVX512, NULL) &&
/* Skylake downclocks unacceptably much when using zmm, but later generations are fast. */
boot_cpu_data.x86_vfm != INTEL_SKYLAKE_X)
static_branch_enable(&poly1305_use_avx512);
diff --git a/lib/crypto/x86/sha1.h b/lib/crypto/x86/sha1.h
index c48a0131fd12..6aff433466e7 100644
--- a/lib/crypto/x86/sha1.h
+++ b/lib/crypto/x86/sha1.h
@@ -59,9 +59,7 @@ static void sha1_mod_init_arch(void)
{
if (boot_cpu_has(X86_FEATURE_SHA_NI)) {
static_call_update(sha1_blocks_x86, sha1_blocks_ni);
- } else if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- NULL) &&
- boot_cpu_has(X86_FEATURE_AVX)) {
+ } else if (boot_cpu_has(X86_FEATURE_AVX)) {
if (boot_cpu_has(X86_FEATURE_AVX2) &&
boot_cpu_has(X86_FEATURE_BMI1) &&
boot_cpu_has(X86_FEATURE_BMI2))
diff --git a/lib/crypto/x86/sha256.h b/lib/crypto/x86/sha256.h
index 0ee69d8e39fe..e98ffdaf4b14 100644
--- a/lib/crypto/x86/sha256.h
+++ b/lib/crypto/x86/sha256.h
@@ -104,9 +104,7 @@ static void sha256_mod_init_arch(void)
boot_cpu_has(X86_FEATURE_PHE_EN) &&
boot_cpu_data.x86 >= 0x07) {
static_call_update(sha256_blocks_x86, sha256_blocks_phe);
- } else if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM,
- NULL) &&
- boot_cpu_has(X86_FEATURE_AVX)) {
+ } else if (boot_cpu_has(X86_FEATURE_AVX)) {
if (boot_cpu_has(X86_FEATURE_AVX2) &&
boot_cpu_has(X86_FEATURE_BMI2))
static_call_update(sha256_blocks_x86,
diff --git a/lib/crypto/x86/sha512.h b/lib/crypto/x86/sha512.h
index 0213c70cedd0..4e177b4606bd 100644
--- a/lib/crypto/x86/sha512.h
+++ b/lib/crypto/x86/sha512.h
@@ -37,8 +37,7 @@ static void sha512_blocks(struct sha512_block_state *state,
#define sha512_mod_init_arch sha512_mod_init_arch
static void sha512_mod_init_arch(void)
{
- if (cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL) &&
- boot_cpu_has(X86_FEATURE_AVX)) {
+ if (boot_cpu_has(X86_FEATURE_AVX)) {
if (boot_cpu_has(X86_FEATURE_AVX2) &&
boot_cpu_has(X86_FEATURE_BMI2))
static_call_update(sha512_blocks_x86,
diff --git a/lib/crypto/x86/sm3.h b/lib/crypto/x86/sm3.h
index 3834780f2f6a..e06d4a22e4fa 100644
--- a/lib/crypto/x86/sm3.h
+++ b/lib/crypto/x86/sm3.h
@@ -33,7 +33,6 @@ static void sm3_blocks(struct sm3_block_state *state,
#define sm3_mod_init_arch sm3_mod_init_arch
static void sm3_mod_init_arch(void)
{
- if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_BMI2) &&
- cpu_has_xfeatures(XFEATURE_MASK_SSE | XFEATURE_MASK_YMM, NULL))
+ if (boot_cpu_has(X86_FEATURE_AVX) && boot_cpu_has(X86_FEATURE_BMI2))
static_call_update(sm3_blocks_x86, sm3_blocks_avx);
}
diff --git a/lib/raid/xor/Makefile b/lib/raid/xor/Makefile
index 9b0fad459cdb..e1e3455c219d 100644
--- a/lib/raid/xor/Makefile
+++ b/lib/raid/xor/Makefile
@@ -31,7 +31,7 @@ xor-$(CONFIG_SPARC32) += sparc/xor-sparc32.o
xor-$(CONFIG_SPARC64) += sparc/xor-sparc64.o sparc/xor-sparc64-glue.o
xor-$(CONFIG_S390) += s390/xor.o
xor-$(CONFIG_X86_32) += x86/xor-avx.o x86/xor-sse.o x86/xor-mmx.o
-xor-$(CONFIG_X86_64) += x86/xor-avx.o x86/xor-sse.o
+xor-$(CONFIG_X86_64) += x86/xor-avx.o x86/xor-sse.o x86/xor-avx512.o
obj-y += tests/
CFLAGS_xor-neon.o += $(CC_FLAGS_FPU) -I$(src)/$(SRCARCH)
diff --git a/lib/raid/xor/x86/xor-avx512.c b/lib/raid/xor/x86/xor-avx512.c
new file mode 100644
index 000000000000..c11d83441875
--- /dev/null
+++ b/lib/raid/xor/x86/xor-avx512.c
@@ -0,0 +1,122 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+/*
+ * AVX-512 optimized implementation of xor_gen()
+ *
+ * Copyright 2026 Google LLC
+ */
+
+#include <linux/types.h>
+#include <asm/fpu/api.h>
+#include "xor_impl.h"
+#include "xor_arch.h"
+
+/*
+ * Implementation notes:
+ *
+ * Unrolling by the number of buffers (2-5) is very important.
+ *
+ * Unrolling by length is less important, especially when using register-indexed
+ * addressing with negative indices from the end of the buffers. That approach
+ * results in just two loop control instructions being needed per iteration,
+ * regardless of the number of buffers.
+ *
+ * In fact, benchmarks showed that the 2 and 3 buffer cases require only 2x
+ * unrolling by length, while the 4 and 5 buffer cases don't require any
+ * unrolling by length. Benchmarks also showed that the register-indexed
+ * addressing isn't a bottleneck either; i.e., we can't do any better by
+ * incrementing the pointers as we go along, even with more unrolling.
+ */
+
+static void xor_avx512_2(long bytes, u8 *p1, const u8 *p2)
+{
+ long i = -bytes;
+
+ asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n"
+ "vmovdqa64 64(%1,%0), %%zmm1\n"
+ "vpxorq (%2,%0), %%zmm0, %%zmm0\n"
+ "vpxorq 64(%2,%0), %%zmm1, %%zmm1\n"
+ "vmovdqa64 %%zmm0, (%1,%0)\n"
+ "vmovdqa64 %%zmm1, 64(%1,%0)\n"
+ "add $128, %0\n"
+ "jnz 1b\n"
+ : "+&r"(i)
+ : "r"(p1 + bytes), "r"(p2 + bytes)
+ : "memory", "cc");
+}
+
+static void xor_avx512_3(long bytes, u8 *p1, const u8 *p2, const u8 *p3)
+{
+ long i = -bytes;
+
+ asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n"
+ "vmovdqa64 64(%1,%0), %%zmm1\n"
+ "vmovdqa64 (%2,%0), %%zmm2\n"
+ "vmovdqa64 64(%2,%0), %%zmm3\n"
+ "vpternlogq $0x96, (%3,%0), %%zmm2, %%zmm0\n"
+ "vpternlogq $0x96, 64(%3,%0), %%zmm3, %%zmm1\n"
+ "vmovdqa64 %%zmm0, (%1,%0)\n"
+ "vmovdqa64 %%zmm1, 64(%1,%0)\n"
+ "add $128, %0\n"
+ "jnz 1b\n"
+ : "+&r"(i)
+ : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes)
+ : "memory", "cc");
+}
+
+static void xor_avx512_4(long bytes, u8 *p1, const u8 *p2, const u8 *p3,
+ const u8 *p4)
+{
+ long i = -bytes;
+
+ asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n"
+ "vmovdqa64 (%2,%0), %%zmm1\n"
+ "vpxorq (%3,%0), %%zmm0, %%zmm0\n"
+ "vpternlogq $0x96, (%4,%0), %%zmm1, %%zmm0\n"
+ "vmovdqa64 %%zmm0, (%1,%0)\n"
+ "add $64, %0\n"
+ "jnz 1b\n"
+ : "+&r"(i)
+ : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes),
+ "r"(p4 + bytes)
+ : "memory", "cc");
+}
+
+static void xor_avx512_5(long bytes, u8 *p1, const u8 *p2, const u8 *p3,
+ const u8 *p4, const u8 *p5)
+{
+ long i = -bytes;
+
+ asm volatile("1: vmovdqa64 (%1,%0), %%zmm0\n"
+ "vmovdqa64 (%2,%0), %%zmm1\n"
+ "vpternlogq $0x96, (%3,%0), %%zmm1, %%zmm0\n"
+ "vmovdqa64 (%4,%0), %%zmm1\n"
+ "vpternlogq $0x96, (%5,%0), %%zmm1, %%zmm0\n"
+ "vmovdqa64 %%zmm0, (%1,%0)\n"
+ "add $64, %0\n"
+ "jnz 1b\n"
+ : "+&r"(i)
+ : "r"(p1 + bytes), "r"(p2 + bytes), "r"(p3 + bytes),
+ "r"(p4 + bytes), "r"(p5 + bytes)
+ : "memory", "cc");
+}
+
+DO_XOR_BLOCKS(avx512_inner, xor_avx512_2, xor_avx512_3, xor_avx512_4,
+ xor_avx512_5);
+
+/*
+ * Preconditions: bytes is a nonzero multiple of 512, and all buffers are
+ * 64-byte aligned.
+ */
+static void xor_gen_avx512(void *dest, void **srcs, unsigned int src_cnt,
+ unsigned int bytes)
+{
+ kernel_fpu_begin();
+ xor_gen_avx512_inner(dest, srcs, src_cnt, bytes);
+ asm volatile("vzeroupper");
+ kernel_fpu_end();
+}
+
+struct xor_block_template xor_block_avx512 = {
+ .name = "avx512",
+ .xor_gen = xor_gen_avx512,
+};
diff --git a/lib/raid/xor/x86/xor_arch.h b/lib/raid/xor/x86/xor_arch.h
index 99fe85a213c6..ed5921d2e2aa 100644
--- a/lib/raid/xor/x86/xor_arch.h
+++ b/lib/raid/xor/x86/xor_arch.h
@@ -6,22 +6,31 @@ extern struct xor_block_template xor_block_p5_mmx;
extern struct xor_block_template xor_block_sse;
extern struct xor_block_template xor_block_sse_pf64;
extern struct xor_block_template xor_block_avx;
+extern struct xor_block_template xor_block_avx512;
-/*
- * When SSE is available, use it as it can write around L2. We may also be able
- * to load into the L1 only depending on how the cpu deals with a load to a line
- * that is being prefetched.
- *
- * When AVX2 is available, force using it as it is better by all measures.
- *
- * 32-bit without MMX can fall back to the generic routines.
- */
static __always_inline void __init arch_xor_init(void)
{
- if (boot_cpu_has(X86_FEATURE_AVX) &&
- boot_cpu_has(X86_FEATURE_OSXSAVE)) {
+ if (IS_ENABLED(CONFIG_X86_64) && boot_cpu_has(X86_FEATURE_AVX512F) &&
+ !boot_cpu_has(X86_FEATURE_PREFER_YMM)) {
+ /*
+ * Use the AVX-512 code on CPUs that support AVX-512 without
+ * overly-eager downclocking. On such CPUs the AVX-512 code
+ * should always work at least as well as the AVX code, so
+ * runtime selection is unnecessary.
+ *
+ * The AVX-512 code can work on X86_32. However, due to lack of
+ * use case for that, for now it's built only for X86_64.
+ */
+ xor_force(&xor_block_avx512);
+ } else if (boot_cpu_has(X86_FEATURE_AVX)) {
+ /* AVX will be the best; no need to try others. */
xor_force(&xor_block_avx);
} else if (IS_ENABLED(CONFIG_X86_64) || boot_cpu_has(X86_FEATURE_XMM)) {
+ /*
+ * When SSE is available, use it as it can write around L2. We
+ * may also be able to load into the L1 only depending on how
+ * the cpu deals with a load to a line that is being prefetched.
+ */
xor_register(&xor_block_sse);
xor_register(&xor_block_sse_pf64);
} else if (boot_cpu_has(X86_FEATURE_MMX)) {
diff --git a/lib/vdso/gettimeofday.c b/lib/vdso/gettimeofday.c
index f7a591aba59f..ef4dcc614489 100644
--- a/lib/vdso/gettimeofday.c
+++ b/lib/vdso/gettimeofday.c
@@ -285,6 +285,7 @@ __cvdso_clock_gettime_common(const struct vdso_time_data *vd, clockid_t clock,
* Convert the clockid to a bitmask and use it to check which
* clocks are handled in the VDSO directly.
*/
+ BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk));
msk = 1U << clock;
if (likely(msk & VDSO_HRES))
vc = &vc[CS_HRES_COARSE];
@@ -438,6 +439,7 @@ bool __cvdso_clock_getres_common(const struct vdso_time_data *vd, clockid_t cloc
* Convert the clockid to a bitmask and use it to check which
* clocks are handled in the VDSO directly.
*/
+ BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk));
msk = 1U << clock;
if (msk & (VDSO_HRES | VDSO_RAW)) {
/*
diff --git a/net/core/pktgen.c b/net/core/pktgen.c
index a89cb0760821..d63eeadddfcc 100644
--- a/net/core/pktgen.c
+++ b/net/core/pktgen.c
@@ -2344,11 +2344,11 @@ static void spin(struct pktgen_dev *pkt_dev, ktime_t spin_until)
set_current_state(TASK_INTERRUPTIBLE);
hrtimer_sleeper_start_expires(&t, HRTIMER_MODE_ABS);
- if (likely(t.task))
+ if (likely(hrtimer_sleeper_task_get(&t)))
schedule();
hrtimer_cancel(&t.timer);
- } while (t.task && pkt_dev->running && !signal_pending(current));
+ } while (hrtimer_sleeper_task_get(&t) && pkt_dev->running && !signal_pending(current));
__set_current_state(TASK_RUNNING);
end_time = ktime_get();
}
diff --git a/tools/objtool/Documentation/klp-test-design.txt b/tools/objtool/Documentation/klp-test-design.txt
new file mode 100644
index 000000000000..2082c277197f
--- /dev/null
+++ b/tools/objtool/Documentation/klp-test-design.txt
@@ -0,0 +1,286 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+======================================
+Design of the objtool klp test harness
+======================================
+
+tools/objtool/tests/ holds unit tests for the klp subcommands of
+objtool -- ``klp checksum``, ``klp diff``, ``klp post-link`` and
+``--klp-symids`` -- which together turn two builds of the kernel into a
+livepatch module.
+
+This document explains how the harness is built and why. For the rules to
+follow when adding a test, see klp-write-tests.txt.
+
+
+TL;DR
+=====
+
+One run covers one compiler and one architecture; CI runs the combinations.
+Build objtool first -- it needs libelf and libxxhash -- and the same ARCH is
+used for both steps.
+
+Natively, with gcc::
+
+ make -C tools/objtool
+ make -C tools/objtool tests
+
+Natively, with clang -- LLVM=1 additionally selects the LLVM binutils::
+
+ CC=clang make -C tools/objtool tests
+ LLVM=1 make -C tools/objtool tests
+
+Cross, with gcc -- an arm64 host running the x86 tests::
+
+ ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool
+ ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool tests
+
+Cross, with clang. It defaults to the host triple however it is invoked, so
+--target= is what makes it emit x86; OBJCOPY is needed because BFD's is
+usually built for the host's target alone::
+
+ ARCH=x86_64 make -C tools/objtool
+ ARCH=x86_64 CC="clang --target=x86_64-linux-gnu" OBJCOPY=llvm-objcopy \
+ make -C tools/objtool tests
+
+A run ends with a totals line; anything other than fail:0 is a real result::
+
+ # pass:48 fail:0 static-skip:1 probe-skip:0 xfail:0 xpass:0
+
+Useful extras::
+
+ tools/objtool/tests/run-tests.sh basic # one test, by name
+ tools/objtool/tests/run-tests.sh --keep basic # and keep what it built
+
+
+Why unit tests are possible at all
+==================================
+
+klp-build is a pipeline: build the kernel twice, checksum both, diff them,
+link the result. Testing that end to end means two kernel builds per case,
+which is too slow to run often and too heavy to keep in the tree.
+
+Three properties make a much cheaper test possible.
+
+**objtool has no configuration-dependent logic.** It never reads ``.config``.
+Every ``CONFIG_`` string in its source is a comment or one error message, and
+its only build-time conditionals are driven by host libraries and the target
+architecture. Configuration reaches objtool through exactly two channels: the
+``objtool-args-$(CONFIG_*)`` lines in scripts/Makefile.lib, and the
+contents of the object handed to it.
+
+**The klp subcommands use none of the first channel.** Of objtool's options
+they consult three -- ``checksum``, ``debug_checksum``, ``dryrun`` -- all from
+their own command line. So klp behaviour varies with configuration *only*
+through the input object.
+
+**Therefore a test can reproduce any configuration's behaviour by reproducing
+its input.** Compile a small freestanding fixture with the flags that
+configuration would have used, and objtool cannot tell the difference. No
+kernel, no ``.config``, no object cache.
+
+The whole suite runs in a few seconds.
+
+
+Shape of a test
+===============
+
+Each test compiles one fixture twice -- once plain, once with ``-DPATCHED`` --
+runs ``klp checksum`` over both, diffs them, and asserts on properties of the
+output object::
+
+ . "$(dirname "$0")/../lib.sh"
+
+ setup
+ build_pair basic.c
+
+ assert_input_symbol changed
+ run_diff
+
+ assert_patched changed
+ assert_not_patched untouched
+
+ pass "changed function cloned, unchanged function left alone"
+
+Assertions check properties, never recorded output. Codegen varies between
+compilers and versions, so a golden file would report churn rather than
+regressions.
+
+
+Layout
+======
+
+::
+
+ tools/objtool/tests/
+ lib.sh the harness: everything a test may call
+ run-tests.sh selects, runs and classifies
+ generic/
+ test-*.sh
+ fixtures/*.c
+ x86/
+ test-*.sh
+ fixtures/*.c
+
+Which architecture a test is for is expressed by where it lives. The runner
+executes ``generic/`` plus the directory matching this architecture, so a test
+which cannot apply is not run rather than running in order to report that it
+did not. There is no ``x86_only`` helper, and no lookup letting an
+architecture fixture shadow a generic one: an architecture-specific test
+carries its own fixtures.
+
+Compilers cannot be expressed the same way, because CI varies ``CC`` over the
+same tree. A compiler requirement stays a declaration inside the test
+(``gcc_only``, ``clang_only``).
+
+
+The environment is established once
+===================================
+
+Sourcing lib.sh runs ``klp_preflight``, which checks that objtool
+exists and has klp support, that ``$CC`` works, that the binutils are present,
+and which architecture this is. The answers are exported, so:
+
+* ``run-tests.sh`` sources lib.sh too, and therefore knows the
+ architecture before it chooses which tests to run;
+* each test inherits the answers rather than repeating the work;
+* a test run on its own establishes them for itself.
+
+Preflight answers only whether the suite can run at all. A suite which cannot
+run must not exit 0 looking like one which passed, so a missing objtool fails
+the whole run with a TAP ``Bail out!`` rather than skipping each test in turn.
+What a *particular* compiler can do is a different question, left to the test
+which cares.
+
+
+Outcomes
+========
+
+Output is TAP. The distinction the harness cares most about is between kinds
+of skip, because a skip is how a suite quietly stops testing anything:
+
+``declared``
+ The test said in advance it does not apply -- ``gcc_only`` on a clang run.
+ Expected indefinitely.
+
+``probe``
+ The construct did not turn up in the built object this time. Weaker: one
+ which becomes permanent is a fixture that has stopped testing anything.
+
+``undeclared``
+ Counted as a **failure**. A test which gives up for a reason it never
+ declared is a hole, not an outcome.
+
+``xfail``/``xpass`` come with them, so a known failure is reported rather than
+commented out, and one which starts passing says so instead of going quietly
+green.
+
+The runner classifies the TAP result line, not everything a test printed:
+objtool warns on stderr and that output is captured, so a stray line ahead of
+the result would otherwise leave the exit status to decide -- and an expected
+failure exits 0.
+
+A run ends with a totals line::
+
+ # pass:48 fail:0 static-skip:1 probe-skip:0 xfail:0 xpass:0
+
+and reports what it left out::
+
+ # not run: 5 tests in x86/ (this run is arm64)
+
+
+Working directories
+===================
+
+A run gets one directory; each test gets a subdirectory of it, mirroring the
+source layout::
+
+ /tmp/klp-tests.XXXXXXXX/
+ generic/test-basic/{orig.o,patched.o,out.o,Module.symvers,...}
+ x86/test-kcfi/...
+
+By default (``KEEP=failed``) only failing tests keep their directories; the
+runner reports where they are. ``KEEP=all`` keeps every test's directory;
+``KEEP=none`` removes them all. The runner ``rmdir``s the run directory when
+it is empty -- which fails if anything was left behind unexpectedly, so a test
+which dies without cleaning up is reported rather than silently leaking.
+
+
+Running
+=======
+
+::
+
+ make -C tools/objtool # needs libelf and libxxhash
+ make -C tools/objtool tests
+
+ CC=clang make -C tools/objtool tests # the other toolchain
+ LLVM=1 make -C tools/objtool tests # and its binutils too
+
+ make -C tools/objtool tests KEEP=all # keep every test's workdir
+ make -C tools/objtool tests KEEP=none # remove all workdirs
+
+ tools/objtool/tests/run-tests.sh basic # one test; failures kept by default
+
+A run covers one compiler and one architecture; CI runs the combinations.
+
+Cross-compiled runs
+-------------------
+
+objtool klp is built only where ARCH_HAS_KLP is set, which today means x86 --
+so an arm64 machine cannot run any of this natively. It can run all of it
+cross, because objtool is a host tool that only reads and rewrites ELF, and
+the tests only compile fixtures and inspect the objects. Nothing has to
+execute target code.
+
+::
+
+ ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool
+ ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- make -C tools/objtool tests
+
+objtool itself stays a native binary: it is built with HOSTCC, not CC, so
+setting a cross compiler cannot produce one the host is unable to run. ARCH
+selects both the objtool target and the directory of tests to run.
+
+clang needs telling, since it defaults to the host triple however it is
+invoked. LLVM=1 with CROSS_COMPILE does that for you -- the --target= it
+derives is forwarded to the tests -- and the fixtures include no kernel
+headers, so no sysroot is needed:
+
+::
+
+ ARCH=x86_64 CROSS_COMPILE=x86_64-linux-gnu- LLVM=1 \
+ make -C tools/objtool tests
+
+Naming the compiler by hand works too, and is what to reach for when the
+target triple is not the one CROSS_COMPILE implies:
+
+::
+
+ ARCH=x86_64 CC="clang --target=x86_64-linux-gnu" \
+ OBJCOPY=llvm-objcopy make -C tools/objtool tests
+
+CROSS_COMPILE picks the binutils, and each can be overridden on its own.
+readelf reads any target and rarely needs overriding; BFD's objcopy is usually
+built for the host's alone, hence OBJCOPY=llvm-objcopy above, or install
+binutils-multiarch.
+
+Either readelf will do. The assertions read readelf's output, and the two
+spell some of it differently -- GNU prints "OS [0xff20]" for SHN_LIVEPATCH
+where llvm-readelf prints "OS[0xff20]" -- so they accept both.
+
+Getting this wrong is easy and the harness refuses rather than producing a
+misleading result. "CC=clang ARCH=x86_64" alone selects the x86 tests and
+then builds arm64 objects; preflight compiles a probe object, hands it to
+objtool, and stops the run if they disagree about the architecture, or if
+ARCH does not match what the compiler emits.
+
+
+What this does not cover
+========================
+
+These are unit tests for objtool's klp subcommands. They do not build a
+kernel, do not run scripts/livepatch/klp-build, and do not load a
+livepatch. Behaviour which only appears when the kernel applies a patch --
+the module loader refusing a relocation, late module patching ordering -- has
+to be tested by booting, and is out of scope here.
diff --git a/tools/objtool/Documentation/klp-write-tests.txt b/tools/objtool/Documentation/klp-write-tests.txt
new file mode 100644
index 000000000000..eb2cacaf5720
--- /dev/null
+++ b/tools/objtool/Documentation/klp-write-tests.txt
@@ -0,0 +1,266 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+=====================================
+Writing a test for objtool's klp code
+=====================================
+
+Instructions for adding a test to tools/objtool/tests/. Read
+klp-test-design.txt first if you need to know how the harness works; this
+document is the procedure and the rules.
+
+Two kinds of request bring you here:
+
+* *"write a test for commit <sha>"* -- a fix went in without one.
+* *"port the test at <location>, written against another harness"* -- a case
+ exists elsewhere and should live in tree.
+
+Both follow the same procedure.
+
+
+The one rule that matters
+=========================
+
+**A test is not finished until you have watched it fail.**
+
+Break the thing it guards -- revert the fix, or sabotage the exact line -- and
+confirm the test fails. Then restore and confirm it passes. A test that has
+never failed is not known to test anything, and this suite has produced
+several that passed against deliberately broken code:
+
+* an alternatives fixture whose empty entry pointed at its own end label rather
+ than the neighbour's replacement, so the bug it guarded made no difference;
+* a sympos fixture where symbol-table order and address order agreed, so
+ counting and reading the linked image gave the same answer;
+* a string fixture using a named ``char[]``, which never reached the
+ contents-hashing path because that keys on ``SHF_STRINGS``;
+* a static array the compiler proved constant, folded to zero, and emitted no
+ relocation for -- so the two builds were byte-identical.
+
+Every one looked correct. Say in the commit message how you verified, and if
+you could not isolate the behaviour to a single line, **say that too** rather
+than implying otherwise.
+
+
+Procedure
+=========
+
+1. **Read the fix.** What input reaches the broken line? What is observable
+ in the output object when it misbehaves -- a missing section, a relocation
+ naming the wrong symbol, an unchanged checksum, a rejected build? If
+ nothing is observable, stop and say so; see `When to give up`_.
+
+2. **Decide where it lives.** ``generic/`` unless the fixture needs
+ architecture-specific assembly or the behaviour is architecture-specific,
+ in which case ``x86/`` (or a new directory named for the architecture).
+
+3. **Write the fixture** in the same directory's ``fixtures/``. Reuse an
+ existing one if it already produces the shape; add a ``-D`` knob rather
+ than copying a fixture to change one line.
+
+4. **Write the test.** Assert the *premise* before the result -- see
+ `State the premise`_.
+
+5. **Verify by breaking the code.** Then restore.
+
+6. **Run the whole suite under both compilers**::
+
+ make -C tools/objtool tests
+ CC=clang make -C tools/objtool tests
+
+7. **Commit** the test and its fixture together, alone. One test per commit.
+
+
+Writing the fixture
+===================
+
+Fixtures are freestanding C. No kernel headers -- write out the kernel
+structure by hand if you need one, as the existing special-section fixtures do.
+
+Every fixture needs a ``.modinfo`` name, because klp diff reads the object's
+module name from it::
+
+ static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+Use ``MODNAME`` if the test needs to vary it. Note that most fixtures hardcode
+``vmlinux``: passing ``-DMODNAME`` to one that does silently does nothing and
+the test quietly becomes a vmlinux test.
+
+The patched build is selected with ``-DPATCHED``. For a fixture with several
+variants, gate each on both, so the original is always the baseline::
+
+ #if defined(PATCHED) && defined(WHICH_CALL)
+ r = callee_b(x);
+ #else
+ r = callee_a(x);
+ #endif
+
+and select one per build: ``build_pair foo.c -DWHICH_CALL``. Without the
+``defined(PATCHED)`` the flag applies to *both* builds and nothing differs.
+
+Traps that have bitten before
+-----------------------------
+
+* **The compiler optimises your fixture away.** A static never written is
+ proved constant, its reads folded, and no relocation emitted. Add a writer
+ the compiler cannot see through.
+* **String literals versus named arrays.** The contents-hashing path keys on
+ ``SHF_STRINGS``, which the compiler sets on the mergeable section a *literal*
+ lands in, not on a ``char[]`` given a section of its own.
+* **Per-function sections hide movement.** With the default
+ ``-ffunction-sections`` every function sits at offset 0 of its own section,
+ so nothing ever moves. A test about position needs
+ ``build_pair foo.c -fno-function-sections``.
+* **Special sections need boundaries.** Either an entsize on the section or an
+ ``ANNOTATE_DATA_SPECIAL`` annotation, or klp diff reports "missing special
+ section entsize or annotations". Their targets need real (global) symbols,
+ or it reports "failed to convert reloc sym".
+* **Prefer letting objtool generate what objtool generates.**
+ ``.static_call_sites``, ``.mcount_loc``, ``.ibt_endbr_seal`` and ORC come
+ from its check pass. Call ``run_objtool_check --mcount`` and let it build
+ them; a hand-written copy tests your reading of the format, not the format.
+
+ Write one by hand only when the test needs a shape objtool will not produce,
+ and say so in the fixture. generic/fixtures/static_call.c does: objtool
+ emits the site but not the ``ANNOTATE_DATA_SPECIAL`` that describes its
+ boundaries -- those come from the kernel's macros -- so a fixture which has
+ to vary whether the annotation is there writes both itself.
+
+
+Writing the test
+================
+
+Start from the shortest existing test, generic/test-basic.sh.
+
+State the premise
+-----------------
+
+A test which asserts only on the output passes when the compiler never emitted
+the construct in the first place, and reads as coverage it does not have. Say
+what the input must contain::
+
+ assert_input_section __jump_table # the fixture must produce it -> fail
+ require_input_section .kcfi_traps # this compiler may not -> skip
+
+Prefer ``assert_*``. Reach for ``require_*`` only where absence genuinely
+depends on compiler version or flags, and follow it with something
+unconditional so the test can never be entirely vacuous.
+
+Where a test would otherwise duplicate a sibling, assert what makes it
+different. ``test-jump-label-module-static-key`` checks that the key really is
+reached through its section symbol -- without that it is a second copy of
+``test-jump-label-module-key``.
+
+Assert both directions
+----------------------
+
+Check that the right thing happened *and* that the wrong thing did not. A klp
+diff which clones everything is as wrong as one which clones nothing::
+
+ assert_patched changed
+ assert_not_patched untouched
+
+Skips
+-----
+
+* ``gcc_only``/``clang_only`` -- a settled fact about the compiler. Declared,
+ so it reads as expected forever.
+* ``probe_skip`` -- this toolchain did not produce the construct. Include what
+ to do about it if there is anything::
+
+ probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_CC and THIN_LD to one"
+
+* A bare ``skip`` is **counted as a failure**. Never use it.
+
+Standing in for a kernel configuration
+--------------------------------------
+
+``FIXTURE_CFLAGS`` is what a fixture is built with. Since objtool reads no
+``.config``, changing these flags is how a test covers a configuration without
+building a kernel. Two ways:
+
+* trailing arguments to ``build_pair``/``build_one``, which win, and cover
+ anything expressible as a negation::
+
+ build_pair foo.c -fno-function-sections
+
+* otherwise assign ``FIXTURE_CFLAGS`` before building.
+
+Either way **say in a comment which kernel configuration the change stands in
+for**. A flag with no stated motive is indistinguishable from a mistake.
+
+
+What the test's comment must say
+================================
+
+The comment at the top is the test's justification. It should let a reader
+decide, without archaeology, whether a skip or a failure matters. Include:
+
+* **what breaks** in the running kernel if the behaviour regresses -- not the
+ mechanism, the consequence;
+* **why it is not caught otherwise**, which is usually "nothing fails at build
+ time";
+* **the fix commit** it guards, if there is one;
+* **anything load-bearing about the fixture** that is not obvious, especially
+ anything you got wrong first.
+
+That last point is the one people skip. If the fixture has to be built without
+per-function sections, or the static must not be named ``__warned``, or the key
+must be file-local -- write it down, or the next person will simplify it away.
+
+
+Porting a test from another harness
+===================================
+
+Read the original's *case*, not its code. The other harness probably builds a
+real kernel module; here you write freestanding C. A transliteration will
+usually test something else.
+
+* Work out which objtool behaviour the case exercises, then produce that shape
+ the cheapest way here.
+* Verify by breaking the code, exactly as for a new test -- a port is not
+ correct because the original was.
+* If the original names a fix commit, cite it.
+* Credit the source in the commit message with the trailers the original
+ carried, followed by your own.
+
+Sometimes the port shows the case is already covered, and sometimes it shows
+the case cannot be reproduced here. Both are results; report them rather than
+committing something that passes vacuously.
+
+
+When to give up
+===============
+
+Some behaviour cannot be reached from a compiled fixture. Say so, with what
+you tried, instead of committing a test that passes either way. Examples that
+were genuinely abandoned:
+
+* a memory leak -- needs valgrind, not an assertion on ELF;
+* ``mkstemp`` with long paths, and other I/O edge cases;
+* a NULL dereference reachable only through a debug path;
+* changes made redundant by a fallback: removing the code changes no output
+ because something else already handles the case;
+* a fix whose code has since been rewritten, so there is nothing left to
+ revert.
+
+Also stop when the behaviour depends on something outside the fixture's
+control -- an ELF library's handling of empty sections, or a compiler version's
+naming of anonymous data. A test which passes for you and skips for everyone
+else is worse than none.
+
+
+Checklist
+=========
+
+Before committing:
+
+* the test fails with the code broken, and passes with it fixed
+* the whole suite passes under **both** gcc and clang
+* the premise is asserted, not assumed
+* both directions are asserted where that applies
+* no bare ``skip``
+* the fixture is in the same directory as the test
+* the comment names the consequence, the fix commit, and anything load-bearing
+* the commit contains one test and its fixtures, and nothing else
+* the commit message says how you verified it
diff --git a/tools/objtool/Makefile b/tools/objtool/Makefile
index 4cc2e756af84..db99c4759b2c 100644
--- a/tools/objtool/Makefile
+++ b/tools/objtool/Makefile
@@ -153,6 +153,19 @@ clean: $(LIBSUBCMD)-clean
mrproper: clean
$(call QUIET_CLEAN, objtool) $(RM) $(OBJTOOL)
+# The kernel's own Makefile sets OBJCOPY, to llvm-objcopy under LLVM and to
+# $(CROSS_COMPILE)objcopy otherwise; tools/scripts/Makefile.include names only
+# LLVM_OBJCOPY and leaves OBJCOPY unset. Forwarding it unset would hand the
+# tests an empty string, and they would fall back to GNU objcopy even when
+# asked for an LLVM toolchain.
+OBJCOPY ?= $(if $(LLVM),$(LLVM_OBJCOPY),$(CROSS_COMPILE)objcopy)
+
+tests: $(OBJTOOL)
+ $(Q)OBJTOOL=$(abspath $(OBJTOOL)) ARCH=$(ARCH) CROSS_COMPILE=$(CROSS_COMPILE) \
+ CC='$(CC) $(CLANG_CROSS_FLAGS)' LD='$(LD)' \
+ READELF='$(READELF)' OBJCOPY='$(OBJCOPY)' \
+ KEEP=$(KEEP) $(srctree)/tools/objtool/tests/run-tests.sh
+
FORCE:
-.PHONY: clean mrproper FORCE
+.PHONY: clean mrproper tests FORCE
diff --git a/tools/objtool/tests/generic/fixtures/abs_and_addressable.c b/tools/objtool/tests/generic/fixtures/abs_and_addressable.c
new file mode 100644
index 000000000000..6392ff99af42
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/abs_and_addressable.c
@@ -0,0 +1,44 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Two constructs which appear all over the kernel and must not upset klp
+ * checksum or klp diff.
+ *
+ * An absolute symbol (SHN_ABS) has no section, so anything walking sym->sec
+ * without checking dereferences NULL. The kernel makes them with linker
+ * scripts and with .set in asm; VDSO and the fixed-address per-cpu bases are
+ * the usual sources.
+ *
+ * __ADDRESSABLE() emits a pointer into .discard.addressable purely to keep a
+ * symbol referenced. It is discarded at link time and means nothing to a
+ * livepatch, but the pointer is a relocation like any other and has to survive
+ * being looked at.
+ *
+ * Neither is the subject of the patch; the point is that their presence does
+ * not disturb the function that is.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/* SHN_ABS, referenced from code. */
+extern char abs_sym[];
+__asm__(".globl abs_sym\n"
+ ".set abs_sym, 0x1234\n");
+
+int helper(int x);
+int helper(int x) { return x + 1; }
+
+/* The shape of __ADDRESSABLE(helper). */
+__asm__(".pushsection .discard.addressable, \"aw\"\n"
+ ".balign 8\n"
+ ".quad helper\n"
+ ".popsection\n");
+
+int target(int x)
+{
+#ifdef PATCHED
+ return helper(x) + (int)(long)abs_sym + 1;
+#else
+ return helper(x) + (int)(long)abs_sym;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/basic.c b/tools/objtool/tests/generic/fixtures/basic.c
new file mode 100644
index 000000000000..811529e7bfb1
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/basic.c
@@ -0,0 +1,20 @@
+// SPDX-License-Identifier: GPL-2.0
+/* One changed function and one unchanged function. */
+
+/* klp diff takes the object's module name from .modinfo */
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int untouched(int x)
+{
+ return x * 3;
+}
+
+int changed(int x)
+{
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/changed_data.c b/tools/objtool/tests/generic/fixtures/changed_data.c
new file mode 100644
index 000000000000..b52461835444
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/changed_data.c
@@ -0,0 +1,16 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Data whose value differs between the two builds. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+#ifdef PATCHED
+int klp_test_data = 2;
+#else
+int klp_test_data = 1;
+#endif
+
+int target(int x)
+{
+ return x + klp_test_data;
+}
diff --git a/tools/objtool/tests/generic/fixtures/checksum_data.c b/tools/objtool/tests/generic/fixtures/checksum_data.c
new file mode 100644
index 000000000000..6310af5c02d3
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/checksum_data.c
@@ -0,0 +1,116 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Data objects whose checksums must move for reasons the raw bytes do not
+ * show.
+ *
+ * checksum_update_object() hashes a data symbol's length and its bytes, and
+ * then walks its relocations: a reference into a string section contributes
+ * the string's *contents*, and any other reference contributes the target
+ * symbol's name and the adjusted addend. So three changes that leave the
+ * object's own bytes identical still have to change its checksum:
+ *
+ * Each variant is selected by a -D on the patched build only, so the original
+ * is always the baseline:
+ *
+ * WHICH_FUNC the function pointer points somewhere else
+ * WHICH_STR the string pointer points at a different literal
+ * STR_CONTENT the string it points at is edited in place
+ * WHICH_SLOT the same array, at a different index: addend only
+ * WHICH_PRIV likewise, but a static, reached through its section symbol
+ *
+ * The last is the interesting one. Nothing in the pointer changes -- same
+ * section, same offset -- so a checksum that hashed only the relocation and
+ * not what it referred to would call the object unchanged, and the patched
+ * kernel would keep the old string.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int callee_a(int x);
+int callee_b(int x);
+int callee_a(int x) { return x + 1; }
+int callee_b(int x) { return x + 2; }
+
+/*
+ * String *literals*, not named arrays. The contents-hashing path keys on
+ * SHF_STRINGS, which the compiler sets on the mergeable .rodata.str1.1 a
+ * literal lands in and not on a named char[] given a section of its own. A
+ * fixture using the latter exercises the ordinary name-and-addend path and
+ * reports nothing when the text changes.
+ */
+#if defined(PATCHED) && defined(STR_CONTENT)
+#define MESSAGE "edited"
+#else
+#define MESSAGE "original"
+#endif
+
+/* A plain data object: only its own bytes decide the checksum. */
+#if defined(PATCHED) && defined(PLAIN_VALUE)
+int plain = 43;
+#else
+int plain = 42;
+#endif
+
+/*
+ * A .bss object, where length is the only thing there is to hash: the section
+ * has no data, so the bytes are skipped and only sym->len distinguishes this
+ * from an object of another size. An initialised array would not isolate it
+ * -- growing one changes the hashed bytes as well.
+ */
+#if defined(PATCHED) && defined(LONGER)
+char sized[4];
+#else
+char sized[2];
+#endif
+
+/*
+ * A reference into the middle of an array: same target symbol, different
+ * addend. Nothing else in the object changes, so this is the only way to see
+ * whether the addend is hashed at all.
+ */
+int slots[4];
+
+/*
+ * A file-local array. A reference to a static lands on its section symbol
+ * plus an offset, so the hash has to resolve that back to the underlying
+ * object before it has a name to hash at all -- a different code path from the
+ * global above, and one that silently contributes nothing when it fails.
+ */
+static int priv_slots[4];
+
+struct desc {
+ int (*fn)(int arg);
+ const char *str;
+ int *slot;
+ int *priv;
+};
+
+const struct desc descriptor = {
+#if defined(PATCHED) && defined(WHICH_FUNC)
+ .fn = callee_b,
+#else
+ .fn = callee_a,
+#endif
+#if defined(PATCHED) && defined(WHICH_STR)
+ .str = "a different literal",
+#else
+ .str = MESSAGE,
+#endif
+#if defined(PATCHED) && defined(WHICH_SLOT)
+ .slot = &slots[2],
+#else
+ .slot = &slots[1],
+#endif
+#if defined(PATCHED) && defined(WHICH_PRIV)
+ .priv = &priv_slots[3],
+#else
+ .priv = &priv_slots[1],
+#endif
+};
+
+int target(int x)
+{
+ return descriptor.fn(x) + plain + sized[0] + (int)descriptor.str[0] +
+ *descriptor.slot + *descriptor.priv;
+}
diff --git a/tools/objtool/tests/generic/fixtures/checksum_insn.c b/tools/objtool/tests/generic/fixtures/checksum_insn.c
new file mode 100644
index 000000000000..10f70a74a976
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/checksum_insn.c
@@ -0,0 +1,78 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Instruction operands whose change must move a function's checksum even
+ * though the instruction bytes themselves do not.
+ *
+ * checksum_update_insn() hashes the raw bytes and then, when the instruction
+ * carries a relocation, what that relocation refers to: a string section
+ * contributes the string's contents, anything else the target symbol's name
+ * and the adjusted addend. A reference to a static arrives as a section
+ * symbol and has to be resolved back to the object first.
+ *
+ * The bytes are identical in every case below -- a rel32 operand is zero in
+ * the object and supplied by the relocation -- so a checksum that stopped at
+ * the bytes would call all of these unchanged.
+ *
+ * Each variant applies to the patched build only:
+ *
+ * WHICH_CALL calls a different function
+ * STR_CONTENT passes a literal whose text was edited
+ * WHICH_SLOT reads a different index of a global array: addend only
+ * WHICH_PRIV the same, for a static, reached through its section symbol
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int callee_a(int x);
+int callee_b(int x);
+int sink(const char *s);
+
+int slots[4];
+
+/*
+ * A file-local array, plus a writer the compiler cannot see through. Without
+ * one it can prove the array is never written, folds every read to zero, and
+ * emits no relocation at all -- so the reference this is here to exercise does
+ * not exist.
+ */
+static int priv_slots[4];
+
+void set_priv(int i, int v);
+void set_priv(int i, int v)
+{
+ priv_slots[i] = v;
+}
+
+#if defined(PATCHED) && defined(STR_CONTENT)
+#define MESSAGE "edited"
+#else
+#define MESSAGE "original"
+#endif
+
+int target(int x)
+{
+ int r;
+
+#if defined(PATCHED) && defined(WHICH_CALL)
+ r = callee_b(x);
+#else
+ r = callee_a(x);
+#endif
+
+ r += sink(MESSAGE);
+
+#if defined(PATCHED) && defined(WHICH_SLOT)
+ r += slots[2];
+#else
+ r += slots[1];
+#endif
+
+#if defined(PATCHED) && defined(WHICH_PRIV)
+ r += priv_slots[3];
+#else
+ r += priv_slots[1];
+#endif
+
+ return r;
+}
diff --git a/tools/objtool/tests/generic/fixtures/checksum_position.c b/tools/objtool/tests/generic/fixtures/checksum_position.c
new file mode 100644
index 000000000000..4320bc220592
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/checksum_position.c
@@ -0,0 +1,42 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A function whose position in the section changes between the two builds,
+ * without the function itself changing.
+ *
+ * target() calls callee() twice, and a call within the same section needs no
+ * relocation: the displacement is in the instruction. It is that displacement
+ * which moves, and hashing those bytes makes the checksum move with it. Both
+ * must therefore share a section, which is why the test passes
+ * -fno-function-sections.
+ *
+ * What has to change is the distance between the two, and PATCHED changes it
+ * by aligning them rather than by inserting a function between them. Where a
+ * compiler puts an added function is its own business: gcc emits these in
+ * source order, so a function written between callee() and target() separates
+ * them, but clang emits target() immediately before callee() whatever the
+ * source says, and an added function lands ahead of both. That moves target()
+ * without moving it relative to callee(), the displacement comes out identical
+ * in both builds, and the test passes without having asked anything.
+ *
+ * Alignment moves the functions apart on both, and moves neither function's
+ * own instructions -- which is exactly the distinction under test.
+ */
+
+#ifdef PATCHED
+#define MOVED __attribute__((aligned(64)))
+#else
+#define MOVED
+#endif
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+__attribute__((noinline)) MOVED static int callee(int x)
+{
+ return x * 5 + 1;
+}
+
+__attribute__((noinline)) MOVED int target(int x)
+{
+ return callee(x) + callee(x + 1);
+}
diff --git a/tools/objtool/tests/generic/fixtures/checksum_skip.c b/tools/objtool/tests/generic/fixtures/checksum_skip.c
new file mode 100644
index 000000000000..973bdc10295d
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/checksum_skip.c
@@ -0,0 +1,47 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Symbols calculate_checksums() must not give an entry of their own.
+ *
+ * Three kinds are skipped, for two different reasons:
+ *
+ * zero-length there is nothing to hash, and an entry keyed on the
+ * symbol's address would collide with whatever really lives
+ * there.
+ * alias a second name for an address already checksummed.
+ * cold part hashed as part of its parent, which func_for_each_insn()
+ * walks into, so a separate entry would double-count it.
+ *
+ * An entry per address is the invariant: .discard.sym_checksum is looked up by
+ * the address a relocation points at, so two entries for one address make the
+ * lookup ambiguous and one of the two checksums unreachable.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/* Zero-length: an object symbol of size 0, in a section of its own. */
+extern char empty_marker[];
+__asm__(".pushsection .data.empty_marker,\"aw\",@progbits\n"
+ ".globl empty_marker\n"
+ ".type empty_marker, @object\n"
+ "empty_marker:\n"
+ ".size empty_marker, 0\n"
+ ".popsection\n");
+
+int real_function(int x);
+int real_function(int x)
+{
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
+
+/* Alias: a second name for real_function's address. */
+int alias_function(int x) __attribute__((alias("real_function")));
+
+int target(int x)
+{
+ return real_function(x) + alias_function(x) + (int)(long)empty_marker;
+}
diff --git a/tools/objtool/tests/generic/fixtures/cold_function.c b/tools/objtool/tests/generic/fixtures/cold_function.c
new file mode 100644
index 000000000000..f6d410983ce2
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/cold_function.c
@@ -0,0 +1,21 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Function the compiler may split into a hot part and a foo.cold part. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+static void __attribute__((cold, noinline)) slow_path(int x)
+{
+ __asm__ volatile("" :: "r"(x));
+}
+
+int target(int x)
+{
+ if (__builtin_expect(x < 0, 0))
+ slow_path(x);
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/cross_module.c b/tools/objtool/tests/generic/fixtures/cross_module.c
new file mode 100644
index 000000000000..c170bde0666f
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/cross_module.c
@@ -0,0 +1,25 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A function which calls out to another object. MODNAME selects which object
+ * this one is, so a test can make the caller a module and the callee's owner
+ * something else.
+ */
+
+#ifndef MODNAME
+#define MODNAME "vmlinux"
+#endif
+
+/* klp diff takes the object's module name from .modinfo */
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME;
+
+extern int other_mod_func(int x);
+
+int target(int x)
+{
+#ifdef PATCHED
+ return other_mod_func(x) + 2;
+#else
+ return other_mod_func(x) + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/data_alignment.c b/tools/objtool/tests/generic/fixtures/data_alignment.c
new file mode 100644
index 000000000000..900253dfb2cb
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/data_alignment.c
@@ -0,0 +1,29 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Data with an alignment stricter than its size.
+ *
+ * A cloned data section has to keep its sh_addralign. The kernel has plenty
+ * of data whose alignment is a correctness property rather than an
+ * optimisation -- per-CPU variables, anything touched by an aligned SSE move,
+ * cacheline-aligned locks -- and a clone that lands under-aligned faults or
+ * silently shares a cacheline it was written to avoid.
+ *
+ * The object is new in the patched build, so klp diff has to clone it rather
+ * than reference the kernel's copy.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+#ifdef PATCHED
+int aligned_data[2] __attribute__((aligned(64))) = { 1, 2 };
+#endif
+
+int target(int x)
+{
+#ifdef PATCHED
+ return x + aligned_data[0];
+#else
+ return x;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/function_removal.c b/tools/objtool/tests/generic/fixtures/function_removal.c
new file mode 100644
index 000000000000..d65ff604c2c5
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/function_removal.c
@@ -0,0 +1,25 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A function the patch deletes, along with its only caller's use of it. The
+ * original has a symbol which the patched object simply does not, so there is
+ * nothing to correlate it against.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+#ifndef PATCHED
+int going_away(int x)
+{
+ return x + 7;
+}
+#endif
+
+int caller(int x)
+{
+#ifdef PATCHED
+ return x + 1;
+#else
+ return going_away(x);
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/init_reference.c b/tools/objtool/tests/generic/fixtures/init_reference.c
new file mode 100644
index 000000000000..9e59bdf708d3
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/init_reference.c
@@ -0,0 +1,17 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Patched function referencing data in an .init section. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/* volatile so the read cannot be folded into a constant, leaving no reference */
+static volatile int init_only __attribute__((section(".init.data"), used)) = 5;
+
+int target(int x)
+{
+#ifdef PATCHED
+ return x + init_only + 1;
+#else
+ return x + init_only;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/jump_label.c b/tools/objtool/tests/generic/fixtures/jump_label.c
new file mode 100644
index 000000000000..ecd3c06912af
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/jump_label.c
@@ -0,0 +1,76 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Static branch in a patched function. The jump table entry is written out by
+ * hand, mirroring JUMP_TABLE_ENTRY(), so the fixture builds without kernel
+ * headers. The key is an STT_OBJECT; anything else is ignored by
+ * validate_special_section_klp_reloc().
+ *
+ * MODNAME selects whether the key is taken to belong to vmlinux or a module.
+ *
+ * NEW_KEY puts the whole static branch behind PATCHED, so the patch introduces
+ * one where the original had none -- a different question from patching code
+ * that already has a key, because the __jump_table entry itself is new.
+ *
+ * KEY_NAME renames the key. Two names are special to
+ * validate_special_section_klp_reloc(): a __tracepoint_* key and the
+ * __UNIQUE_ID_ddebug_* one pr_debug() generates are both unsupported in a
+ * module, but are disabled with a warning rather than rejected, because the
+ * kernel is full of them and refusing outright would make ordinary functions
+ * unpatchable.
+ *
+ * STATIC_KEY makes the key file-local. That changes the shape of the
+ * relocation rather than the meaning of the code: a reference to a static lands
+ * on the section symbol plus an addend, so the key has to be resolved from the
+ * section before it can be recognised as a key at all.
+ */
+
+#ifndef MODNAME
+#define MODNAME "vmlinux"
+#endif
+
+#ifndef KEY_NAME
+#define KEY_NAME klp_test_key
+#endif
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME;
+
+#ifdef STATIC_KEY
+static long KEY_NAME;
+#else
+long KEY_NAME;
+#endif
+
+int target(int x)
+{
+ int r = x;
+
+#if defined(NEW_KEY) && !defined(PATCHED)
+ /* The original has no static branch at all. */
+ return r + 1;
+#else
+ asm goto(
+ "1: nop\n\t"
+ ".pushsection __jump_table, \"aw\"\n\t"
+ ".balign 8\n\t"
+ "912:\n\t"
+ ".pushsection .discard.annotate_data, \"M\", @progbits, 8\n\t"
+ ".long 912b - ., 1\n\t"
+ ".popsection\n\t"
+ ".long 1b - ., %l[l_yes] - .\n\t"
+ ".quad %c0 - .\n\t"
+ ".popsection\n\t"
+ : : "i" (&KEY_NAME) : : l_yes);
+
+ r += 1;
+ goto out;
+l_yes:
+ r += 2;
+out:
+#endif
+#ifdef PATCHED
+ return r + 100;
+#else
+ return r;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/klp_funcs.c b/tools/objtool/tests/generic/fixtures/klp_funcs.c
new file mode 100644
index 000000000000..3f0d3e206cef
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/klp_funcs.c
@@ -0,0 +1,31 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Two changed functions and one untouched, so the patch's function list has a
+ * length worth checking and something that must not appear in it.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int first(int x)
+{
+#ifdef PATCHED
+ return x + 11;
+#else
+ return x + 1;
+#endif
+}
+
+int second(int x)
+{
+#ifdef PATCHED
+ return x + 22;
+#else
+ return x + 2;
+#endif
+}
+
+int third(int x)
+{
+ return x + 3;
+}
diff --git a/tools/objtool/tests/generic/fixtures/local_to_global.c b/tools/objtool/tests/generic/fixtures/local_to_global.c
new file mode 100644
index 000000000000..3c9eb9200ce5
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/local_to_global.c
@@ -0,0 +1,34 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A function which the patch changes from static to non-static, and a variable
+ * that goes the other way. The names are unchanged; only the binding moves.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/* noinline, or the static one is folded into its caller and has no symbol */
+#ifdef PATCHED
+__attribute__((noinline)) int flipped_up(int x) /* was static */
+#else
+__attribute__((noinline)) static int flipped_up(int x)
+#endif
+{
+ return x + 1;
+}
+
+#ifdef PATCHED
+static volatile int flipped_down = 5; /* was global */
+#else
+volatile int flipped_down = 5;
+#endif
+
+int caller(int x)
+{
+ flipped_down += x;
+#ifdef PATCHED
+ return flipped_up(x) + flipped_down + 2;
+#else
+ return flipped_up(x) + flipped_down + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/new_data.c b/tools/objtool/tests/generic/fixtures/new_data.c
new file mode 100644
index 000000000000..ac364622f12a
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/new_data.c
@@ -0,0 +1,23 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Data introduced by the patch. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+#ifdef PATCHED
+/*
+ * Not an arithmetic progression: { 1, 2, 3, 4 } indexed by x & 3 is something
+ * a compiler can compute instead of load, and then target() has no reference
+ * to the array and there is nothing for klp diff to carry.
+ */
+static const int klp_new_data[4] __attribute__((used)) = { 7, 3, 11, 5 };
+#endif
+
+int target(int x)
+{
+#ifdef PATCHED
+ return x + klp_new_data[x & 3];
+#else
+ return x;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/new_export_ref.c b/tools/objtool/tests/generic/fixtures/new_export_ref.c
new file mode 100644
index 000000000000..73210aacb4ae
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/new_export_ref.c
@@ -0,0 +1,35 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A reference which only exists in the patched build. The symbol has no twin
+ * in the original object, so what klp diff may do with it depends entirely on
+ * whether Module.symvers says it is exported, and by what.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+extern int newly_referenced(int x);
+
+/*
+ * A reference both builds have. When Module.symvers says a module exports
+ * this one, the original already depends on that module, which is what makes
+ * a new reference to it safe -- the loader will not let the patched module
+ * load without it. EXISTING_DEP leaves it out, for the case where there is
+ * no such dependency to inherit.
+ */
+extern int existing_dep(int x);
+
+int target(int x)
+{
+#ifdef EXISTING_DEP
+ int base = existing_dep(x);
+#else
+ int base = x;
+#endif
+
+#ifdef PATCHED
+ return newly_referenced(base);
+#else
+ return base + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/new_function.c b/tools/objtool/tests/generic/fixtures/new_function.c
new file mode 100644
index 000000000000..e7886eaaff3e
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/new_function.c
@@ -0,0 +1,21 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Function introduced by the patch. noinline keeps it from being folded. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+#ifdef PATCHED
+static __attribute__((noinline)) int klp_new_helper(int x)
+{
+ return x * 7;
+}
+#endif
+
+int target(int x)
+{
+#ifdef PATCHED
+ return klp_new_helper(x);
+#else
+ return x;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/no_modinfo.c b/tools/objtool/tests/generic/fixtures/no_modinfo.c
new file mode 100644
index 000000000000..e46374702b1a
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/no_modinfo.c
@@ -0,0 +1,11 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Deliberately has no .modinfo section. */
+
+int target(int x)
+{
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/special_section.c b/tools/objtool/tests/generic/fixtures/special_section.c
new file mode 100644
index 000000000000..d28c5541e337
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/special_section.c
@@ -0,0 +1,24 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Special section entry with no ANNOTATE_DATA_SPECIAL annotation and a local
+ * label at offset 0, the shape Clang produces for .kcfi_traps.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int target(int x)
+{
+ asm volatile(
+ "1:\n\t"
+ ".pushsection .kcfi_traps, \"a\"\n\t"
+ ".balign 4\n\t"
+ "trap_marker:\n\t"
+ ".long 1b - .\n\t"
+ ".popsection\n\t");
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/special_section_shared.c b/tools/objtool/tests/generic/fixtures/special_section_shared.c
new file mode 100644
index 000000000000..f54e24f862c6
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/special_section_shared.c
@@ -0,0 +1,31 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Two functions contribute to one special section; only one is patched. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int other(int x)
+{
+ asm volatile(
+ "2:\n\t"
+ ".pushsection .kcfi_traps, \"a\"\n\t"
+ ".balign 4\n\t"
+ ".long 2b - .\n\t"
+ ".popsection\n\t");
+ return x * 5;
+}
+
+int target(int x)
+{
+ asm volatile(
+ "1:\n\t"
+ ".pushsection .kcfi_traps, \"a\"\n\t"
+ ".balign 4\n\t"
+ ".long 1b - .\n\t"
+ ".popsection\n\t");
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/static_call.c b/tools/objtool/tests/generic/fixtures/static_call.c
new file mode 100644
index 000000000000..4a0c4c25321e
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/static_call.c
@@ -0,0 +1,59 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Static call site in a patched function, laid out by hand as for
+ * jump_label.c. MODNAME selects whether the key belongs to vmlinux or a
+ * module.
+ *
+ * objtool's check pass would emit the site, and klp-write-tests.txt says to
+ * let it. Not here: it does not emit the ANNOTATE_DATA_SPECIAL describing
+ * the entry boundaries -- in the kernel that comes from the static_call
+ * macros -- and NO_ANNOTATE below has to be able to take it away. A fixture
+ * which varies the annotation has to write the entry that goes with it.
+ *
+ * NO_ANNOTATE drops the ANNOTATE_DATA_SPECIAL block from the patched build,
+ * leaving .static_call_sites with no annotation to describe its entry
+ * boundaries. The section carries no entsize either, so klp diff has to fall
+ * back on the annotations it can still see -- and when the patched object is
+ * the only one that lost them, the two sides disagree about how the section is
+ * divided up.
+ *
+ * NEW_CALL puts the call site behind PATCHED, so the patch introduces one
+ * where the original had none. The .static_call_sites entry is then new, with
+ * nothing in the original to correlate it against.
+ */
+
+#ifndef MODNAME
+#define MODNAME "vmlinux"
+#endif
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=" MODNAME;
+
+long __SCK__klp_test_call;
+
+int target(int x)
+{
+#if defined(NEW_CALL) && !defined(PATCHED)
+ /* The original has no static call at all. */
+ return x + 1;
+#else
+ __asm__ volatile(
+ "1: nop\n\t"
+ ".pushsection .static_call_sites, \"aw\"\n\t"
+ ".balign 8\n\t"
+ "912:\n\t"
+#if !(defined(PATCHED) && defined(NO_ANNOTATE))
+ ".pushsection .discard.annotate_data, \"M\", @progbits, 8\n\t"
+ ".long 912b - ., 1\n\t"
+ ".popsection\n\t"
+#endif
+ ".long 1b - ., %c0 - .\n\t"
+ ".popsection\n\t"
+ :: "i" (&__SCK__klp_test_call));
+#endif
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/static_local.c b/tools/objtool/tests/generic/fixtures/static_local.c
new file mode 100644
index 000000000000..f2f025d00a39
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/static_local.c
@@ -0,0 +1,17 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Static local in a patched function. */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int target(int x)
+{
+ static int counter;
+
+ counter += 1;
+#ifdef PATCHED
+ return x + counter + 1;
+#else
+ return x + counter;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c b/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c
new file mode 100644
index 000000000000..cb4cdd7a496e
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c
@@ -0,0 +1,41 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Static locals of three kinds, in one patched function.
+ *
+ * Most static locals must be correlated, so the patched code keeps using the
+ * running kernel's copy. Two kinds must not:
+ *
+ * - anything in .data..once, the flag behind WARN_ONCE and friends. Sharing
+ * it would mean a patch inherits "already warned" from before the patch.
+ * - the well-known names the kernel generates for such things (__warned,
+ * __key, __func__, ...), which are per-instance by nature. gcc names them
+ * <var>.<id> and Clang <func>.<var>, so both spellings have to be caught.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int target(int x)
+{
+ /*
+ * A .data..once variable whose name is *not* on the list below, so
+ * only the section can disqualify it. Naming it __warned would let
+ * the name rule catch it and the section rule go untested.
+ */
+ static int once_flag __attribute__((section(".data..once")));
+ /* a never-correlate name, in an ordinary section */
+ static int __key;
+ /* and one that must be correlated */
+ static int ordinary;
+
+ if (!once_flag)
+ once_flag = 1;
+ __key += x;
+ ordinary += x;
+
+#ifdef PATCHED
+ return __key + ordinary + once_flag + 2;
+#else
+ return __key + ordinary + once_flag + 1;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/switch_rodata.c b/tools/objtool/tests/generic/fixtures/switch_rodata.c
new file mode 100644
index 000000000000..817ddac92d81
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/switch_rodata.c
@@ -0,0 +1,31 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A switch dense enough that Clang builds a jump table for it, in a section of
+ * its own: .rodata..Lswitch.table.<function>.
+ *
+ * The table belongs to the function and has to travel with it. It is named
+ * after the function but is not part of it, so klp diff has to associate the
+ * two rather than treating the table as unrelated data.
+ *
+ * The patch adds a case, which changes the table's contents and length.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+const char *status_to_string(unsigned int c)
+{
+ switch (c) {
+ case 0: return "idle";
+ case 1: return "running";
+ case 2: return "stopped";
+ case 3: return "error";
+ case 4: return "paused";
+ case 5: return "waiting";
+ case 6: return "starting";
+ case 7: return "stopping";
+#ifdef PATCHED
+ case 8: return "completed";
+#endif
+ }
+ return "unknown";
+}
diff --git a/tools/objtool/tests/generic/fixtures/symid_discarded.c b/tools/objtool/tests/generic/fixtures/symid_discarded.c
new file mode 100644
index 000000000000..573cc2d4474b
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/symid_discarded.c
@@ -0,0 +1,25 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Compiled twice and partially linked so the result has duplicate locals,
+ * which is what symid_needed() requires. dup_normal is in a live section,
+ * dup_discarded in one the vmlinux link throws away. DISCARDED_SEC selects
+ * which discarded section, since there is more than one and each was its own
+ * bug.
+ */
+
+#ifndef DISCARDED_SEC
+#define DISCARDED_SEC ".exitcall.exit"
+#endif
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+static int dup_normal = 1;
+
+static void *dup_discarded
+ __attribute__((section(DISCARDED_SEC), used)) = &dup_normal;
+
+int FUNC_NAME(void)
+{
+ return dup_normal + (dup_discarded != (void *)0);
+}
diff --git a/tools/objtool/tests/generic/fixtures/sympos_dup.c b/tools/objtool/tests/generic/fixtures/sympos_dup.c
new file mode 100644
index 000000000000..7eded9b12cfc
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/sympos_dup.c
@@ -0,0 +1,32 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A static whose name recurs in every translation unit that includes it.
+ * Compiled once for a single-copy object and twice, partially linked, for one
+ * with duplicates -- which is the only case where sympos is non-zero.
+ *
+ * FUNC_NAME keeps the referencing functions distinct so both get patched.
+ * Only the first copy carries .modinfo; two would be a second thing to
+ * disambiguate and is not what this fixture is about.
+ */
+
+#ifndef FUNC_NAME
+#define FUNC_NAME use_a
+#endif
+
+#ifndef NO_MODINFO
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+#endif
+
+/* volatile so it survives as an STT_OBJECT rather than being folded away */
+static volatile int dup_counter = 1;
+
+int FUNC_NAME(int x)
+{
+ dup_counter += x;
+#ifdef PATCHED
+ return dup_counter + 1;
+#else
+ return dup_counter;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c b/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c
new file mode 100644
index 000000000000..d5e70994c582
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/sympos_vmlinux.c
@@ -0,0 +1,40 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Two translation units with a same-named static, placed so that the linker
+ * puts them in the opposite order to the one they appear in the symbol table.
+ *
+ * VARSEC selects the section the static lands in. Linking with
+ * --sort-section=name then orders them alphabetically rather than by object
+ * order, so the first symbol in the symbol table ends up at the *higher*
+ * address. That is the whole point: counting symbol table order and reading
+ * the linked image's addresses now give different answers, which is what makes
+ * it possible to tell which one klp diff used.
+ *
+ * Only use_a is patched, so exactly one sympos is emitted and there is nothing
+ * to attribute.
+ */
+
+#ifndef FUNC_NAME
+#define FUNC_NAME use_a
+#endif
+#ifndef VARSEC
+#define VARSEC ".data.mmm"
+#endif
+
+#ifndef NO_MODINFO
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+#endif
+
+/* volatile so it survives as an STT_OBJECT rather than being folded away */
+static volatile int dup_counter __attribute__((section(VARSEC))) = 1;
+
+int FUNC_NAME(int x)
+{
+ dup_counter += x;
+#ifdef PATCHED
+ return dup_counter + 1;
+#else
+ return dup_counter;
+#endif
+}
diff --git a/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c b/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c
new file mode 100644
index 000000000000..b88830f41a92
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c
@@ -0,0 +1,57 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Three translation units linked with ThinLTO, two of which have a file-local
+ * helper of the same name.
+ *
+ * TU_C calls into both of the others, so ThinLTO imports entry_a and entry_b
+ * and with them the static helper each one calls. A file-local symbol which
+ * has to become visible is renamed helper.llvm.<hash>, and the hash is content
+ * derived -- so the two helpers get different hashes from each other, and
+ * TU_A's gets a different one again after the patch changes it. Only TU_A's
+ * changes: were both bodies to change, both would be cloned whichever way
+ * they were paired, and the pairing would not be observable.
+ *
+ * That leaves klp diff with two symbols in the original and two in the patched
+ * object, all four named differently, which have to be paired up correctly.
+ * Demangling alone gives "helper" for all of them; something else has to
+ * decide which is which.
+ *
+ * Only TU_A's helper changes. That is what makes a wrong pairing observable:
+ * paired correctly, one helper is changed and the other is not, so exactly one
+ * is cloned. Paired the wrong way round, both look changed -- or the wrong
+ * one does. If both bodies changed the outcome would be the same either way
+ * and the test would prove nothing.
+ *
+ * BASE differs between the two so their bodies are not identical to begin
+ * with.
+ */
+
+#if defined(TU_C)
+extern int entry_a(int x);
+extern int entry_b(int x);
+int glue(int x) { return entry_a(x) + entry_b(x + 1); }
+#else
+#ifdef TU_B
+#define ENTRY entry_b
+#define BASE 5
+#else
+#define ENTRY entry_a
+#define BASE 10
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+#endif
+static __attribute__((noinline)) int helper(int x, int len)
+{
+ int sum = 0, i;
+
+ for (i = 0; i < len; i++)
+#if defined(PATCHED) && !defined(TU_B)
+ sum += i * 2 + BASE; /* only TU_A's helper changes */
+#else
+ sum += i + BASE;
+#endif
+ return sum + x;
+}
+
+int ENTRY(int x) { return helper(x, 4); }
+#endif
diff --git a/tools/objtool/tests/generic/fixtures/thinlto_local.c b/tools/objtool/tests/generic/fixtures/thinlto_local.c
new file mode 100644
index 000000000000..fe9f9e6bf44f
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/thinlto_local.c
@@ -0,0 +1,39 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Two translation units (TU_B selects the second) linked with ThinLTO.
+ * Importing bump() promotes the file-local counter, renaming it
+ * counter.llvm.<hash>. The hash is content derived, so it differs between the
+ * original and patched builds.
+ */
+
+#ifdef TU_B
+
+extern int bump(void);
+
+int other_entry(void)
+{
+ return bump() + bump();
+}
+
+#else
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+static int counter;
+
+int bump(void)
+{
+ return ++counter;
+}
+
+int target(void)
+{
+#ifdef PATCHED
+ return counter + 1;
+#else
+ return counter;
+#endif
+}
+
+#endif
diff --git a/tools/objtool/tests/generic/fixtures/ubsan_noise.c b/tools/objtool/tests/generic/fixtures/ubsan_noise.c
new file mode 100644
index 000000000000..bf5999254163
--- /dev/null
+++ b/tools/objtool/tests/generic/fixtures/ubsan_noise.c
@@ -0,0 +1,49 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A translation unit built with UBSAN, where only one of two functions is
+ * patched.
+ *
+ * Every instrumented operation gets a per-callsite metadata object in an
+ * anonymous data section -- .data..Lubsan_data and .data..Lubsan_type from
+ * GCC, .data..L__unnamed_ from Clang -- and a call to a __ubsan_handle_*
+ * routine. The names are compiler-generated and carry no meaning across a
+ * rebuild, so klp diff has to treat those sections as uncorrelated rather than
+ * pairing them up by name.
+ *
+ * untouched() is byte-identical in both builds and exists to catch the false
+ * positive: if the metadata were correlated by name, its shifts would look
+ * changed and it would be dragged into the patch.
+ *
+ * The shifts are what draw the instrumentation. A bounds check would do as
+ * well but neither compiler emits one for an index it can prove in range.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int shift_by(int v, int n);
+
+int untouched(int v, int n)
+{
+ int s = 0;
+
+ s += v << (n & 31);
+ s += v << ((n + 1) & 31);
+ s += shift_by(v, n);
+
+ return s;
+}
+
+int touched(int v, int n)
+{
+ int s = 0;
+
+ s += v << (n & 31);
+#ifdef PATCHED
+ s += v << ((n + 3) & 31);
+#else
+ s += v << ((n + 2) & 31);
+#endif
+
+ return s;
+}
diff --git a/tools/objtool/tests/generic/test-abs-and-addressable.sh b/tools/objtool/tests/generic/test-abs-and-addressable.sh
new file mode 100755
index 000000000000..6adb23ed4b88
--- /dev/null
+++ b/tools/objtool/tests/generic/test-abs-and-addressable.sh
@@ -0,0 +1,50 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# An absolute symbol and an __ADDRESSABLE() pointer must not disturb the
+# function being patched.
+#
+# A SHN_ABS symbol has no section, so any walk of sym->sec which does not check
+# dereferences NULL -- and the kernel has plenty, from linker scripts and from
+# .set in assembly. __ADDRESSABLE() emits a pointer into .discard.addressable
+# to keep a symbol referenced; it means nothing to a livepatch and is discarded
+# at link time, but it is a relocation like any other and gets looked at.
+#
+# Neither is what the patch changes. The failure this guards against is not a
+# wrong answer but a crash or an error on input the kernel produces routinely,
+# which would make any function near one unpatchable.
+#
+# Not isolated to a single guard: the absolute symbol here has zero length, so
+# it is excluded before the section check is reached and removing that check
+# alone changes nothing observable. This stands as a check on the behaviour
+# rather than on the line which produces it.
+#
+# Covers the same ground as corpus/x86_64/checksum-abs-sym-skip and
+# addressable-symbols in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair abs_and_addressable.c
+
+# The premise: the fixture really did produce both.
+in_symbols orig.o | grep -q 'ABS.*abs_sym' ||
+ probe_skip "assembler did not make abs_sym absolute here"
+assert_input_section .discard.addressable
+
+# Checksumming has to survive them, and still see the function that changed.
+run_checksum
+assert_checksum_differs target
+assert_checksum_matches helper
+
+# So does the diff.
+run_diff
+assert_patched target
+assert_not_patched helper
+
+# An absolute symbol has no address to record a checksum against, so it gets
+# no entry -- the reference to it is what mattered, not the symbol itself.
+in_relocs orig.o | awk '/rela\.discard\.sym_checksum/,/^$/' | grep -qw abs_sym &&
+ fail "absolute symbol got a checksum entry"
+
+pass "absolute and __ADDRESSABLE symbols do not disturb the patched function"
diff --git a/tools/objtool/tests/generic/test-basic.sh b/tools/objtool/tests/generic/test-basic.sh
new file mode 100755
index 000000000000..562edfbc8646
--- /dev/null
+++ b/tools/objtool/tests/generic/test-basic.sh
@@ -0,0 +1,17 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Only functions whose code changed get cloned into the patch.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair basic.c
+run_diff
+
+assert_patched changed
+assert_not_patched untouched
+assert_section ".init.klp_funcs"
+assert_section ".init.klp_objects"
+
+pass "changed function cloned, unchanged function left alone"
diff --git a/tools/objtool/tests/generic/test-changed-data.sh b/tools/objtool/tests/generic/test-changed-data.sh
new file mode 100755
index 000000000000..c5c7381bb409
--- /dev/null
+++ b/tools/objtool/tests/generic/test-changed-data.sh
@@ -0,0 +1,18 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Livepatching replaces functions, not data. A changed data symbol must be
+# rejected.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair changed_data.c
+run_diff 255
+
+diff_log | grep -q 'changed data: klp_test_data' ||
+ fail "expected rejection, got: $(diff_log | tail -1)"
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "changed data symbol rejected"
diff --git a/tools/objtool/tests/generic/test-checksum-data.sh b/tools/objtool/tests/generic/test-checksum-data.sh
new file mode 100755
index 000000000000..e915026b79a7
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-data.sh
@@ -0,0 +1,61 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# What a data object's checksum has to cover.
+#
+# checksum_update_object() hashes the symbol's length, its bytes (when the
+# section has any -- .bss does not), and then
+# every relocation it carries -- as the target's name plus the adjusted addend,
+# except for a reference into a string section, which contributes the string's
+# contents instead.
+#
+# Each of those is load-bearing, and the failure is always the same shape: a
+# checksum that ignores one of them calls a changed object unchanged, klp diff
+# leaves it out of the patch, and the patched code goes on reading the
+# kernel's old copy. Nothing says so at build time.
+#
+# The string case is the one that cannot be caught by hashing bytes alone. The
+# pointer is identical -- same section, same offset -- and only the text it
+# refers to moved.
+#
+# Covers the same ground as corpus/x86_64/checksum-data-basic,
+# checksum-data-func-ptr, checksum-data-string-ptr and checksum-string-reloc in
+# Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# check <flag> <symbol> <what changed>
+#
+# Build the pair with one difference and require that symbol's checksum to move.
+check()
+{
+ build_pair checksum_data.c "-D$1"
+ run_checksum
+
+ assert_checksum_differs "$2"
+}
+
+# The object's own bytes.
+check PLAIN_VALUE plain
+# Its length, for a .bss object whose bytes are not hashed at all.
+check LONGER sized
+# A relocation's target: same bytes in the object, different symbol named.
+check WHICH_FUNC descriptor
+check WHICH_STR descriptor
+# The contents of a string the object points at, with the pointer untouched.
+check STR_CONTENT descriptor
+# A relocation's addend: same target symbol, different offset into it.
+check WHICH_SLOT descriptor
+# The same, for a static reached through its section symbol: the reference has
+# to be resolved back to the object before there is a name or offset to hash.
+check WHICH_PRIV descriptor
+
+# Having shown five things that must change it, show one that must not: an
+# unrelated edit elsewhere in the file leaves this object alone.
+build_pair checksum_data.c -DPLAIN_VALUE
+run_checksum
+assert_checksum_matches descriptor
+
+pass "data checksums cover length, bytes, reloc targets and string contents"
diff --git a/tools/objtool/tests/generic/test-checksum-debug.sh b/tools/objtool/tests/generic/test-checksum-debug.sh
new file mode 100755
index 000000000000..78856b191636
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-debug.sh
@@ -0,0 +1,49 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# "klp checksum --debug-checksum" prints a per-instruction checksum stream,
+# and klp-build -f (--show-first-changed) parses it to report where a function
+# first differs between the original and patched builds.
+#
+# It is a debugging aid, so nothing fails when it breaks: klp-build greps the
+# stream, and an unmatched grep just yields no output, which reads as "no
+# instruction changed". That is exactly how the format drifted out from under
+# it once already. Pin the shape klp-build depends on:
+#
+# DEBUG: <object>: checksum: <func>(): <sym>+0x<offset> <16 hex digits>
+#
+# and that --dry-run leaves the object alone, since klp-build runs this against
+# objects it is going to checksum again for real.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair basic.c
+
+before="$(md5sum < "$workdir/orig.o")"
+
+"$OBJTOOL" klp checksum --dry-run --debug-checksum=changed \
+ "$workdir/orig.o" > "$workdir/debug.log" 2>&1 ||
+ fail "klp checksum --debug-checksum failed"
+
+# --dry-run has to mean it: klp-build checksums these objects again afterwards,
+# and "already has .discard.sym_checksum, skipping" would lose the real run.
+[ "$(md5sum < "$workdir/orig.o")" = "$before" ] ||
+ fail "--dry-run modified the object"
+has_input_section orig.o .discard.sym_checksum &&
+ fail "--dry-run created .discard.sym_checksum"
+
+grep -qE '^DEBUG: .*: checksum: changed\(\): [^ ]+\+0x[0-9a-f]+ [0-9a-f]{16}$' \
+ "$workdir/debug.log" ||
+ fail "unexpected --debug-checksum format: $(head -1 "$workdir/debug.log")"
+
+# This is the pattern klp-build greps with. Keep it working verbatim.
+grep -qE "^DEBUG: .*checksum: changed\(\): " "$workdir/debug.log" ||
+ fail "klp-build's --show-first-changed pattern no longer matches"
+
+# Only the requested function, or klp-build attributes instructions to the
+# wrong one.
+grep -qE 'checksum: untouched\(\)' "$workdir/debug.log" &&
+ fail "--debug-checksum=changed also dumped untouched()"
+
+pass "--debug-checksum format is the one klp-build -f parses"
diff --git a/tools/objtool/tests/generic/test-checksum-insn.sh b/tools/objtool/tests/generic/test-checksum-insn.sh
new file mode 100755
index 000000000000..e1c04a1f518a
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-insn.sh
@@ -0,0 +1,49 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# What a function's checksum has to cover beyond its instruction bytes.
+#
+# checksum_update_insn() hashes the raw bytes, and then what any relocation on
+# the instruction refers to: a string section contributes the string's
+# contents, anything else the target symbol's name and the adjusted addend,
+# with a reference to a static resolved back through its section symbol first.
+#
+# None of these show up in the bytes. A rel32 operand is zero in the object
+# and supplied by the relocation, so every change below leaves the encoded
+# instruction byte-identical. A checksum stopping at the bytes reports the
+# function unchanged, klp diff omits it, and the patch silently does not
+# contain the fix.
+#
+# test-checksum-position is the other half of this: it covers what must *not*
+# change the checksum when a function merely moves.
+#
+# Covers the same ground as corpus/x86_64/checksum-reloc-sym,
+# checksum-pc-relative-addend, checksum-string-reloc and
+# checksum-sec-sym-resolve in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# check <flag> <what it changes>
+check()
+{
+ build_pair checksum_insn.c "-D$1"
+ run_checksum
+
+ # The premise for all of them: the operand is a relocation, not bytes.
+ assert_checksum_differs target
+}
+
+check WHICH_CALL # relocation target name
+check STR_CONTENT # contents of a string the code passes
+check WHICH_SLOT # addend, same target symbol
+check WHICH_PRIV # addend via a static's section symbol
+
+# The converse: rebuilding identical source leaves it alone, so the above is
+# not just "any rebuild moves the checksum".
+build_pair checksum_insn.c
+run_checksum
+assert_checksum_matches target
+
+pass "instruction checksums cover reloc targets, addends and string contents"
diff --git a/tools/objtool/tests/generic/test-checksum-position.sh b/tools/objtool/tests/generic/test-checksum-position.sh
new file mode 100755
index 000000000000..5ae759e719eb
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-position.sh
@@ -0,0 +1,51 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A function's checksum must not depend on where the function sits.
+#
+# A jump or call without a relocation encodes its target as an offset from the
+# instruction. Hashing those bytes makes the checksum change whenever anything
+# ahead of the function changes size -- so an unrelated edit elsewhere in the
+# file reports this function as changed too, and the patch grows to include it
+# and everything it references. Nothing fails; the livepatch is just larger and
+# riskier than the patch it came from.
+#
+# Here the "patch" moves target() away from callee() and changes nothing else:
+# both are aligned to 64 in the patched build, which shifts them apart without
+# touching a byte of either. See the fixture for why it is done that way.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# -fno-function-sections, or each function is at offset 0 of its own section
+# and target() never moves.
+build_pair checksum_position.c -fno-function-sections
+
+assert_input_symbol target
+
+# The fixture is only meaningful if the displacement target's calls encode
+# actually changed, and that is the distance to callee() -- not target's own
+# offset. A compiler which shifted the two by the same amount would move
+# target and leave the distance alone, and then the bytes are identical and
+# the checksum matches for the uninteresting reason. Ask about the distance.
+sym_off() # $1 object, $2 symbol
+{
+ in_symbols "$1" | awk -v n="$2" '$NF == n { print $2; exit }'
+}
+
+orig_t="$(sym_off orig.o target)"; orig_c="$(sym_off orig.o callee)"
+new_t="$(sym_off patched.o target)"; new_c="$(sym_off patched.o callee)"
+
+[ -n "$orig_t" ] && [ -n "$orig_c" ] && [ -n "$new_t" ] && [ -n "$new_c" ] ||
+ fail "target or callee missing from one of the objects"
+
+orig_gap=$(( 16#$orig_t - 16#$orig_c ))
+new_gap=$(( 16#$new_t - 16#$new_c ))
+[ "$orig_gap" != "$new_gap" ] ||
+ probe_skip "this compiler kept target() and callee() the same distance" \
+ "apart; the call displacement did not change"
+
+assert_checksum_matches target
+
+pass "checksum unchanged when the function only moves"
diff --git a/tools/objtool/tests/generic/test-checksum-skip.sh b/tools/objtool/tests/generic/test-checksum-skip.sh
new file mode 100755
index 000000000000..f245535a0f36
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-skip.sh
@@ -0,0 +1,81 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Symbols which must not get a checksum entry of their own.
+#
+# .discard.sym_checksum is an array of { address, checksum } looked up by the
+# address a relocation points at, so the invariant is one entry per address.
+# calculate_checksums() skips three kinds of symbol to keep it:
+#
+# zero-length nothing to hash, and its address belongs to whatever really
+# lives there
+# alias a second name for an address already covered
+# cold part hashed into its parent, which func_for_each_insn() walks
+# into, so its own entry would double-count
+#
+# A duplicate entry is not a build failure. It makes the lookup ambiguous, and
+# whichever checksum loses is simply never consulted again -- so a function
+# whose code changed can be read as unchanged and dropped from the patch.
+#
+# Of the three, only the alias skip is isolated here: removing it makes this
+# test fail. A zero-length symbol is excluded by more than one of the guards
+# at once -- its section has no data either -- so no single change makes that
+# assertion fail, and it stands as a check on the behaviour rather than on the
+# line which produces it. Nothing here reaches the cold-part skip, which
+# wants a compiler that splits functions; test-cold-function covers that
+# symbol surviving into the patch, not its checksum.
+#
+# Covers the same ground as corpus/x86_64/checksum-zero-len-sym,
+# checksum-alias-skip and checksum-cold-skip in Joe Lawrence's klp-build unit
+# test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair checksum_skip.c
+
+assert_input_symbol empty_marker
+assert_input_symbol alias_function
+run_checksum
+
+# entries_for <object> [-t]
+#
+# The symbol names .discard.sym_checksum has an entry for, one per line. With
+# -t, what each entry points at instead: the name and its addend, which is the
+# address the kernel looks the entry up by. A name alone is not that address,
+# since a relocation against a section symbol names the section and puts the
+# offset in the addend, and several entries can then share a name honestly.
+entries_for()
+{
+ in_relocs "$1" | awk -v target="${2:-}" '/rela\.discard\.sym_checksum/,/^$/ {
+ if ($1 !~ /^[0-9a-f]{8,}/)
+ next
+ if (target == "-t")
+ print $5, $6, $7
+ else
+ print $5
+ }'
+}
+
+entries="$(entries_for orig.o)"
+
+# The control: something real did get an entry, so an empty listing cannot
+# make the rest of this pass by default.
+echo "$entries" | grep -qx target ||
+ fail "no checksum entry for target"
+
+echo "$entries" | grep -qx empty_marker &&
+ fail "zero-length symbol got a checksum entry"
+
+# One of the two names for that address is kept and the other skipped; which
+# one falls out of symbol table order and is not the point. Two would be.
+n="$(echo "$entries" | grep -cxE 'real_function|alias_function')"
+[ "$n" = 1 ] ||
+ fail "expected 1 checksum entry across real_function and its alias, found $n"
+
+# One entry per address, which is what the skipping is for.
+dupes="$(entries_for orig.o -t | sort | uniq -d)"
+[ -z "$dupes" ] ||
+ fail "two checksum entries for one address: $dupes"
+
+pass "zero-length symbols and aliases get no checksum entry of their own"
diff --git a/tools/objtool/tests/generic/test-checksum-value.sh b/tools/objtool/tests/generic/test-checksum-value.sh
new file mode 100755
index 000000000000..feae6a12e98d
--- /dev/null
+++ b/tools/objtool/tests/generic/test-checksum-value.sh
@@ -0,0 +1,37 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# The per-function checksums klp checksum records are what klp diff uses to
+# decide which functions changed. A checksum covering too little misses a real
+# change and the patch silently omits the function; one covering too much, or
+# unstable across identical input, clones functions nobody patched and drags
+# their dependencies in with them.
+#
+# test-basic covers which functions got cloned, which is downstream of this and
+# passes for either kind of wrong checksum as long as the two errors do not
+# happen to cancel. This checks the checksums themselves.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair basic.c
+
+assert_input_symbol changed
+assert_input_symbol untouched
+
+run_checksum
+
+# The edited function's checksum has to move, the untouched one's must not.
+assert_checksum_differs changed
+assert_checksum_matches untouched
+
+# And it has to be a function of the code, not of the build: checksumming the
+# same input twice has to give the same answer, or every rebuild reports
+# spurious changes.
+first="$(checksum_of orig.o changed)"
+build_pair basic.c
+run_checksum
+[ "$(checksum_of orig.o changed)" = "$first" ] ||
+ fail "checksum for 'changed' differs between builds of identical source"
+
+pass "checksums track the changed function and are stable across rebuilds"
diff --git a/tools/objtool/tests/generic/test-cold-function.sh b/tools/objtool/tests/generic/test-cold-function.sh
new file mode 100755
index 000000000000..a02652fe5341
--- /dev/null
+++ b/tools/objtool/tests/generic/test-cold-function.sh
@@ -0,0 +1,39 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Both halves of a split function belong to the patch; carrying only the hot
+# part leaves the cold path branching into unpatched code.
+
+. "$(dirname "$0")/../lib.sh"
+
+# Clang does not split functions into a cold part at all, so there is nothing
+# for this test to look at there. A given gcc may or may not split, which is a
+# version property rather than a compiler choice -- that stays a probe below.
+gcc_only "clang does not split functions into a cold part"
+
+setup
+
+split_flag=-freorder-blocks-and-partition
+cc_supports "$split_flag" || split_flag=
+
+build_pair cold_function.c $split_flag
+
+# Find what the compiler called the cold half -- target.cold, target.cold.0,
+# depending on version -- and name it exactly from here on.
+cold_sym="$(in_symbols orig.o | awk '$NF ~ /^target\.cold/ { print $NF; exit }')"
+[ -n "$cold_sym" ] ||
+ probe_skip "compiler did not split the function into a cold part"
+
+run_diff
+
+assert_patched target
+# Match the name field exactly, and require it to be defined. Had the cold
+# half been left behind, the branch to it would appear as an undefined
+# .klp.sym.vmlinux.target.cold,0 -- a different name, which happens to contain
+# this one. Asking about a column instead of the name would not tell them
+# apart: readelf prints SHN_LIVEPATCH as "OS [0xff20]" and llvm-readelf as
+# "OS[0xff20]", so the fields either side of the name shift between the two.
+out_symbols | awk -v n="$cold_sym" '$NF == n && $(NF - 1) != "UND"' | grep -q . ||
+ fail "cold half ($cold_sym) was not carried into the patch"
+
+pass "cold half carried into the patch with its parent"
diff --git a/tools/objtool/tests/generic/test-data-alignment.sh b/tools/objtool/tests/generic/test-data-alignment.sh
new file mode 100755
index 000000000000..8e389544a3b1
--- /dev/null
+++ b/tools/objtool/tests/generic/test-data-alignment.sh
@@ -0,0 +1,40 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A cloned data section keeps its alignment.
+#
+# Plenty of kernel data is aligned for correctness rather than speed: per-CPU
+# variables, anything touched by an aligned vector move, structures padded to
+# own a cacheline. A clone that lands under-aligned either faults on first use
+# or silently shares a line it was laid out to avoid, and neither shows up
+# until the patch is loaded on hardware that cares.
+#
+# Fixed by 2f2600decb30 ("objtool/klp: Fix alignment of cloned data
+# sections").
+#
+# Covers the same ground as corpus/x86_64/cloned-data-alignment in Joe
+# Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair data_alignment.c
+
+# The premise: the compiler really did over-align it, and the object is new in
+# the patch so it has to be cloned rather than referenced.
+want="$(in_sections patched.o | sed 's/^ *\[[ 0-9]*\] *//' |
+ awk '$1 == ".data.aligned_data" { print $NF }')"
+[ "$want" = 64 ] ||
+ probe_skip "compiler gave .data.aligned_data alignment '$want', not 64"
+has_input_section orig.o .data.aligned_data &&
+ fail "fixture put aligned_data in the original; nothing to clone"
+
+run_diff
+assert_section .data.aligned_data
+
+got="$(out_sections | sed 's/^ *\[[ 0-9]*\] *//' |
+ awk '$1 == ".data.aligned_data" { print $NF }')"
+[ "$got" = "$want" ] ||
+ fail "cloned .data.aligned_data has alignment $got, expected $want"
+
+pass "cloned data section keeps its alignment"
diff --git a/tools/objtool/tests/generic/test-export-symbol-for-modules.sh b/tools/objtool/tests/generic/test-export-symbol-for-modules.sh
new file mode 100755
index 000000000000..7e7bdde6a7ac
--- /dev/null
+++ b/tools/objtool/tests/generic/test-export-symbol-for-modules.sh
@@ -0,0 +1,39 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# EXPORT_SYMBOL_FOR_MODULES() puts a vmlinux symbol in a "module:<names>"
+# namespace, and the module loader grants access by matching the importing
+# module's name against that list. A livepatch module is never on the list, so
+# referencing such a symbol with a normal relocation fails modpost, and if that
+# is silenced, fails to load with "Unknown symbol". It needs a klp relocation,
+# the same as an unexported symbol.
+#
+# Ordinary namespaces are not affected: copy_import_ns() propagates the patched
+# object's import tags to the patch module, so a normal relocation works.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair cross_module.c
+
+sym=other_mod_func
+
+# Plain vmlinux export: a normal relocation is what we want.
+export_syms "$sym"
+run_diff
+assert_no_klp_sym "$sym"
+
+# Ordinary namespace: still a normal relocation.
+export_syms
+add_exports_ns vmlinux MY_NS "$sym"
+run_diff
+assert_no_klp_sym "$sym"
+
+# module: namespace: has to become a klp relocation.
+export_syms
+add_exports_ns vmlinux module:kvm "$sym"
+run_diff
+assert_klp_sym "$sym" vmlinux
+assert_section __klp_relocs.vmlinux
+
+pass "EXPORT_SYMBOL_FOR_MODULES symbol referenced with a klp relocation"
diff --git a/tools/objtool/tests/generic/test-function-removal.sh b/tools/objtool/tests/generic/test-function-removal.sh
new file mode 100755
index 000000000000..df76e81e3936
--- /dev/null
+++ b/tools/objtool/tests/generic/test-function-removal.sh
@@ -0,0 +1,34 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A patch which deletes a function leaves a symbol in the original with no
+# counterpart in the patched object. klp diff cannot correlate it, and must
+# say so and carry on: livepatching cannot remove code from a running kernel,
+# so what matters is that the surviving caller is patched and the deleted
+# function is not dragged into the patch module.
+#
+# Cloning it would be worse than useless -- dead code in the patch, plus
+# whatever it references, resolved against a kernel where it may not exist.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair function_removal.c
+
+# One-sided by construction: present in the original, gone from the patched.
+has_input_symbol orig.o going_away ||
+ fail "fixture has no going_away in the original"
+has_input_symbol patched.o going_away &&
+ fail "fixture still has going_away in the patched object"
+
+run_diff
+
+assert_diff_log 'no correlation: going_away'
+
+# The caller changed, so it is patched ...
+assert_patched caller
+# ... and the deleted function comes along in no form at all.
+assert_not_patched going_away
+assert_no_symbol going_away
+
+pass "deleted function reported as uncorrelated and left out of the patch"
diff --git a/tools/objtool/tests/generic/test-init-reference.sh b/tools/objtool/tests/generic/test-init-reference.sh
new file mode 100755
index 000000000000..921cf493cd79
--- /dev/null
+++ b/tools/objtool/tests/generic/test-init-reference.sh
@@ -0,0 +1,29 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Init code and data are freed after boot, so a klp relocation against them can
+# never resolve. Such a patch must be rejected.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair init_reference.c
+
+# The rejection can only happen if the patched build really does reference the
+# init symbol. The fixture makes init_only volatile so the read cannot be
+# folded away, and this checks that it worked: without a relocation klp diff
+# would succeed, and the failure below would read as a missing check rather
+# than as a fixture which stopped posing the question.
+# Either spelling will do. A file-local variable is reached through its
+# section symbol -- .init.data -- and a global one by name; which of the two
+# the compiler picks is its business, and the reference is what matters.
+in_relocs "$patched_obj" |
+ awk '$5 == ".init.data" || $5 == "init_only"' | grep -q . ||
+ fail "patched object has no reference into .init.data; the fixture tests nothing"
+
+run_diff 255
+
+diff_log | grep -q "can't patch or reference init code/data" ||
+ fail "expected rejection, got: $(diff_log | tail -1)"
+
+pass "reference to init data rejected"
diff --git a/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh b/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh
new file mode 100755
index 000000000000..fa2f913d860b
--- /dev/null
+++ b/tools/objtool/tests/generic/test-jump-label-exempt-keys.sh
@@ -0,0 +1,51 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Two kinds of module-owned static branch key are disabled with a warning
+# instead of rejected.
+#
+# A module-local key is normally fatal: late module patching lets the livepatch
+# load before the module it depends on, so the unresolved __jump_table entry is
+# dereferenced by jump_label_add_module(). test-jump-label-module-key covers
+# that rejection.
+#
+# Tracepoints and pr_debug() generate such keys everywhere, though, and
+# refusing them outright would make any function containing a trace_*() call or
+# a pr_debug() unpatchable. So klp diff drops the entry, says so, and carries
+# on: the patched code keeps working with that one tracepoint or debug print
+# permanently off.
+#
+# Both halves matter. A build that fails is a function nobody can patch; an
+# entry left in place is the memory corruption the rejection exists to prevent.
+#
+# The two exemptions are isolated: remove either and this fails. That the
+# entry is then dropped is asserted but not isolated -- making the caller keep
+# it anyway produces no output difference here, so that assertion stands as a
+# check on the behaviour rather than on the line which produces it.
+#
+# Covers the same ground as corpus/x86_64/static-call-module-tracepoint and
+# pr-debug-unsupported in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# check <key name> <expected warning>
+check()
+{
+ build_pair jump_label.c -DKEY_NAME="$1" -DMODNAME='"klp_testmod"'
+ require_input_section __jump_table
+
+ # Accepted, not rejected: this is the whole point.
+ run_diff
+ assert_diff_log "$2"
+
+ # And the entry is gone, not merely complained about.
+ assert_patched target
+ assert_no_section __jump_table
+}
+
+check __tracepoint_klp_test 'disabling unsupported tracepoint klp_test'
+check __UNIQUE_ID_ddebug_klp_test 'disabling unsupported pr_debug'
+
+pass "tracepoint and pr_debug keys disabled with a warning, not rejected"
diff --git a/tools/objtool/tests/generic/test-jump-label-key.sh b/tools/objtool/tests/generic/test-jump-label-key.sh
new file mode 100755
index 000000000000..142f94bd38ec
--- /dev/null
+++ b/tools/objtool/tests/generic/test-jump-label-key.sh
@@ -0,0 +1,44 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A cloned __jump_table entry must keep a relocation in its key slot, whether
+# the key needs a klp relocation or not.
+
+. "$(dirname "$0")/../lib.sh"
+
+key=klp_test_key
+
+setup
+build_pair jump_label.c
+
+has_input_section orig.o __jump_table ||
+ probe_skip "fixture produced no __jump_table on this arch"
+
+key_slot_relocs()
+{
+ out_relocs | awk '/rela__jump_table/,/^$/' | grep -c "^0*8[[:space:]]"
+}
+
+# Unexported: klp relocation, key slot holds a tombstone.
+export_syms
+run_diff
+
+[ "$(key_slot_relocs)" = 1 ] ||
+ fail "unexported key: key slot has no relocation"
+out_relocs | awk '/rela__jump_table/,/^$/' | grep -q "\.klp\.tombstone\.$key" ||
+ fail "unexported key: expected a .klp.tombstone.$key relocation"
+out_symbols | grep -q "\.klp\.sym\..*\.$key," ||
+ fail "unexported key: no .klp.sym reference for the real relocation"
+
+# Exported: ordinary relocation, no klp machinery.
+export_syms "$key"
+run_diff
+
+[ "$(key_slot_relocs)" = 1 ] ||
+ fail "exported key: key slot has no relocation"
+out_relocs | awk '/rela__jump_table/,/^$/' | grep -q "[[:space:]]$key[[:space:]]*+" ||
+ fail "exported key: expected a direct relocation to $key"
+out_symbols | grep -q '\.klp\.tombstone\.' &&
+ fail "exported key: tombstone emitted for an exported symbol"
+
+pass "key slot populated for exported and unexported vmlinux keys"
diff --git a/tools/objtool/tests/generic/test-jump-label-module-key.sh b/tools/objtool/tests/generic/test-jump-label-module-key.sh
new file mode 100755
index 000000000000..7c3f82cdd3c1
--- /dev/null
+++ b/tools/objtool/tests/generic/test-jump-label-module-key.sh
@@ -0,0 +1,22 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A static branch key owned by a module cannot be reached with a klp
+# relocation and must be rejected.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair jump_label.c -DMODNAME='"klp_testmod"'
+
+has_input_section orig.o __jump_table ||
+ probe_skip "fixture produced no __jump_table on this arch"
+
+run_diff 255
+
+diff_log | grep -q 'unsupported static branch key klp_test_key' ||
+ fail "expected rejection, got: $(diff_log | tail -1)"
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "module-owned static branch key rejected"
diff --git a/tools/objtool/tests/generic/test-jump-label-module-static-key.sh b/tools/objtool/tests/generic/test-jump-label-module-static-key.sh
new file mode 100755
index 000000000000..7bdcbaf2fa60
--- /dev/null
+++ b/tools/objtool/tests/generic/test-jump-label-module-static-key.sh
@@ -0,0 +1,45 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A static branch key owned by a module is rejected whether the key is global
+# or file-local.
+#
+# The rejection matters because late module patching allows the livepatch
+# module to load before the module it depends on: the __jump_table klp reloc is
+# then unresolved, and jump_label_add_module() dereferences an uninitialized
+# pointer. Catching it at build time is the only defence.
+#
+# test-jump-label-module-key covers the global key. A file-local one reaches
+# the same check by a different route: the compiler emits the reference against
+# the section symbol plus an addend, so validate_special_section_klp_reloc()
+# has to resolve it to the underlying object before it can see a key at all.
+# Until it did, a static key was passed over as "not STT_OBJECT" and the
+# unsupported reference was emitted with nothing said.
+#
+# Fixed by f9fb44b0ecef ("objtool/klp: Fix detection of corrupt static
+# branch/call entries").
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair jump_label.c -DSTATIC_KEY -DMODNAME='"klp_testmod"'
+
+require_input_section __jump_table
+
+# The premise: the key is reached through its section symbol, not by name.
+# Without that this is just a second copy of test-jump-label-module-key.
+input_jump_relocs="$(in_relocs orig.o | awk '/rela__jump_table/,/^$/')"
+
+echo "$input_jump_relocs" | grep -q klp_test_key ||
+ fail "fixture produced no __jump_table reference to the key"
+echo "$input_jump_relocs" | grep -qE '\.(bss|data)\.klp_test_key' ||
+ probe_skip "compiler referenced the static key by name, not through its section"
+
+run_diff 255
+
+diff_log | grep -q 'unsupported static branch key klp_test_key' ||
+ fail "expected rejection, got: $(diff_log | tail -1)"
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "module-owned file-local static branch key rejected"
diff --git a/tools/objtool/tests/generic/test-jump-label-new-key.sh b/tools/objtool/tests/generic/test-jump-label-new-key.sh
new file mode 100755
index 000000000000..3bc6005fc776
--- /dev/null
+++ b/tools/objtool/tests/generic/test-jump-label-new-key.sh
@@ -0,0 +1,51 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A patch may introduce a static branch where the original function had none.
+#
+# That is not the same case as patching a function which already has one. The
+# __jump_table entry is itself new, so there is no counterpart in the original
+# to correlate it against: klp diff has to carry the entry and the key into the
+# patch from scratch, and the key has to be reached the way any other reference
+# to a vmlinux symbol is.
+#
+# Get it wrong and the entry is dropped, leaving a static branch the kernel
+# never patches -- the code takes the wrong arm forever, silently.
+#
+# Where the key lives still decides whether that is allowed, exactly as it does
+# for a key the original already had: a module-owned one cannot be reached, so
+# introducing one has to stop the build rather than emit an entry nothing will
+# resolve.
+#
+# Covers the same ground as corpus/x86_64/static-branch-vmlinux-new and
+# static-branch-module-new in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup klp_test_key
+build_pair jump_label.c -DNEW_KEY
+
+# The premise: the original really has no jump table, and the patched one does.
+has_input_section orig.o __jump_table &&
+ fail "fixture put a __jump_table in the original; nothing new to add"
+has_input_section patched.o __jump_table ||
+ probe_skip "compiler produced no __jump_table on this arch"
+
+run_diff
+
+assert_patched target
+assert_section __jump_table
+assert_reloc_sym __jump_table target
+
+# The same new branch, with the key owned by a module. Drop the vmlinux export
+# first: while it is exported the key is reachable and being new changes
+# nothing, which is what the first version of this got wrong.
+export_syms
+rm -f "$workdir/out.o"
+build_pair jump_label.c -DNEW_KEY -DMODNAME='"klp_testmod"'
+run_diff 255
+assert_diff_log 'unsupported static branch key klp_test_key'
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "static branch introduced by the patch carried in, or rejected for a module key"
diff --git a/tools/objtool/tests/generic/test-klp-funcs-content.sh b/tools/objtool/tests/generic/test-klp-funcs-content.sh
new file mode 100755
index 000000000000..32fbd5f6a6fa
--- /dev/null
+++ b/tools/objtool/tests/generic/test-klp-funcs-content.sh
@@ -0,0 +1,45 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# .init.klp_funcs is the list the kernel walks to decide what to patch, and
+# .init.klp_objects points at it. Existing tests assert only that the sections
+# exist, which they do whether the list names the right functions, the wrong
+# ones, or none at all -- and a patch module with an empty function list loads
+# perfectly happily and patches nothing.
+#
+# Each entry pairs a name string in .rodata.klp.str1.1 with a relocation to the
+# new function, so both halves are checkable.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair klp_funcs.c
+run_diff
+
+assert_section .init.klp_funcs
+assert_section .init.klp_objects
+
+# Two functions changed, so two entries, each contributing a name relocation
+# and a function relocation.
+assert_reloc_count .init.klp_funcs 4
+
+# The functions that changed are named ...
+assert_reloc_sym .init.klp_funcs first
+assert_reloc_sym .init.klp_funcs second
+# ... and the one that did not is absent, from the list and from the patch.
+assert_no_reloc_sym .init.klp_funcs third
+assert_not_patched third
+
+# The names the kernel matches on are real strings, not just relocations.
+# readelf prints one per line as "[ offset] <string>", so compare the whole
+# name: a word-boundary match would also accept ".text.first", since a dot is
+# not a word character.
+for name in first second; do
+ out_strings .rodata.klp.str1.1 | awk -v n="$name" '$NF == n' | grep -q . ||
+ fail "no '$name' string in .rodata.klp.str1.1"
+done
+
+# The object list has to reach the function list, or nothing is walked.
+assert_reloc_sym .init.klp_objects .init.klp_funcs
+
+pass "klp_funcs lists exactly the changed functions, by name and relocation"
diff --git a/tools/objtool/tests/generic/test-local-to-global-flip.sh b/tools/objtool/tests/generic/test-local-to-global-flip.sh
new file mode 100755
index 000000000000..45e92797190b
--- /dev/null
+++ b/tools/objtool/tests/generic/test-local-to-global-flip.sh
@@ -0,0 +1,63 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A patch can change a symbol's linkage without renaming it: dropping "static"
+# from a helper so something else can call it, or adding it to one that is no
+# longer shared. Correlation keys off more than the name, so a symbol whose
+# binding moved can fail to pair with itself.
+#
+# Failing to correlate is not a build failure. The symbol looks new, and a
+# "new" data symbol is either rejected or cloned as a second copy -- at which
+# point the patched code updates its own private variable and the rest of the
+# kernel keeps reading the original.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair local_to_global.c
+
+# Confirm the fixture really moved the bindings, in both directions.
+#
+# Match the binding and the name as fields, not as substrings of the line.
+# -ffunction-sections and -fdata-sections give these symbols sections of their
+# own, and the section symbols -- .text.flipped_up, .data.flipped_down -- are
+# always LOCAL, so "does a LOCAL line mention flipped_up" is answered by the
+# wrong symbol. GNU readelf 2.35 happens to leave section symbol names blank,
+# but llvm-readelf prints them, and a premise that holds on one readelf and
+# not the other is no premise at all.
+has_binding() # $1 object, $2 binding, $3 symbol
+{
+ in_symbols "$1" | awk -v b="$2" -v n="$3" '$5 == b && $NF == n' | grep -q .
+}
+
+has_binding orig.o LOCAL flipped_up ||
+ fail "flipped_up is not local in the original"
+has_binding patched.o GLOBAL flipped_up ||
+ fail "flipped_up is not global in the patched object"
+has_binding orig.o GLOBAL flipped_down ||
+ fail "flipped_down is not global in the original"
+has_binding patched.o LOCAL flipped_down ||
+ fail "flipped_down is not local in the patched object"
+
+run_diff
+
+assert_diff_log 'changed function: caller'
+
+# Correlated means each pairs with its own counterpart in the original, so the
+# patch refers back to the kernel's copy ...
+assert_klp_sym flipped_up vmlinux
+assert_klp_sym flipped_down vmlinux
+
+# ... rather than carrying its own. A second copy of flipped_down is the bad
+# outcome: patched code would update its private one while the rest of the
+# kernel keeps reading the original.
+assert_not_patched flipped_up
+assert_no_section .data.flipped_down
+assert_no_section .bss.flipped_down
+
+diff_log | grep -q 'no correlation' &&
+ fail "linkage change reported as an uncorrelated symbol"
+diff_log | grep -q 'changed data' &&
+ fail "linkage change reported as changed data"
+
+pass "symbols correlated across a change of linkage"
diff --git a/tools/objtool/tests/generic/test-local-vs-export.sh b/tools/objtool/tests/generic/test-local-vs-export.sh
new file mode 100755
index 000000000000..5b91101dadd6
--- /dev/null
+++ b/tools/objtool/tests/generic/test-local-vs-export.sh
@@ -0,0 +1,32 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# find_export() matched on symbol name alone, so a static function or variable
+# sharing a name with an export was mistaken for a reference to that export.
+# For a vmlinux export that means no klp relocation at all: the normal
+# relocation left behind is resolved by the module loader to the vmlinux
+# symbol, and the patched code quietly reads and writes the wrong object.
+#
+# Exports are always global, so a local symbol is never one.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_local.c
+
+# The static the fixture uses. Compilers mangle statics variously -- gcc says
+# counter.0, clang says target.counter -- so find what this one produced rather
+# than assuming a shape.
+local_sym="$(in_symbols orig.o |
+ awk '$4 == "OBJECT" && $5 == "LOCAL" && $8 ~ /counter/ { print $8; exit }')"
+[ -n "$local_sym" ] ||
+ fail "fixture produced no local 'counter' symbol"
+
+# Contrive the collision: something else exports that same name.
+export_syms "$local_sym" counter
+run_diff
+
+# Still treated as the local it is, not as the export.
+assert_klp_sym "$local_sym" vmlinux
+
+pass "local symbol not mistaken for an export of the same name"
diff --git a/tools/objtool/tests/generic/test-missing-checksum.sh b/tools/objtool/tests/generic/test-missing-checksum.sh
new file mode 100755
index 000000000000..89a05b1869ed
--- /dev/null
+++ b/tools/objtool/tests/generic/test-missing-checksum.sh
@@ -0,0 +1,18 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Without .discard.sym_checksum there is nothing to compare; concluding that
+# nothing changed would be worse than failing.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair basic.c
+
+( cd "$workdir" && "$OBJTOOL" klp diff orig.o patched.o out.o ) \
+ > "$workdir/diff.log" 2>&1 && fail "klp diff accepted an unchecksummed object"
+
+grep -q 'sym_checksum' "$workdir/diff.log" ||
+ fail "expected a complaint about the checksum section, got: $(tail -1 "$workdir/diff.log")"
+
+pass "unchecksummed input rejected"
diff --git a/tools/objtool/tests/generic/test-missing-modinfo.sh b/tools/objtool/tests/generic/test-missing-modinfo.sh
new file mode 100755
index 000000000000..6f242f7eb756
--- /dev/null
+++ b/tools/objtool/tests/generic/test-missing-modinfo.sh
@@ -0,0 +1,16 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# The module name in .modinfo ends up in the livepatch's klp_object, so an
+# object without one cannot be diffed.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair no_modinfo.c
+run_diff 255
+
+diff_log | grep -q 'modinfo' ||
+ fail "expected a complaint about .modinfo, got: $(diff_log | tail -1)"
+
+pass "object without .modinfo rejected"
diff --git a/tools/objtool/tests/generic/test-modname-normalize.sh b/tools/objtool/tests/generic/test-modname-normalize.sh
new file mode 100755
index 000000000000..a8059bdbd667
--- /dev/null
+++ b/tools/objtool/tests/generic/test-modname-normalize.sh
@@ -0,0 +1,26 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Module.symvers records the build-tree path of the object that exports a
+# symbol, not the name the module has at runtime: "arch/x86/kvm/kvm-intel",
+# where the kernel knows the module as "kvm_intel".
+#
+# The klp symbol name embeds the owning object, and livepatch matches it
+# against loaded modules by name. Left unnormalized it names a module that
+# does not exist, and the relocation is never resolved -- at load time, with no
+# build-time complaint.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_module_pair cross_module.c klp_testmod
+
+# A path with directory components, a dash, and no extension.
+export_syms
+add_exports "arch/x86/kvm/kvm-intel" other_mod_func
+run_diff
+
+# Directories stripped, dash to underscore.
+assert_klp_sym other_mod_func kvm_intel
+
+pass "Module.symvers paths normalized to runtime module names"
diff --git a/tools/objtool/tests/generic/test-module-object.sh b/tools/objtool/tests/generic/test-module-object.sh
new file mode 100755
index 000000000000..95c8402a523a
--- /dev/null
+++ b/tools/objtool/tests/generic/test-module-object.sh
@@ -0,0 +1,31 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A klp relocation section is named for the object being patched, not for the
+# object which happens to own the symbol being referenced. Deriving it from
+# the symbol means a cross-module reference lands in a section for an object
+# the patch may not even touch, so the relocation is never applied and the call
+# goes somewhere arbitrary.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_module_pair cross_module.c klp_testmod
+
+# The fixture has to have built as a module for any of this to mean anything.
+in_sections orig.o | grep -q '\.modinfo' ||
+ fail "fixture has no .modinfo"
+
+# other_mod_func belongs to a different module than the one being patched.
+add_exports other_mod other_mod_func
+run_diff
+
+# Named for the patched object ...
+assert_section __klp_relocs.klp_testmod
+# ... not for the object owning the symbol.
+assert_no_section __klp_relocs.other_mod
+
+run_post_link
+assert_klp_rela klp_testmod .text.target
+
+pass "klp relocation section named for the patched object"
diff --git a/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh b/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh
new file mode 100755
index 000000000000..635a75f6b8ba
--- /dev/null
+++ b/tools/objtool/tests/generic/test-module-vmlinux-reloc.sh
@@ -0,0 +1,40 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Patching a module, where the patched code references a vmlinux symbol which
+# needs a klp relocation.
+#
+# The kernel does not allow a module-targeted klp relocation to reference a
+# vmlinux symbol, and a symbol exported with EXPORT_SYMBOL_FOR_MODULES gets a
+# klp relocation. Put together, filing that relocation under the patched
+# module produces a patch the kernel refuses to apply to its target.
+#
+# So it goes under vmlinux instead, and is applied when the patch module loads
+# rather than when the patched module does. That is the opposite of the rule
+# for a reference to a module's symbol, which test-module-object covers; this
+# is the other branch of the same decision.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+# The object being patched is a module ...
+build_module_pair cross_module.c klp_testmod
+
+# ... and the symbol it references belongs to vmlinux, exported in a way that
+# still requires a klp relocation.
+export_syms
+add_exports_ns vmlinux module:kvm other_mod_func
+run_diff
+
+# Filed against vmlinux, applied when the patch loads.
+assert_section __klp_relocs.vmlinux
+assert_klp_sym other_mod_func vmlinux
+
+# Not against the patched module: that is the relocation the kernel rejects.
+assert_no_section __klp_relocs.klp_testmod
+
+run_post_link
+assert_klp_rela vmlinux .text.target
+assert_no_section ".klp.rela.klp_testmod..text.target"
+
+pass "klp relocation to a vmlinux symbol filed under vmlinux, not the patched module"
diff --git a/tools/objtool/tests/generic/test-new-data.sh b/tools/objtool/tests/generic/test-new-data.sh
new file mode 100755
index 000000000000..acb68ff19c99
--- /dev/null
+++ b/tools/objtool/tests/generic/test-new-data.sh
@@ -0,0 +1,27 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Data added by the patch has no counterpart in the running kernel and must be
+# carried into the livepatch.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair new_data.c
+
+# State the premise on both sides. The array is new in the patched build and
+# absent from the original; if the compiler folded it into the code instead of
+# emitting it, the assertion below would fail without saying why.
+has_input_symbol "$orig_obj" klp_new_data &&
+ fail "fixture put klp_new_data in the original; nothing new to carry"
+has_input_symbol "$patched_obj" klp_new_data ||
+ fail "compiler did not emit klp_new_data; the fixture tests nothing"
+in_relocs "$patched_obj" | grep -q 'klp_new_data' ||
+ fail "target() does not reference klp_new_data; the fixture tests nothing"
+
+run_diff
+
+assert_patched target
+assert_symbol klp_new_data
+
+pass "new data carried into the patch"
diff --git a/tools/objtool/tests/generic/test-new-export-ref.sh b/tools/objtool/tests/generic/test-new-export-ref.sh
new file mode 100755
index 000000000000..f0be2cf87fe3
--- /dev/null
+++ b/tools/objtool/tests/generic/test-new-export-ref.sh
@@ -0,0 +1,46 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A reference the patch adds has no counterpart in the original object. klp
+# diff used to reject any such reference needing a klp relocation, which ruled
+# out patches that call something they did not call before -- a common enough
+# thing for a fix to do.
+#
+# Module.symvers is what makes it safe: it says the symbol exists and who owns
+# it. But that is only sufficient for a vmlinux export. A new reference to a
+# module's export is a dependency the patch module does not declare, and the
+# relocation would resolve only if that module happened to be loaded, so it
+# stays an error.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair new_export_ref.c
+
+# Exported by vmlinux, in a module: namespace so it needs a klp relocation
+# rather than an ordinary one. Allowed.
+export_syms
+add_exports_ns vmlinux module:kvm newly_referenced
+run_diff
+assert_klp_sym newly_referenced vmlinux
+
+# Exported by a module the patched object does not depend on. Rejected, and
+# for that reason rather than some other.
+export_syms
+add_exports other_mod newly_referenced
+run_diff 255
+assert_diff_log 'undeclared module dependency'
+
+# ... unless the original already referenced something that module exports.
+# The loader will not let the patched object load without other_mod, so the
+# klp relocation has something to resolve against, and klp diff allows it.
+# This is the other half of the rule, and it fails in the opposite direction:
+# refusing here would reject a patch which is safe to apply.
+rm -f "$workdir/out.o"
+build_pair new_export_ref.c -DEXISTING_DEP
+export_syms
+add_exports other_mod newly_referenced existing_dep
+run_diff
+assert_klp_sym newly_referenced other_mod
+
+pass "new reference allowed for vmlinux and for a module already depended on"
diff --git a/tools/objtool/tests/generic/test-new-function.sh b/tools/objtool/tests/generic/test-new-function.sh
new file mode 100755
index 000000000000..b0b7f6443110
--- /dev/null
+++ b/tools/objtool/tests/generic/test-new-function.sh
@@ -0,0 +1,16 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A function added by the patch has no original to correlate against and must
+# still be carried into the livepatch.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair new_function.c
+run_diff
+
+assert_patched target
+assert_symbol klp_new_helper
+
+pass "new function carried into the patch with its caller"
diff --git a/tools/objtool/tests/generic/test-post-link.sh b/tools/objtool/tests/generic/test-post-link.sh
new file mode 100755
index 000000000000..a6d40ae1b18d
--- /dev/null
+++ b/tools/objtool/tests/generic/test-post-link.sh
@@ -0,0 +1,42 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# klp post-link converts the intermediate __klp_relocs.* sections into the
+# .klp.rela.* form the kernel applies at patch load. Getting this wrong is
+# invisible at build time: the module links and loads, and the relocations are
+# simply never applied.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_local.c
+
+# An unexported symbol is what produces a klp relocation in the first place.
+# The static local is renamed by the compiler -- counter.0 with gcc -- so the
+# fixture's premise is that some local of that shape exists, not that a symbol
+# called "counter" does. Find it first; everything below names it.
+local_sym="$(in_symbols orig.o |
+ awk '$4 == "OBJECT" && $5 == "LOCAL" && $8 ~ /counter/ { print $8; exit }')"
+[ -n "$local_sym" ] || fail "fixture produced no local 'counter' symbol"
+
+run_diff
+assert_section __klp_relocs.vmlinux
+
+# Nothing has converted them yet.
+assert_no_section ".klp.rela.vmlinux..text.target"
+
+# The original relocation is neutralised by pointing it at a tombstone, which
+# is what stops the module loader resolving it behind livepatch's back.
+#
+# Compilers mangle a static local differently -- gcc says counter.0, clang
+# target.counter -- so find what this one produced rather than assuming.
+assert_tombstone "$local_sym"
+
+run_post_link
+
+# One .klp.rela section per base section, carrying SHF_RELA_LIVEPATCH, against
+# a symbol in SHN_LIVEPATCH for the kernel to resolve.
+assert_klp_rela vmlinux .text.target
+assert_livepatch_sym "$local_sym"
+
+pass "klp relocations converted to .klp.rela with SHN_LIVEPATCH symbols"
diff --git a/tools/objtool/tests/generic/test-special-section-shared.sh b/tools/objtool/tests/generic/test-special-section-shared.sh
new file mode 100755
index 000000000000..05c6110b6aef
--- /dev/null
+++ b/tools/objtool/tests/generic/test-special-section-shared.sh
@@ -0,0 +1,26 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Only the patched function's special section entry may be extracted.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair special_section_shared.c
+
+has_input_section orig.o .kcfi_traps ||
+ probe_skip "fixture produced no .kcfi_traps on this arch"
+
+run_diff
+
+assert_patched target
+assert_not_patched other
+assert_section ".kcfi_traps"
+
+entries="$(out_relocs | awk '/rela\.kcfi_traps/,/^$/' | grep -c 'target')"
+[ "$entries" = 1 ] || fail "expected one .kcfi_traps entry, found $entries"
+
+out_relocs | awk '/rela\.kcfi_traps/,/^$/' | grep -q 'other' &&
+ fail "the untouched function's entry was dragged in"
+
+pass "only the patched function's entry extracted"
diff --git a/tools/objtool/tests/generic/test-special-section.sh b/tools/objtool/tests/generic/test-special-section.sh
new file mode 100755
index 000000000000..b6a9139c0062
--- /dev/null
+++ b/tools/objtool/tests/generic/test-special-section.sh
@@ -0,0 +1,20 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A .kcfi_traps entry belonging to a patched function must be extracted even
+# without ANNOTATE_DATA_SPECIAL and with a local label already at offset 0.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair special_section.c
+
+in_symbols orig.o | grep -q 'trap_marker' ||
+ probe_skip "fixture produced no .kcfi_traps on this arch"
+
+run_diff
+
+assert_patched target
+assert_section ".kcfi_traps"
+
+pass ".kcfi_traps extracted despite a local label at offset 0"
diff --git a/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh b/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh
new file mode 100755
index 000000000000..818b047ed4b6
--- /dev/null
+++ b/tools/objtool/tests/generic/test-static-call-annotate-stripped.sh
@@ -0,0 +1,42 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A patch may remove the last ANNOTATE_DATA_SPECIAL in a translation unit while
+# leaving the special section it described in place.
+#
+# klp diff needs entry boundaries for a special section: either an entsize, or
+# annotations naming where each entry starts. .static_call_sites has no
+# entsize, so the annotations are all there is -- and when the patched object
+# is the only side that lost them, the two sides no longer agree on how the
+# section divides up.
+#
+# The section must still be handled. Dropping it would leave the patched
+# function's static call unregistered; misreading its boundaries would attach
+# the entry to the wrong code. Either way nothing is reported at build time.
+#
+# Fixed by commit 3de711fba73a ("objtool/klp: Fix create_fake_symbols()
+# skipping entsize-based sections").
+#
+# Covers the same ground as corpus/x86_64/static-call-annotate-stripped in Joe
+# Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_call.c -DNO_ANNOTATE
+
+# The premise: the original describes its entry, the patched one no longer
+# does, and both still have the section itself.
+has_input_section orig.o .discard.annotate_data ||
+ fail "fixture produced no annotation in the original"
+has_input_section patched.o .discard.annotate_data &&
+ fail "patched object still has the annotation; nothing was stripped"
+assert_input_section .static_call_sites
+
+run_diff
+
+assert_patched target
+assert_section .static_call_sites
+assert_reloc_sym .static_call_sites target
+
+pass "static call site kept when the patch strips its data annotation"
diff --git a/tools/objtool/tests/generic/test-static-call-module-key.sh b/tools/objtool/tests/generic/test-static-call-module-key.sh
new file mode 100755
index 000000000000..260c4af72e05
--- /dev/null
+++ b/tools/objtool/tests/generic/test-static-call-module-key.sh
@@ -0,0 +1,36 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# As for static branches, a static call key owned by a module must be rejected
+# while a vmlinux-owned one is accepted.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_call.c
+
+has_input_section orig.o .static_call_sites ||
+ probe_skip "fixture produced no .static_call_sites on this arch"
+
+run_diff
+assert_patched target
+
+# The accepted half has to show the entry was carried, not just that the
+# function was: dropping the section silently would leave the patched call
+# unregistered, and "target was cloned" cannot tell the two apart.
+assert_section .static_call_sites
+assert_reloc_sym .static_call_sites target
+
+rm -f "$workdir/out.o"
+build_pair static_call.c -DMODNAME='"klp_testmod"'
+run_diff 255
+
+diff_log | grep -q 'unsupported static call key __SCK__klp_test_call' ||
+ fail "expected rejection, got: $(diff_log | tail -1)"
+
+# A rejection has to leave nothing behind. out.o was removed above, so
+# anything here was written by the run which was supposed to refuse.
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "module-owned static call key rejected, vmlinux-owned accepted"
diff --git a/tools/objtool/tests/generic/test-static-call-new.sh b/tools/objtool/tests/generic/test-static-call-new.sh
new file mode 100755
index 000000000000..f7d2b39c0fec
--- /dev/null
+++ b/tools/objtool/tests/generic/test-static-call-new.sh
@@ -0,0 +1,45 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A patch may introduce a static call where the original function had none.
+#
+# The .static_call_sites entry is then new, with nothing in the original to
+# correlate it against, so klp diff has to carry it into the patch from
+# scratch. Where the key lives still decides whether that is allowed: a
+# vmlinux key is reachable, and a module-owned one is not, for the same reason
+# an existing module key is not -- late module patching lets the livepatch load
+# first, and the unresolved entry is dereferenced when the module arrives.
+#
+# Both halves are here because they fail in opposite directions. Dropping the
+# new entry leaves a static call the kernel never patches; accepting a new
+# module-owned one is the corruption the check exists to prevent.
+#
+# Covers the same ground as corpus/x86_64/static-call-vmlinux-new and
+# static-call-module-new in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_call.c -DNEW_CALL
+
+# The premise: the original really has no static call, the patched one does.
+has_input_section orig.o .static_call_sites &&
+ fail "fixture put a .static_call_sites in the original; nothing new to add"
+has_input_section patched.o .static_call_sites ||
+ probe_skip "compiler produced no .static_call_sites on this arch"
+
+run_diff
+assert_patched target
+assert_section .static_call_sites
+assert_reloc_sym .static_call_sites target
+
+# The same new call, with the key owned by a module: not reachable, so the
+# build has to stop rather than emit a relocation nothing will resolve.
+rm -f "$workdir/out.o"
+build_pair static_call.c -DNEW_CALL -DMODNAME='"klp_testmod"'
+run_diff 255
+assert_diff_log 'unsupported static call key __SCK__klp_test_call'
+[ -e "$workdir/out.o" ] &&
+ fail "output object produced for a rejected input"
+
+pass "static call introduced by the patch carried in, or rejected for a module key"
diff --git a/tools/objtool/tests/generic/test-static-local-uncorrelated.sh b/tools/objtool/tests/generic/test-static-local-uncorrelated.sh
new file mode 100755
index 000000000000..d7711587d2d4
--- /dev/null
+++ b/tools/objtool/tests/generic/test-static-local-uncorrelated.sh
@@ -0,0 +1,41 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Some static locals must not be correlated with their counterparts in the
+# running kernel; the patched code has to use a fresh copy instead.
+#
+# .data..once holds the "have we warned yet" flags behind WARN_ONCE. Correlate
+# one and the patched function inherits the flag from before the patch, so the
+# warning the patch was written to produce never fires. The same goes for the
+# names the kernel generates for per-instance state -- __warned, __key,
+# __func__ and friends.
+#
+# Both directions matter, so an ordinary static local is here too: a rule that
+# refuses to correlate anything would pass a test that only checks the
+# refusals.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_local_uncorrelated.c
+run_diff
+
+# Compilers mangle static locals differently -- gcc gives __key.1, Clang
+# target.__key -- so match on the base name.
+
+# Correlated: referenced through a klp symbol, pointing at the kernel's copy.
+out_symbols | grep -q '\.klp\.sym\..*ordinary' ||
+ fail "ordinary static local was not correlated"
+
+# Not correlated: no klp symbol, and a copy cloned into the patch instead.
+out_symbols | grep -q '\.klp\.sym\..*__key' &&
+ fail "__key was correlated; it must use a fresh copy"
+# .sbss/.sdata on the architectures with a small-data area.
+out_sections | grep -qE '\.s?(bss|data)[^ ]*__key' ||
+ fail "__key was neither correlated nor cloned"
+
+out_symbols | grep -q '\.klp\.sym\..*once_flag' &&
+ fail ".data..once variable was correlated; it must use a fresh copy"
+assert_section '.data..once'
+
+pass "per-instance static locals cloned, ordinary ones correlated"
diff --git a/tools/objtool/tests/generic/test-static-local.sh b/tools/objtool/tests/generic/test-static-local.sh
new file mode 100755
index 000000000000..d57efa64dfd9
--- /dev/null
+++ b/tools/objtool/tests/generic/test-static-local.sh
@@ -0,0 +1,24 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A static local must be correlated with the original, not duplicated: a second
+# copy would discard the state the running kernel accumulated.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair static_local.c
+
+in_symbols orig.o | grep -q 'counter' ||
+ probe_skip "compiler emitted no distinct static local symbol"
+
+run_diff
+assert_patched target
+
+out_symbols | grep -q '\.klp\.sym\..*\.counter' ||
+ fail "static local not referenced through a klp relocation"
+
+out_symbols | grep 'counter' | grep -qvE 'UND|\.klp\.(sym|tombstone)' &&
+ fail "static local was given a fresh definition"
+
+pass "static local correlated rather than duplicated"
diff --git a/tools/objtool/tests/generic/test-switch-rodata.sh b/tools/objtool/tests/generic/test-switch-rodata.sh
new file mode 100755
index 000000000000..fb27e96c65f6
--- /dev/null
+++ b/tools/objtool/tests/generic/test-switch-rodata.sh
@@ -0,0 +1,53 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A Clang switch jump table travels with the function it belongs to.
+#
+# For a dense enough switch Clang emits the targets as a table in
+# .rodata..Lswitch.table.<function>, named after the function but not part of
+# it. klp diff has to associate the two: the patched function indexes into
+# that table, so a clone which does not bring it along jumps through whatever
+# the kernel's copy holds -- which, when the patch changed the switch, is the
+# wrong set of targets.
+#
+# That is an indirect jump to a stale address, not a missing symbol, so nothing
+# reports it at build or load time.
+#
+# objtool has no switch-specific code: the table is carried by the general
+# mechanism for data a cloned function references. So this is a regression
+# test on that mechanism reaching a shape it is easy to get wrong, not a guard
+# on a particular line -- making the table uncorrelated, the nearest sabotage,
+# does not change the outcome.
+#
+# Covers the same ground as corpus/x86_64-llvm-switch-rodata/
+# clang-switch-rodata-assoc in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+clang_only "only Clang emits switch jump tables in their own section"
+
+setup
+build_pair switch_rodata.c
+
+# The premise: this Clang really did build a table rather than a chain of
+# comparisons, and the added case really did change it.
+tbl=.rodata..Lswitch.table.status_to_string
+has_input_section orig.o "$tbl" ||
+ probe_skip "this clang built no jump table for the switch"
+# readelf prefixes each line with "[nn]", which splits into one or two fields
+# depending on the index, so strip it before counting columns.
+tbl_size()
+{
+ in_sections "$1" | sed 's/^ *\[[ 0-9]*\] *//' |
+ awk -v s="$tbl" '$1 == s { print $5 }'
+}
+[ "$(tbl_size orig.o)" != "$(tbl_size patched.o)" ] ||
+ fail "fixture's added case did not change the jump table"
+
+run_diff
+
+assert_patched status_to_string
+assert_section "$tbl"
+assert_reloc_sym .text.status_to_string "$tbl"
+
+pass "Clang switch jump table carried with the function it belongs to"
diff --git a/tools/objtool/tests/generic/test-symid-discarded.sh b/tools/objtool/tests/generic/test-symid-discarded.sh
new file mode 100755
index 000000000000..388a24984ec0
--- /dev/null
+++ b/tools/objtool/tests/generic/test-symid-discarded.sh
@@ -0,0 +1,44 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# .klp.symid must not reference symbols in sections the vmlinux link discards.
+# Each such section has been its own bug, found only when someone built a
+# config where a duplicate happened to land there, so cover the whole list
+# rather than whichever one was reported last.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# Allocated sections which vmlinux.lds.h discards unconditionally. A symid
+# referencing one of these fails the vmlinux link outright:
+#
+# `__exitcall_hid_exit' referenced in section `.klp.symid' of vmlinux.o:
+# defined in discarded section `.exitcall.exit' of vmlinux.o
+for sec in .exitcall.exit .no_trim_symbol; do
+ build_one symid_discarded.c a.o \
+ -DFUNC_NAME=use_a -DDISCARDED_SEC="\"$sec\""
+ build_one symid_discarded.c b.o \
+ -DFUNC_NAME=use_b -DDISCARDED_SEC="\"$sec\""
+
+ # --klp-symids only runs on a file named vmlinux.o
+ rm -f "$workdir/vmlinux.o"
+ partial_link "$workdir/vmlinux.o" "$workdir/a.o" "$workdir/b.o" ||
+ probe_skip "partial link unavailable"
+
+ "$OBJTOOL" --klp-symids --link "$workdir/vmlinux.o" ||
+ fail "objtool --klp-symids failed"
+
+ symids="$(in_relocs vmlinux.o |
+ awk '/rela.klp.symid/,/^$/')"
+
+ # Without this the test would also pass if symid generation stopped
+ # entirely.
+ echo "$symids" | grep -q 'dup_normal' ||
+ fail "$sec: no symid for the duplicate in a live section"
+
+ echo "$symids" | grep -q 'dup_discarded' &&
+ fail "symid emitted for a symbol in discarded section $sec"
+done
+
+pass "no symids for symbols in discarded sections"
diff --git a/tools/objtool/tests/generic/test-sympos-vmlinux.sh b/tools/objtool/tests/generic/test-sympos-vmlinux.sh
new file mode 100755
index 000000000000..b2b44001446f
--- /dev/null
+++ b/tools/objtool/tests/generic/test-sympos-vmlinux.sh
@@ -0,0 +1,57 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# sympos for vmlinux, which is resolved differently from sympos for a module.
+#
+# A module's .ko preserves symbol table order, so klp diff can count -- that is
+# what test-sympos covers. vmlinux cannot be counted: the final link reorders
+# sub-sections, so the order in vmlinux.o is not the order the running kernel
+# has. klp diff bridges that with .klp.symid, a table of { id, address }
+# emitted into vmlinux.o whose addresses the linker resolves, read back out of
+# the linked vmlinux.
+#
+# Getting it wrong points the relocation at a different symbol of the same
+# name. Nothing fails to build or load; the patched code uses the wrong
+# object.
+#
+# The fixture is arranged so the two answers differ: the static that comes
+# first in the symbol table is placed at the *higher* address, so counting
+# gives 1 and reading the linked image gives 2. Without that, both paths agree
+# and the test cannot tell them apart.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# use_a's static sorts last by section name, use_b's first. Only use_a is
+# patched, so exactly one sympos comes out.
+build_one sympos_vmlinux.c orig_a.o -DFUNC_NAME=use_a -DVARSEC='".data.zzz"'
+build_one sympos_vmlinux.c patched_a.o -DFUNC_NAME=use_a -DVARSEC='".data.zzz"' -DPATCHED
+build_one sympos_vmlinux.c b.o -DFUNC_NAME=use_b -DVARSEC='".data.aaa"' -DNO_MODINFO
+
+make_vmlinux_pair "$workdir/orig_a.o" "$workdir/b.o" \
+ -- "$workdir/patched_a.o" "$workdir/b.o"
+
+[ "$(count_input_symbols vmlinux.o dup_counter)" = 2 ] ||
+ fail "fixture did not produce two dup_counter symbols"
+has_input_section vmlinux.o .klp.symid ||
+ fail "objtool --klp-symids emitted no .klp.symid table"
+has_input_section vmlinux .klp.symid ||
+ fail ".klp.symid did not survive the link"
+
+# The premise: symbol table order and address order must disagree, or the test
+# proves nothing.
+first_addr="$(in_symbols vmlinux | awk '$8 == "dup_counter" { print $2; exit }')"
+low_addr="$(in_symbols vmlinux | awk '$8 == "dup_counter" { print $2 }' | sort | head -1)"
+[ "$first_addr" != "$low_addr" ] ||
+ probe_skip "linker did not reorder the two statics"
+
+assert_input_symbol dup_counter
+run_diff
+
+# Address order says 2. Counting symbol table order would say 1.
+assert_klp_sympos dup_counter 2
+out_symbols | grep -q 'dup_counter,1' &&
+ fail "sympos 1 emitted: counted symbol table order instead of reading the linked vmlinux"
+
+pass "vmlinux sympos taken from the linked image, not from symbol table order"
diff --git a/tools/objtool/tests/generic/test-sympos.sh b/tools/objtool/tests/generic/test-sympos.sh
new file mode 100755
index 000000000000..b71d4930a22a
--- /dev/null
+++ b/tools/objtool/tests/generic/test-sympos.sh
@@ -0,0 +1,51 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# sympos is what livepatch uses to tell duplicate symbol names apart in the
+# patched object: which "dup_counter" of several the relocation means. Get it
+# wrong and the patch resolves to the wrong object at load time, silently.
+#
+# klp_find_sympos() reports 0 when a name is unique and a 1-based position when
+# it is not, so both need checking -- always reporting a position, or never,
+# each looks right in one of the two cases.
+#
+# This is the module path, counting symbol table order. vmlinux is reordered
+# by the final link and goes through .klp.symid instead; that needs a linked
+# vmlinux next to vmlinux.o and is not covered here.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# One copy: the name is unique, so there is nothing to disambiguate.
+build_one sympos_dup.c orig.o -DFUNC_NAME=use_a
+build_one sympos_dup.c patched.o -DFUNC_NAME=use_a -DPATCHED
+run_diff
+
+assert_klp_sympos dup_counter 0
+
+# Two copies: positions, in symbol table order.
+for p in "" "-DPATCHED"; do
+ # shellcheck disable=SC2086
+ build_one sympos_dup.c "a$p.o" -DFUNC_NAME=use_a $p
+ # shellcheck disable=SC2086
+ build_one sympos_dup.c "b$p.o" -DFUNC_NAME=use_b -DNO_MODINFO $p
+done
+partial_link "$workdir/orig.o" "$workdir/a.o" "$workdir/b.o" ||
+ probe_skip "partial link unavailable"
+partial_link "$workdir/patched.o" "$workdir/a-DPATCHED.o" "$workdir/b-DPATCHED.o" ||
+ probe_skip "partial link unavailable"
+
+# Without duplicates in the input there is nothing for sympos to number.
+[ "$(count_input_symbols orig.o dup_counter)" = 2 ] ||
+ fail "fixture did not produce two dup_counter symbols"
+
+run_diff
+
+assert_klp_sympos dup_counter 1
+assert_klp_sympos dup_counter 2
+# ... and nothing still claiming the name is unique
+out_symbols | grep -qE '\.klp\.sym\.[^.]+\.dup_counter,0([[:space:]]|$)' &&
+ fail "sympos 0 emitted for a duplicated symbol"
+
+pass "sympos numbers duplicate symbols and stays 0 for unique ones"
diff --git a/tools/objtool/tests/generic/test-symvers-parse-error.sh b/tools/objtool/tests/generic/test-symvers-parse-error.sh
new file mode 100755
index 000000000000..d597f28e5617
--- /dev/null
+++ b/tools/objtool/tests/generic/test-symvers-parse-error.sh
@@ -0,0 +1,23 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A malformed Module.symvers has to be reported against the line it is on.
+# Module.symvers has tens of thousands of lines and is generated, so a wrong
+# line number sends whoever has to fix it to the wrong place, and "line 1" is
+# wrong in a way that looks plausible.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair basic.c
+
+# Three well-formed lines, then one with no tabs at all.
+export_syms a b c
+echo 'this line has no fields' >> "$workdir/Module.symvers"
+
+run_diff 255
+
+assert_diff_log 'malformed Module.symvers'
+assert_diff_log 'at line 4'
+
+pass "malformed Module.symvers reported against the offending line"
diff --git a/tools/objtool/tests/generic/test-thinlto-ambiguity.sh b/tools/objtool/tests/generic/test-thinlto-ambiguity.sh
new file mode 100755
index 000000000000..34f4f3a58e20
--- /dev/null
+++ b/tools/objtool/tests/generic/test-thinlto-ambiguity.sh
@@ -0,0 +1,77 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Two ThinLTO-promoted symbols sharing a demangled name must be paired up
+# correctly.
+#
+# A file-local symbol which ThinLTO has to make visible is renamed
+# helper.llvm.<hash>. With two such helpers in one link the original and the
+# patched object hold two each, all four spelled differently, and demangling
+# gives "helper" for all of them -- so the name is not enough to say which
+# corresponds to which.
+#
+# Getting it wrong is silent and specific: the patch is built against the wrong
+# body, so one call site gets the other helper's arithmetic. Nothing fails to
+# build and nothing fails to load.
+#
+# test-thinlto-local covers the unambiguous case, one promoted symbol whose
+# hash moved. This is the case where demangling alone is not an answer.
+#
+# The outcome is asserted, not the machinery: with the clang tested here the
+# pairing succeeds even with the .llvm.<hash> suffix map disabled and with
+# llvm_suffix() stubbed out, so no single-line sabotage distinguishes it. The
+# tiered matcher this case was written for is not needed for this shape.
+#
+# Covers the same ground as corpus/x86_64-llvm-thinlto/
+# thin-lto-demangled-ambiguity and thin-lto-demangled-global-match in Joe
+# Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+clang_only "ThinLTO requires clang"
+
+find_thinlto_toolchain ||
+ probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_LD to one"
+
+build_thinlto() # $1 output object, $2 extra flags
+{
+ local t
+ for t in "" -DTU_B -DTU_C; do
+ $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections \
+ $2 $t -c "$FIXTURES_DIR/thinlto_ambiguity.c" \
+ -o "$workdir/tu$t.o" 2>/dev/null || return 1
+ done
+ "$THIN_LD" -r "$workdir/tu.o" "$workdir/tu-DTU_B.o" "$workdir/tu-DTU_C.o" \
+ -o "$1" 2>/dev/null || return 1
+}
+
+build_thinlto "$workdir/orig.o" "" ||
+ probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)"
+build_thinlto "$workdir/patched.o" -DPATCHED ||
+ probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)"
+
+# The premise: two promoted helpers per object, and exactly one of them kept
+# its hash -- the one the patch did not touch. Without that there is nothing
+# to disambiguate.
+orig_syms="$(in_symbols orig.o | grep -oE 'helper\.llvm\.[0-9]+' | sort -u)"
+new_syms="$( in_symbols patched.o | grep -oE 'helper\.llvm\.[0-9]+' | sort -u)"
+[ "$(echo "$orig_syms" | wc -l)" = 2 ] && [ "$(echo "$new_syms" | wc -l)" = 2 ] ||
+ probe_skip "ThinLTO did not promote two distinct helpers here"
+
+kept="$(comm -12 <(echo "$orig_syms") <(echo "$new_syms"))"
+moved="$(comm -13 <(echo "$orig_syms") <(echo "$new_syms"))"
+[ "$(echo "$kept" | wc -w)" = 1 ] && [ "$(echo "$moved" | wc -w)" = 1 ] ||
+ probe_skip "expected one helper to keep its hash and one to move"
+
+run_diff
+
+# Exactly one helper is cloned, and it is the one whose body changed. Cloning
+# the other, or both, is what a wrong pairing looks like.
+assert_not_patched "$kept"
+
+n="$(out_sections | grep -cE '[[:space:]]\.text\.helper\.llvm\.[0-9]+[[:space:]]')"
+[ "$n" = 1 ] ||
+ fail "expected 1 cloned helper, found $n"
+
+pass "ThinLTO helpers sharing a demangled name paired up correctly"
diff --git a/tools/objtool/tests/generic/test-thinlto-local.sh b/tools/objtool/tests/generic/test-thinlto-local.sh
new file mode 100755
index 000000000000..a266263676ed
--- /dev/null
+++ b/tools/objtool/tests/generic/test-thinlto-local.sh
@@ -0,0 +1,48 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Correlating ThinLTO-promoted locals requires demangling the .llvm.<hash>
+# suffix, and the resulting klp relocation must name the original symbol: that
+# is the one in the running kernel's kallsyms.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+clang_only "ThinLTO requires clang"
+
+find_thinlto_toolchain ||
+ probe_skip "no matching clang/lld pair for a ThinLTO link; set THIN_LD to one"
+
+build_thinlto() # $1 output object, $2 extra flags
+{
+ $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections $2 \
+ -c "$FIXTURES_DIR/thinlto_local.c" -o "$workdir/tu_a.o" 2>/dev/null || return 1
+ $THIN_CC -flto=thin -O2 -ffunction-sections -fdata-sections $2 -DTU_B \
+ -c "$FIXTURES_DIR/thinlto_local.c" -o "$workdir/tu_b.o" 2>/dev/null || return 1
+ "$THIN_LD" -r "$workdir/tu_a.o" "$workdir/tu_b.o" -o "$1" 2>/dev/null || return 1
+}
+
+build_thinlto "$workdir/orig.o" "" ||
+ probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)"
+build_thinlto "$workdir/patched.o" -DPATCHED ||
+ probe_skip "ThinLTO build failed ($THIN_CC, $THIN_LD)"
+
+orig_sym="$(in_symbols orig.o | grep -o 'counter\.llvm\.[0-9]*' | head -1)"
+new_sym="$( in_symbols patched.o | grep -o 'counter\.llvm\.[0-9]*' | head -1)"
+
+[ -n "$orig_sym" ] && [ -n "$new_sym" ] ||
+ probe_skip "$THIN_CC did not promote the local symbol"
+
+# Equal hashes would make plain name matching work, testing nothing.
+[ "$orig_sym" != "$new_sym" ] ||
+ probe_skip "$THIN_CC gave the same ThinLTO hash for both builds"
+
+run_diff
+assert_patched target
+
+out_symbols | grep -q "\.klp\.sym\.vmlinux\.$orig_sym," ||
+ fail "expected a klp relocation naming $orig_sym"
+out_symbols | grep -q "\.klp\.sym\.vmlinux\.$new_sym," &&
+ fail "klp relocation names $new_sym, which the running kernel does not have"
+
+pass "ThinLTO-mangled local correlated across differing hashes ($THIN_CC, $THIN_LD)"
diff --git a/tools/objtool/tests/generic/test-ubsan-noise.sh b/tools/objtool/tests/generic/test-ubsan-noise.sh
new file mode 100755
index 000000000000..b415eb16bfd2
--- /dev/null
+++ b/tools/objtool/tests/generic/test-ubsan-noise.sh
@@ -0,0 +1,48 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# UBSAN instrumentation in an unchanged function must not make it look changed.
+#
+# Every instrumented operation gets a per-callsite metadata object in an
+# anonymous data section -- .data..Lubsan_data and .data..Lubsan_type from GCC,
+# .data..L__unnamed_ from Clang -- whose names are compiler-generated and mean
+# nothing across a rebuild. is_uncorrelated_section() exists so klp diff does
+# not try to pair them up.
+#
+# Without that, the metadata belonging to a function nobody touched compares as
+# different and drags the function into the patch. A livepatch which replaces
+# functions the patch never changed is not a build failure: it is a larger
+# patch than intended, taking its dependencies with it, and every extra
+# function is one more that can fail to correlate or to apply.
+#
+# Covers the same ground as corpus/x86_64-ubsan/{ubsan-shift-noise,
+# ubsan-metadata-data-section,gcc-ubsan-anonymous-data,ubsan-handler-cloning}
+# and corpus/x86_64-llvm-ubsan/{clang-ubsan-bounds-noise,
+# clang-ubsan-handler-cloning} in Joe Lawrence's klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair ubsan_noise.c -fsanitize=shift
+
+# The premise: this compiler really did instrument, and left its metadata in an
+# anonymous section. Without that the test is just test-basic again.
+ubsan_sec="$(in_sections orig.o |
+ grep -oE '\.data\.\.L(ubsan_data|__unnamed_)[A-Za-z0-9_.]*' | head -1)"
+[ -n "$ubsan_sec" ] ||
+ probe_skip "compiler emitted no anonymous UBSAN data section"
+assert_input_symbol untouched
+
+run_diff
+
+# The changed function is patched, and the untouched one is left alone despite
+# carrying instrumentation of its own.
+assert_patched touched
+assert_not_patched untouched
+
+# The handler the patched code calls has to come with it, or the clone calls
+# nothing when its check fires.
+out_symbols | grep -q '__ubsan_handle_' ||
+ fail "no __ubsan_handle_* reference in the patched output"
+
+pass "UBSAN metadata in an unchanged function does not drag it into the patch"
diff --git a/tools/objtool/tests/lib.sh b/tools/objtool/tests/lib.sh
new file mode 100644
index 000000000000..4da168d0ddca
--- /dev/null
+++ b/tools/objtool/tests/lib.sh
@@ -0,0 +1,911 @@
+# SPDX-License-Identifier: GPL-2.0
+#
+# Helpers for the objtool klp tests. A test builds a fixture twice, as the
+# original and (with -DPATCHED) the patched object, runs both through
+# "klp checksum" and diffs them, then asserts on the result.
+#
+# Assertions check properties rather than compare against recorded output:
+# codegen varies between compilers and golden files would report churn instead
+# of regressions.
+
+set -u
+
+TESTS_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+
+# Tests live in generic/ or in an architecture directory beside it, and each
+# carries its own fixtures.
+FIXTURES_DIR="$(cd "$(dirname "$0")/fixtures" 2>/dev/null && pwd)"
+
+# The kernel's convention: CROSS_COMPILE is the one knob, with per-tool
+# overrides for what it does not cover. objtool itself is always a host binary
+# -- it is built with HOSTCC and only reads ELF -- so an arm64 machine can run
+# the x86 tests against x86 objects given a compiler that emits them.
+#
+# readelf reads any target, so it rarely needs overriding, and either GNU
+# readelf or llvm-readelf will do: the assertions match on fields rather than
+# on columns, and where the two spell something differently -- "OS [0xff20]"
+# against "OS[0xff20]" for SHN_LIVEPATCH -- they accept both. BFD's objcopy is
+# usually built for the host's target alone, and llvm-objcopy is the
+# target-agnostic replacement.
+CROSS_COMPILE="${CROSS_COMPILE:-}"
+CC="${CC:-${CROSS_COMPILE}gcc}"
+LD="${LD:-${CROSS_COMPILE}ld}"
+READELF="${READELF:-${CROSS_COMPILE}readelf}"
+OBJCOPY="${OBJCOPY:-${CROSS_COMPILE}objcopy}"
+
+OBJTOOL="${OBJTOOL:-$TESTS_DIR/../objtool}"
+
+# klp_preflight
+#
+# Check the environment once, before any test runs, and report what was found.
+#
+klp_preflight()
+{
+ local tmp tool cc_version arch host cc_arch
+
+ bail() { echo "Bail out! $*" >&2; exit 1; }
+
+ # A relative $OBJTOOL is relative to the objtool directory, not tests/.
+ [ -x "$OBJTOOL" ] || [ ! -x "$TESTS_DIR/../$OBJTOOL" ] ||
+ OBJTOOL="$TESTS_DIR/../$OBJTOOL"
+
+ [ -x "$OBJTOOL" ] ||
+ bail "objtool not found at '$OBJTOOL' -- build it first"
+
+ # run_diff() runs objtool from inside the test's working directory, so
+ # a relative path would resolve against that instead.
+ OBJTOOL="$(realpath "$OBJTOOL")"
+
+ "$OBJTOOL" klp 2>&1 | grep -q checksum ||
+ bail "objtool was built without klp support; install libxxhash (>= 0.8) and rebuild"
+
+ command -v "${CC%% *}" >/dev/null || bail "compiler not found: $CC"
+
+ for tool in "$READELF" "$OBJCOPY" "$LD"; do
+ command -v "${tool%% *}" >/dev/null || bail "$tool not found"
+ done
+
+ tmp="$(mktemp -d)" || bail "mktemp failed"
+ echo 'int probe(void) { return 0; }' > "$tmp/probe.c"
+ $CC -c -o "$tmp/probe.o" "$tmp/probe.c" 2>/dev/null ||
+ { rm -rf "$tmp"; bail "$CC cannot compile a trivial object"; }
+
+ # $CC, $ARCH and objtool have to agree about the target, and cross runs
+ # are where they stop agreeing: plain "CC=clang ARCH=x86_64" on an arm64
+ # box selects the x86 tests and then builds arm64 objects, because clang
+ # needs --target= to emit anything but the host's.
+ #
+ # Ask objtool rather than comparing machine names. It rejects an object
+ # it was not built for -- "unexpected ELF machine type" -- so one check
+ # covers every way the three can disagree, and says so once instead of
+ # failing every test for the same reason.
+ "$OBJTOOL" klp checksum "$tmp/probe.o" >/dev/null 2>&1 ||
+ { rm -rf "$tmp"
+ bail "objtool rejects an object built by '$CC'; they target" \
+ "different architectures (set CROSS_COMPILE, or" \
+ "--target= for clang)"; }
+
+ # BFD objcopy is usually built for the host's target alone, and
+ # checksum_of() needs it to read the object under test.
+ $OBJCOPY -O binary --only-section=.text "$tmp/probe.o" "$tmp/probe.bin" 2>/dev/null ||
+ { rm -rf "$tmp"
+ bail "$OBJCOPY cannot read objects built by '$CC'; install" \
+ "binutils-multiarch or set OBJCOPY=llvm-objcopy"; }
+ # $ARCH only chooses which directory of tests runs, so it can disagree
+ # with what $CC builds without objtool noticing -- and the result is the
+ # wrong set of tests, quietly.
+ case "$($READELF -hW "$tmp/probe.o" | sed -n 's/.*Machine: *//p')" in
+ *X86-64*|*Intel*80386*) cc_arch=x86 ;;
+ *AArch64*) cc_arch=arm64 ;;
+ *) cc_arch= ;;
+ esac
+ rm -rf "$tmp"
+
+ # Normalize to the kernel's SRCARCH.
+ case "${ARCH:-$(uname -m)}" in
+ x86_64|i?86) arch=x86 ;;
+ aarch64*) arch=arm64 ;;
+ *) arch="${ARCH:-$(uname -m)}" ;;
+ esac
+
+ case "$(uname -m)" in
+ x86_64|i?86) host=x86 ;;
+ aarch64*) host=arm64 ;;
+ *) host="$(uname -m)" ;;
+ esac
+
+ [ -z "$cc_arch" ] || [ "$cc_arch" = "$arch" ] ||
+ bail "ARCH says $arch but '$CC' builds $cc_arch objects;" \
+ "the $arch tests would run against the wrong architecture"
+
+ KLP_TEST_ARCH="$arch"
+ KLP_TEST_PREFLIGHT=done
+ export OBJTOOL CC KLP_TEST_ARCH KLP_TEST_PREFLIGHT
+
+ cc_version="$($CC --version 2>/dev/null | head -1)"
+ cat <<EOF
+# preflight
+# objtool $OBJTOOL (klp: yes)
+# compiler $cc_version
+# arch $KLP_TEST_ARCH$([ "$arch" = "$host" ] || echo " (host $host, cross)")
+# tmpdir ${TMPDIR:-/tmp} (each test builds in a fresh directory here)
+EOF
+}
+
+[ -n "${KLP_TEST_PREFLIGHT:-}" ] || klp_preflight
+
+# What every fixture is built with. These describe the kernel a fixture stands
+# in for; -c is build_one's contract rather than a property of that kernel, so
+# it lives at the compile where an override cannot drop it.
+#
+# -O2 the kernel's default
+# -ffunction-sections -fdata-sections klp-build passes these itself, through
+# KCFLAGS, whatever the configuration
+# -fno-asynchronous-unwind-tables arch/x86/Makefile sets this always, so
+# kernel objects carry no .eh_frame
+# -fno-common the kernel's Makefile sets it, so an
+# uninitialised global there lands in
+# .bss rather than being SHN_COMMON,
+# which has no section and so no
+# checksum
+#
+# A test overrides it; see tools/objtool/Documentation/klp-write-tests.txt.
+FIXTURE_CFLAGS="-O2 -ffunction-sections -fdata-sections -fno-common \
+ -fno-asynchronous-unwind-tables"
+
+test_name="$(basename "$0" .sh)"
+workdir=
+
+# The pair run_diff() and the checksum helpers work on. build_pair() names
+# them again and make_vmlinux_pair() repoints orig_obj at the image it links,
+# but a test which builds its objects itself with build_one() sets neither, so
+# the default belongs here.
+orig_obj=orig.o
+patched_obj=patched.o
+
+pass() { KLP_TEST_REPORTED=1; echo "ok - $test_name${1:+: $1}"; exit 0; }
+fail() { KLP_TEST_FAILED=1; echo "not ok - $test_name: $*"; exit 1; }
+
+# Two kinds of skip, and the runner tells them apart.
+#
+# declared_skip the test said in advance it does not apply here, e.g.
+# gcc_only on a clang run. Expected indefinitely.
+# probe_skip the construct did not turn up in the built object this
+# time. Weaker: it may appear on another compiler version,
+# and one which becomes permanent is a fixture that quietly
+# stopped testing anything.
+#
+# A bare skip() is neither, and the runner counts it as a failure: a test which
+# gives up for a reason it never declared is a hole, not an outcome.
+declared_skip()
+{
+ KLP_TEST_REPORTED=1
+ echo "ok - $test_name # SKIP (declared) $*"
+ exit 0
+}
+
+probe_skip()
+{
+ KLP_TEST_REPORTED=1
+ echo "ok - $test_name # SKIP (probe) $*"
+ exit 0
+}
+skip() { echo "ok - $test_name # SKIP $*"; exit 0; }
+
+# TAP directives. A test which is known to fail reports it rather than being
+# commented out and forgotten, and one which starts passing again says so
+# instead of quietly going green: the expectation has to be removed by hand,
+# which is the point.
+xfail()
+{
+ KLP_TEST_REPORTED=1
+ echo "not ok - $test_name${1:+: $1} # TODO known failure"
+ exit 0
+}
+
+xpass()
+{
+ KLP_TEST_FAILED=1
+ echo "ok - $test_name${1:+: $1} # TODO expected failure, but passed"
+ exit 1
+}
+
+# cleanup [exit]
+#
+# Called with "exit" from the trap, when the test is over and what it built may
+# be worth keeping. Called bare by a test which has finished with one segment
+# and is about to setup() another: that one is done with, whatever the outcome
+# of the segments still to come, so it goes.
+cleanup()
+{
+ [ -n "$workdir" ] || return 0
+
+ # run-tests.sh exports KLP_TEST_KEEP, having validated it; a test run on
+ # its own reads KEEP itself, so the same setting means the same thing
+ # either way.
+ case "${KLP_TEST_KEEP:-${KEEP:-failed}}" in
+ all) return 0 ;;
+ none) rm -rf "$workdir" ;;
+ failed|*)
+ [ "${1:-}" = exit ] || {
+ rm -rf "$workdir"
+ return 0
+ }
+ # Keep what the runner is going to point at. It counts as a
+ # failure anything which did not report an expected outcome --
+ # including a test which died before printing one, and an
+ # undeclared skip -- and none of those set KLP_TEST_FAILED, so
+ # the question to ask is whether a result was reported at all.
+ # An exit status cannot answer it: a test killed by a signal
+ # runs this trap with the status of whatever ran last.
+ [ -n "${KLP_TEST_REPORTED:-}" ] && [ -z "${KLP_TEST_FAILED:-}" ] && {
+ rm -rf "$workdir"
+ return 0
+ }
+ # run on its own there is no runner to say where it was kept
+ [ -n "${KLP_TEST_WORKDIR:-}" ] ||
+ echo "# kept $workdir"
+ ;;
+ esac
+}
+
+# setup [exported symbol...]
+setup()
+{
+ if [ -n "${KLP_TEST_WORKDIR:-}" ]; then
+ workdir="$KLP_TEST_WORKDIR"
+ mkdir -p "$workdir" || fail "cannot create $workdir"
+ else
+ workdir="$(mktemp -d)" || fail "mktemp failed"
+ fi
+ trap 'cleanup exit' EXIT
+
+ export_syms "$@"
+}
+
+# export_syms [symbol...]
+#
+# Rewrite Module.symvers so exactly these symbols are exported by vmlinux.
+# Whether a symbol is listed decides between an ordinary relocation and a klp
+# relocation, so tests flip it to cover both.
+export_syms()
+{
+ : > "$workdir/Module.symvers"
+ add_exports vmlinux "$@"
+}
+
+# add_exports <object> [symbol...]
+#
+# Append exports owned by one object, without clearing what is already there,
+# so a test can describe a kernel where several objects export things.
+#
+# Which object owns a symbol is not cosmetic: a reference to a vmlinux symbol
+# is applied when the patch module loads, and a reference to a module's symbol
+# when that patched module loads, so klp diff files them in different sections.
+add_exports()
+{
+ local owner="$1"; shift
+
+ add_exports_ns "$owner" "" "$@"
+}
+
+# add_exports_ns <object> <namespace> [symbol...]
+#
+# Exports in a symbol namespace, the last field of a Module.symvers line.
+#
+# A "module:<names>" namespace is EXPORT_SYMBOL_FOR_MODULES(), where the module
+# loader grants access by matching the importing module's name against the
+# list. A livepatch module is never on that list, so such a symbol has to be
+# referenced the way an unexported one is. Ordinary namespaces are not
+# special here.
+add_exports_ns()
+{
+ local owner="$1" ns="$2"; shift 2
+
+ local sym
+
+ for sym in "$@"; do
+ printf '0x00000000\t%s\t%s\tEXPORT_SYMBOL\t%s\n' \
+ "$sym" "$owner" "$ns" >> "$workdir/Module.symvers"
+ done
+}
+
+# gcc_only / clang_only <reason>
+gcc_only()
+{
+ case "$($CC --version 2>/dev/null | head -1)" in
+ *[Gg][Cc][Cc]*) return 0 ;;
+ esac
+ declared_skip "gcc only${1:+: $1}"
+}
+
+clang_only()
+{
+ case "$($CC --version 2>/dev/null | head -1)" in
+ *clang*) return 0 ;;
+ esac
+ declared_skip "clang only${1:+: $1}"
+}
+
+# build_one <fixture.c> <output object> [cflags...]
+build_one()
+{
+ local fixture out
+ fixture="$FIXTURES_DIR/$1"
+ out="$workdir/$2"
+ shift 2
+
+ [ -f "$fixture" ] || fail "missing fixture $fixture"
+
+ # run_checksum only runs once per workdir. A fresh object has no
+ # checksums in it, so anything built now needs that to happen again.
+ rm -f "$workdir/.checksummed"
+
+ $CC -c $FIXTURE_CFLAGS "$@" -o "$out" "$fixture" 2>"$workdir/cc.log" ||
+ fail "$(basename "$fixture") does not build: $(tail -1 "$workdir/cc.log")"
+}
+
+# build_pair <fixture.c> [cflags...]
+build_pair()
+{
+ local fixture="$1"; shift
+
+ # Name what this builds. A test may run several segments, and
+ # make_vmlinux_pair() repoints orig_obj at the image it links, so
+ # without this the next run_diff() would still be reading that.
+ orig_obj=orig.o
+ patched_obj=patched.o
+
+ build_one "$fixture" orig.o "$@"
+ build_one "$fixture" patched.o "$@" -DPATCHED
+}
+
+# run_objtool_check <objtool arguments...>
+#
+# Run objtool's ordinary check pass over the pair, as the kernel build does.
+#
+# Some of what klp diff consumes is produced by this pass rather than by the
+# compiler: .static_call_sites, .mcount_loc, .ibt_endbr_seal, ORC.
+#
+# Only module objects see it before klp-build -- with CONFIG_KLP_BUILD the
+# per-object pass is deferred, so built-in objects reach klp diff exactly as
+# the compiler left them.
+run_objtool_check()
+{
+ local obj
+
+ # This rewrites both objects, so checksums taken before it describe
+ # something that no longer exists. As in build_one(), drop the marker
+ # so run_checksum() takes them again.
+ rm -f "$workdir/.checksummed"
+
+ for obj in "$orig_obj" "$patched_obj"; do
+ "$OBJTOOL" "$@" "$workdir/$obj" ||
+ fail "objtool $* failed on $obj"
+ done
+}
+
+run_checksum()
+{
+ # Checksums live in the objects, and a test may ask for them more than
+ # once -- diffing the same pair again with a different Module.symvers,
+ # say. objtool does the right thing when asked twice, leaving the
+ # object alone, but it says so, and that warning would be most of what
+ # a passing run prints. Remember instead, and keep quiet.
+ [ -e "$workdir/.checksummed" ] && return 0
+
+ "$OBJTOOL" klp checksum "$workdir/$orig_obj" ||
+ fail "klp checksum $orig_obj failed"
+ "$OBJTOOL" klp checksum "$workdir/$patched_obj" ||
+ fail "klp checksum $patched_obj failed"
+ touch "$workdir/.checksummed"
+}
+
+# run_diff [expected exit status]
+run_diff()
+{
+ local expect="${1:-0}" rc=0
+
+ run_checksum
+
+ # klp diff looks for Module.symvers relative to the working directory.
+ ( cd "$workdir" && "$OBJTOOL" klp diff "$orig_obj" "$patched_obj" out.o ) \
+ > "$workdir/diff.log" 2>&1 || rc=$?
+
+ [ "$rc" = "$expect" ] ||
+ fail "klp diff exited $rc, expected $expect: $(tail -2 "$workdir/diff.log")"
+}
+
+cc_supports()
+{
+ echo 'int f(void) { return 0; }' > "$workdir/flagtest.c"
+ $CC $1 -c "$workdir/flagtest.c" -o "$workdir/flagtest.o" 2>/dev/null
+}
+
+# partial_link <output> <object...>
+#
+# "ld -r" through the compiler driver so the link targets the same
+# architecture as the objects.
+partial_link()
+{
+ local out="$1"; shift
+
+ rm -f "$workdir/.checksummed"
+
+ $CC -r -nostdlib -o "$out" "$@" 2>/dev/null ||
+ $CC -r -nostdlib -fuse-ld=lld -o "$out" "$@" 2>/dev/null
+}
+
+# link_vmlinux <output> <object...>
+#
+# Link objects into an executable, the way the kernel's final link produces
+# vmlinux from vmlinux.o. Entry point 0 and no libc: nothing runs it, it only
+# has to be a linked image with resolved addresses.
+#
+# The sub-sections have to come out in name order rather than object order,
+# the way the kernel's linker script gathers .text.unlikely and .data.. apart
+# from the rest. That reordering is the entire reason .klp.symid exists: a
+# link which preserves order cannot tell a correct sympos from one that merely
+# counted, and the caller checks the two orders really did diverge.
+#
+# A linker script rather than --sort-section=name, because lld accepts that
+# option and ignores it -- so on a host where only lld can link the target, the
+# test would quietly stop testing the thing it is named for.
+#
+# Three attempts because a cross run has neither $LD nor the compiler's default
+# linker able to touch the target: on an arm64 host linking x86 objects, only
+# lld will do it.
+link_vmlinux()
+{
+ local out="$1" lds="$workdir/sort.lds"; shift
+
+ echo 'SECTIONS { .data : { *(SORT_BY_NAME(.data.*)) } }' > "$lds"
+
+ $LD -e 0 -T "$lds" -o "$out" "$@" 2>/dev/null ||
+ $CC -nostdlib -Wl,-e,0 -Wl,-T,"$lds" \
+ -o "$out" "$@" 2>/dev/null ||
+ $CC -nostdlib -fuse-ld=lld -Wl,-e,0 -Wl,-T,"$lds" \
+ -o "$out" "$@" 2>/dev/null
+}
+
+# make_vmlinux_pair <orig object...> -- <patched object...>
+#
+# Build the vmlinux.o / vmlinux pair klp diff needs to resolve sympos the way
+# it does for built-in code, and point the diff at it.
+#
+# For a module, sympos is a count in symbol table order, which klp diff can do
+# from the object alone. vmlinux is different: the final link reorders
+# sub-sections, so the position comes from the linked image, bridged by
+# .klp.symid. klp diff only looks for that when the object it was handed is
+# called vmlinux.o and a vmlinux sits beside it -- so both the name and the
+# linked image matter.
+make_vmlinux_pair()
+{
+ local orig=() patched=() seen= arg
+
+ for arg in "$@"; do
+ if [ "$arg" = -- ]; then seen=y; continue; fi
+ if [ -n "$seen" ]; then patched+=( "$arg" ); else orig+=( "$arg" ); fi
+ done
+
+ # Both sides have to have been named. Without this, forgetting the --
+ # leaves one list empty, the link of nothing fails, and the test skips
+ # saying the toolchain cannot link -- which is a test bug wearing the
+ # costume of an environment one.
+ [ "${#orig[@]}" -gt 0 ] && [ "${#patched[@]}" -gt 0 ] ||
+ fail "make_vmlinux_pair needs objects either side of --"
+
+ partial_link "$workdir/vmlinux.o" "${orig[@]}" ||
+ probe_skip "partial link unavailable"
+ partial_link "$workdir/patched.o" "${patched[@]}" ||
+ probe_skip "partial link unavailable"
+
+ "$OBJTOOL" --klp-symids --link "$workdir/vmlinux.o" ||
+ fail "objtool --klp-symids failed"
+
+ link_vmlinux "$workdir/vmlinux" "$workdir/vmlinux.o" ||
+ probe_skip "cannot link a vmlinux here"
+
+ orig_obj=vmlinux.o
+}
+
+# build_module_pair <fixture.c> <module name> [cflags...]
+#
+# Build the pair as objects belonging to a module rather than to vmlinux. klp
+# diff reads the object's module name from .modinfo, and that decides which
+# object a relocation is attributed to and whether a reference counts as
+# cross-module, so a good deal of the code has a module path the vmlinux
+# fixtures never reach.
+#
+# The fixture defines its .modinfo name from MODNAME. Passing that through
+# -D needs two levels of quoting, which is easy to get wrong at the call site.
+build_module_pair()
+{
+ local fixture="$1" modname="$2"; shift 2
+
+ build_pair "$fixture" -DMODNAME="\"$modname\"" "$@"
+}
+
+# find_thinlto_toolchain
+#
+# Set $THIN_LD to an lld from the same LLVM release as $CC (or THIN_CC). A
+# mismatched pair fails with "Invalid summary version", which reads like a
+# broken test rather than a broken environment.
+#
+# ThinLTO is clang-only; callers must use clang_only before calling this.
+# Only $CC (or an explicit THIN_CC override) is consulted -- the harness does
+# not search for a second compiler beside a gcc $CC.
+find_thinlto_toolchain()
+{
+ local cc ver ld
+
+ for cc in "${THIN_CC:-}" "$CC"; do
+ [ -n "$cc" ] || continue
+ command -v "${cc%% *}" >/dev/null 2>&1 || return 1
+
+ ver=$($cc -dumpversion 2>/dev/null | cut -d. -f1)
+
+ for ld in "${THIN_LD:-}" "ld.lld-$ver" ld.lld; do
+ [ -n "$ld" ] || continue
+ command -v "$ld" >/dev/null 2>&1 || continue
+
+ echo 'int probe(void) { return 0; }' > "$workdir/probe.c"
+ $cc -flto=thin -O2 -c "$workdir/probe.c" \
+ -o "$workdir/probe.o" 2>/dev/null || continue
+ "$ld" -r "$workdir/probe.o" -o "$workdir/probe.elf" \
+ 2>/dev/null || continue
+
+ THIN_CC="$cc"
+ THIN_LD="$ld"
+ return 0
+ done
+ done
+
+ return 1
+}
+
+out_sections() { $READELF -S -W "$workdir/out.o" 2>/dev/null; }
+out_relocs() { $READELF -r -W "$workdir/out.o" 2>/dev/null; }
+out_symbols() { $READELF -s -W "$workdir/out.o" 2>/dev/null; }
+diff_log() { cat "$workdir/diff.log"; }
+
+# out_strings <section>
+#
+# The strings in one section of the output, for the names livepatch matches on.
+out_strings() { $READELF -p "$1" "$workdir/out.o" 2>/dev/null; }
+
+# Checks on the input objects, to run before klp diff. The two forms differ in
+# what an absent construct means:
+#
+# require_* the compiler cannot produce it here -> skip
+# assert_* the fixture is supposed to produce it -> fail
+
+in_sections() { $READELF -S -W "$workdir/$1" 2>/dev/null; }
+in_symbols() { $READELF -s -W "$workdir/$1" 2>/dev/null; }
+in_relocs() { $READELF -r -W "$workdir/$1" 2>/dev/null; }
+
+# count_input_symbols <object> <name>
+#
+# How many object symbols of exactly that name the input has. Deliberately not
+# a grep: readelf lists section symbols too, and a newer binutils prints their
+# name -- ".data.<name>" -- where an older one leaves the column blank. A dot
+# is not a word character, so "grep -w <name>" counts that line as well, and
+# the same object gives a different answer depending on which readelf reads it.
+count_input_symbols()
+{
+ in_symbols "$1" | awk -v n="$2" '$4 == "OBJECT" && $8 == n' | wc -l
+}
+
+# re_quote <string>
+#
+# A string as a literal basic regular expression. Nearly every name these
+# assertions match on contains a dot -- .text.target, .klp.rela.vmlinux -- and
+# an unescaped dot matches any character, so an assertion for one section can be
+# satisfied by a different one whose name merely lines up.
+re_quote() { printf '%s' "$1" | sed 's|[].[^$*\\/]|\\&|g'; }
+
+has_input_section() { in_sections "$1" | grep -q "[[:space:]]$(re_quote "$2")[[:space:]]"; }
+has_input_symbol() { in_symbols "$1" | awk -v n="$2" '$NF == n' | grep -q .; }
+
+assert_input_section()
+{
+ local obj
+
+ for obj in "$orig_obj" "$patched_obj"; do
+ has_input_section "$obj" "$1" ||
+ fail "fixture produced no section '$1' in $obj"
+ done
+}
+
+assert_input_symbol()
+{
+ local obj
+
+ for obj in "$orig_obj" "$patched_obj"; do
+ has_input_symbol "$obj" "$1" ||
+ fail "fixture produced no symbol '$1' in $obj"
+ done
+}
+
+require_input_section()
+{
+ local obj
+
+ for obj in "$orig_obj" "$patched_obj"; do
+ has_input_section "$obj" "$1" ||
+ probe_skip "compiler produced no section '$1' here"
+ done
+}
+
+assert_section()
+{
+ out_sections | grep -q "[[:space:]]$(re_quote "$1")[[:space:]]" ||
+ fail "expected section '$1' in output"
+}
+
+assert_no_section()
+{
+ out_sections | grep -q "[[:space:]]$(re_quote "$1")[[:space:]]" &&
+ fail "unexpected section '$1' in output"
+ return 0
+}
+
+assert_patched()
+{
+ assert_section ".text.$1"
+}
+
+assert_not_patched()
+{
+ out_sections | grep -q "[[:space:]]$(re_quote ".text.$1")[[:space:]]" &&
+ fail "function '$1' should not have been cloned"
+ return 0
+}
+
+# section_relocs <section>
+#
+# The relocations against one section. readelf prints every relocation section
+# in turn, so a test asking about ".smp_locks" has to cut its block out of the
+# listing first.
+section_relocs()
+{
+ local sec="${1//./\\.}"
+
+ out_relocs | awk "/rela$sec'/,/^\$/"
+}
+
+assert_reloc_sym()
+{
+ section_relocs "$1" | awk -v n="$2" '$5 == n' | grep -q . ||
+ fail "expected a relocation to '$2' in '$1'"
+}
+
+assert_no_reloc_sym()
+{
+ section_relocs "$1" | awk -v n="$2" '$5 == n' | grep -q . &&
+ fail "unexpected relocation to '$2' in '$1'"
+ return 0
+}
+
+# assert_reloc_count <section> <count>
+#
+# Counts relocation entries, not header or blank lines: whether a special
+# section entry was extracted once, twice or not at all is usually the whole
+# question.
+#
+# A count of zero is ambiguous on its own -- a section with no relocations and
+# no section at all both read as zero -- so require the section to exist. A
+# test expecting nothing there wants assert_no_section.
+assert_reloc_count()
+{
+ local n
+
+ assert_section "$1"
+
+ n="$(section_relocs "$1" | grep -cE '^[0-9a-f]{8,}')"
+ [ "$n" = "$2" ] ||
+ fail "expected $2 relocations in '$1', found $n"
+}
+
+# assert_klp_sym <symbol> [object]
+#
+# A klp symbol is named .klp.sym.<object>.<symbol>,<sympos>. The object
+# defaults to any, since most tests care that the reference was converted at
+# all rather than which object it resolved against.
+assert_klp_sym()
+{
+ out_symbols | grep -q "\.klp\.sym\.${2:-[^.]*}\.$(re_quote "$1")," ||
+ fail "expected klp symbol for '$1'"
+}
+
+# assert_klp_sympos <symbol> <sympos>
+#
+# The number after the comma in .klp.sym.<object>.<symbol>,<sympos> says which
+# of several same-named symbols livepatch should resolve to, counting from 1;
+# 0 means the name is unique and no disambiguation is needed. Resolving to the
+# wrong one is not a load failure, it is a patch quietly wired to the wrong
+# object.
+assert_klp_sympos()
+{
+ out_symbols | grep -qE "\.klp\.sym\.[^.]+\.$(re_quote "$1"),$2([[:space:]]|\$)" ||
+ fail "expected klp symbol for '$1' with sympos $2, found:$(
+ out_symbols | grep -o "\.klp\.sym\.[^.]*\.$(re_quote "$1"),[0-9]*" |
+ sort -u | tr '\n' ' ')"
+}
+
+assert_no_klp_sym()
+{
+ out_symbols | grep -q "\.klp\.sym\.${2:-[^.]*}\.$(re_quote "$1")," &&
+ fail "unexpected klp symbol for '$1'"
+ return 0
+}
+
+assert_tombstone()
+{
+ out_symbols | grep -qE "\.klp\.tombstone\.$(re_quote "$1")([[:space:]]|\$)" ||
+ fail "expected a tombstone for '$1'"
+}
+
+assert_symbol()
+{
+ out_symbols | awk -v n="$1" '$NF == n' | grep -q . ||
+ fail "expected symbol '$1' in output"
+}
+
+assert_no_symbol()
+{
+ out_symbols | awk -v n="$1" '$NF == n' | grep -q . &&
+ fail "unexpected symbol '$1' in output"
+ return 0
+}
+
+# assert_diff_log <regex>
+#
+# klp diff's combined output, for tests asserting on a diagnostic. Error
+# messages are part of the interface when the whole point is that a construct
+# gets rejected, and a rejection for the wrong reason is not a pass.
+assert_diff_log()
+{
+ diff_log | grep -qE -- "$1" ||
+ fail "expected '$1' in klp diff output: $(tail -2 "$workdir/diff.log")"
+}
+
+# checksum_of <object> <symbol>
+#
+# The checksum "klp checksum" recorded for one symbol, as a hex string.
+#
+# .discard.sym_checksum is an array of { u64 addr; u64 checksum; }, where addr
+# is the target of a relocation naming the symbol. Nothing in the section
+# itself says which symbol an entry belongs to, so the relocation is what
+# locates the entry; the checksum is the eight bytes after it.
+# Callers use this in a command substitution, where fail() would only exit the
+# subshell and the test would carry on with an empty checksum. So this returns
+# non-zero and prints nothing, and the assertions below check for that. For
+# the same reason it does not run run_checksum() itself: that one does call
+# fail(), and from in here the message would be captured as the checksum
+# rather than ending the test. The caller runs it first.
+checksum_of()
+{
+ local obj="$workdir/$1" sym="$2" off
+
+ off="$($READELF -rW "$obj" 2>/dev/null |
+ awk -v s="$sym" '/rela\.discard\.sym_checksum/,/^$/ {
+ if ($5 == s) { print $1; exit }
+ }')"
+
+ [ -n "$off" ] || return 1
+
+ $OBJCOPY -O binary --only-section=.discard.sym_checksum \
+ "$obj" "$workdir/checksums.bin" 2>/dev/null || return 1
+
+ dd if="$workdir/checksums.bin" bs=1 skip=$((16#$off + 8)) count=8 \
+ status=none | od -An -tx1 | tr -d ' \n'
+}
+
+# assert_checksum_differs <symbol> / assert_checksum_matches <symbol>
+#
+# Compare what klp checksum recorded for a symbol in the original against the
+# patched object. This is what decides whether klp diff treats a function as
+# changed, so a test asserting only that the right functions were cloned cannot
+# tell a correct checksum from one which happens to differ.
+checksum_pair()
+{
+ run_checksum
+
+ orig_checksum="$(checksum_of "$orig_obj" "$1")"
+ patched_checksum="$(checksum_of "$patched_obj" "$1")"
+
+ [ -n "$orig_checksum" ] ||
+ fail "no checksum recorded for '$1' in $orig_obj"
+ [ -n "$patched_checksum" ] ||
+ fail "no checksum recorded for '$1' in $patched_obj"
+}
+
+assert_checksum_differs()
+{
+ checksum_pair "$1"
+
+ [ "$orig_checksum" != "$patched_checksum" ] ||
+ fail "checksum for '$1' unchanged at $orig_checksum, expected it to differ"
+}
+
+assert_checksum_matches()
+{
+ checksum_pair "$1"
+
+ [ "$orig_checksum" = "$patched_checksum" ] ||
+ fail "checksum for '$1' changed from $orig_checksum to" \
+ "$patched_checksum, expected no change"
+}
+
+# run_post_link [expected exit status]
+#
+# klp post-link runs last in a livepatch build, converting the intermediate
+# __klp_relocs.* sections into the .klp.rela.* form the kernel consumes. It
+# needs nothing but an object containing those sections, which is what klp diff
+# produces, so it runs on out.o here rather than on a built module. Rewrites
+# out.o in place, so the out_* helpers show the result afterwards.
+run_post_link()
+{
+ local expect="${1:-0}" rc=0
+
+ "$OBJTOOL" klp post-link "$workdir/out.o" \
+ > "$workdir/post-link.log" 2>&1 || rc=$?
+
+ [ "$rc" = "$expect" ] ||
+ fail "klp post-link exited $rc, expected $expect:" \
+ "$(tail -2 "$workdir/post-link.log")"
+}
+
+# The flags readelf prints for a section, or nothing when it has none. The
+# leading "[nn]" index is stripped first so the columns can be counted.
+section_flags()
+{
+ out_sections | sed 's/^ *\[[ 0-9]*\] *//' |
+ awk -v s="$1" '$1 == s && $7 ~ /^[A-Za-z]+$/ { print $7 }'
+}
+
+# assert_section_flag <section> <letter>
+#
+# SHF_RELA_LIVEPATCH is OS-specific, so readelf renders it as "o". A klp rela
+# section which lost it is an ordinary rela section, which the linker may apply
+# and the livepatch code will not.
+assert_section_flag()
+{
+ local flags; flags="$(section_flags "$1")"
+
+ [ -n "$flags" ] ||
+ fail "section '$1' has no flags, expected '$2'"
+ case "$flags" in
+ *"$2"*) ;;
+ *) fail "section '$1' has flags '$flags', expected '$2'" ;;
+ esac
+}
+
+# assert_klp_rela <object> <section>
+#
+# post-link names the converted sections .klp.rela.<object>.<section>, one per
+# base section. Also checks SHF_RELA_LIVEPATCH, since the name alone is not
+# what makes the kernel process it.
+assert_klp_rela()
+{
+ local name=".klp.rela.$1.$2"
+
+ out_sections | grep -q "[[:space:]]$(re_quote "$name")[[:space:]]" ||
+ fail "expected section '$name' in output"
+
+ assert_section_flag "$name" o
+}
+
+# assert_livepatch_sym <symbol>
+#
+# Symbols a klp relocation resolves against live in SHN_LIVEPATCH, which
+# readelf prints as "OS [0xff20]" -- llvm-readelf without the space, so match
+# either. The kernel resolves these itself at patch load; anything else is a
+# symbol the module loader will try, and fail, to resolve normally.
+assert_livepatch_sym()
+{
+ out_symbols | grep -E 'OS ?\[0xff20\]' |
+ grep -qE "\.klp\.sym\.[^.]+\.$(re_quote "$1")," ||
+ fail "expected a klp symbol for '$1' in SHN_LIVEPATCH"
+}
diff --git a/tools/objtool/tests/run-tests.sh b/tools/objtool/tests/run-tests.sh
new file mode 100755
index 000000000000..c0762f532d2a
--- /dev/null
+++ b/tools/objtool/tests/run-tests.sh
@@ -0,0 +1,239 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Run the objtool klp tests. Each test-*.sh prints one TAP result line.
+#
+# Tests live in generic/ and in a directory per architecture. A run executes
+# generic/ plus the one matching this architecture, so a test which cannot
+# apply here is not run rather than reporting a skip; what was left out is
+# reported once, as a comment, so differing coverage is still visible.
+#
+# A run covers one compiler and one architecture; CI runs the combinations.
+# The harness checks the environment once up front and fails the run if the
+# suite cannot execute, rather than letting every test skip and exit 0.
+
+set -u
+
+# Determinism: the order test-*.sh expands in, and how grep's character ranges
+# and sort's collation behave inside the tests, are all locale-dependent. A
+# suite whose results depend on the invoking shell's locale is a suite whose
+# failures cannot be reproduced.
+export LC_ALL=C
+
+usage()
+{
+ cat <<EOF
+usage: $(basename "$0") [-k|--keep] [test...]
+
+Run the objtool klp tests for this architecture: everything in generic/, plus
+everything in the directory named for it. With no arguments, runs all of them.
+A test may be named with or without its "test-" prefix and ".sh" suffix, and is
+looked for in both directories.
+
+Options:
+ -k, --keep same as KEEP=all (see below)
+
+Environment:
+ OBJTOOL objtool binary to test (default ../objtool)
+ CC compiler used to build fixtures (default gcc)
+ ARCH architecture the tests are for (default: uname -m)
+ KEEP failed keep only failing tests (default)
+ all keep every test's working directory
+ none remove all working directories
+
+A test which needs something of its own says so in its skip message.
+EOF
+ exit "${1:-0}"
+}
+
+cd "$(dirname "$0")" || exit 1
+
+keep_from_args=
+while [ $# -gt 0 ]; do
+ case "$1" in
+ -h|--help) usage ;;
+ -k|--keep) keep_from_args=all; shift ;;
+ --) shift; break ;;
+ -*) echo "unknown option: $1" >&2; usage 1 ;;
+ *) break ;;
+ esac
+done
+
+KLP_TEST_KEEP="${KEEP:-failed}"
+[ -n "$keep_from_args" ] && KLP_TEST_KEEP="$keep_from_args"
+case "$KLP_TEST_KEEP" in
+all|none|failed) ;;
+*)
+ echo "invalid KEEP=$KLP_TEST_KEEP (want failed, all, or none)" >&2
+ exit 1
+ ;;
+esac
+export KLP_TEST_KEEP
+
+echo "TAP version 13"
+
+# Sourcing the harness runs its preflight, which decides which architecture
+# this run is for -- so the test list cannot be built before it has, and the
+# tests inherit the answers rather than working them out again.
+. ./lib.sh
+
+dirs=( generic )
+[ -d "$KLP_TEST_ARCH" ] && dirs+=( "$KLP_TEST_ARCH" )
+
+if [ $# -gt 0 ]; then
+ tests=()
+ for arg in "$@"; do
+ name="test-${arg#test-}"; name="${name%.sh}.sh"
+ found=
+ for d in "${dirs[@]}"; do
+ [ -f "$d/$name" ] || continue
+ [ -x "$d/$name" ] ||
+ { echo "not executable: $d/$name" >&2; exit 1; }
+ tests+=( "$d/$name" ); found=y
+ done
+ [ -n "$found" ] ||
+ { echo "no such test for $KLP_TEST_ARCH: $arg" >&2; exit 1; }
+ done
+else
+ tests=()
+ for d in "${dirs[@]}"; do
+ for t in "$d"/test-*.sh; do
+ [ -f "$t" ] && tests+=( "$t" )
+ done
+ done
+ [ "${#tests[@]}" -gt 0 ] ||
+ { echo "1..0 # SKIP no tests found"; exit 0; }
+
+ # Tests for another architecture are absent from this run entirely. Say
+ # how many, so a run which covers less than the tree holds does not look
+ # like one that covers all of it.
+ for d in */; do
+ d="${d%/}"
+ case "$d" in generic|"$KLP_TEST_ARCH") continue ;; esac
+ n=$(ls "$d"/test-*.sh 2>/dev/null | wc -l)
+ [ "$n" -gt 0 ] || continue
+ echo "# not run: $n test$( [ "$n" = 1 ] || echo s ) in $d/" \
+ "(this run is $KLP_TEST_ARCH)"
+ done
+fi
+
+# One directory for the whole run, one per test inside it, mirroring the
+# source layout. A run then leaves a single thing behind instead of 39
+# scattered among everything else using mktemp.
+rundir="$(mktemp -d "${TMPDIR:-/tmp}/klp-tests.XXXXXXXX")" ||
+ { echo "Bail out! cannot create a working directory" >&2; exit 1; }
+
+# The run's own two levels: each test's directory, and the one per source
+# directory holding them. Take them away if the tests left them empty, and
+# say so if they did not. Either rmdir may fail -- the glob stays unexpanded
+# when nothing was created -- so ask the directory itself rather than trusting
+# the status. Never rm -rf: what to keep is the tests' decision, made in
+# cleanup() as each one exits, and this must not overrule it.
+reap_rundir()
+{
+ rmdir "$rundir"/*/ 2>/dev/null
+ rmdir "$rundir" 2>/dev/null
+ [ -d "$rundir" ]
+}
+
+# An interrupted run has the same directory to answer for, and the tests it
+# never reached will not clean up on their way out. The one it was running
+# has, and under the default it kept what it had built, so say where.
+interrupted()
+{
+ reap_rundir && echo "# interrupted; what was built is in $rundir"
+ exit 130
+}
+trap interrupted INT TERM HUP
+
+echo "1..${#tests[@]}"
+
+pass=0 fail=0 static_skip=0 probe_skip=0 xfail=0 xpass=0
+failed_dirs=()
+
+for t in "${tests[@]}"; do
+ out="$(KLP_TEST_WORKDIR="$rundir/${t%.sh}" ./"$t" 2>&1)"
+ rc=$?
+
+ # A test prints one result line, but it is not necessarily the only
+ # thing it prints: objtool warns on stderr, and the runner captures
+ # that. Classify the result line itself rather than the whole of the
+ # output, or a stray line ahead of it makes every pattern below miss and
+ # the exit status decide -- which would count an expected failure, which
+ # exits 0, as a pass.
+ result="$(printf '%s\n' "$out" | grep -E '^(ok|not ok)' | tail -1)"
+ rest="$(printf '%s\n' "$out" | grep -Ev '^(ok|not ok)')"
+
+ # Classify from the result line, not the exit status: a skip and a pass
+ # both exit 0, and telling them apart is the point of counting.
+ #
+ # The two skip kinds differ in what they promise. A static skip was
+ # declared before the test ran ("clang does not do this"), so it is
+ # expected indefinitely. A probe skip means the construct did not turn
+ # up this time, which is weaker and worth watching: one that becomes
+ # permanent is a fixture that quietly stopped testing anything.
+ case "$result" in
+ *"# SKIP (declared)"*) static_skip=$((static_skip + 1)) ;;
+ *"# SKIP (probe)"*) probe_skip=$((probe_skip + 1)) ;;
+ *"# SKIP"*)
+ # An undeclared skip: the test gave up for a reason it never
+ # said it might. That is a hole, not an expected outcome.
+ #
+ # Replace the line rather than adding one. Every test owes the
+ # plan exactly one result, and a consumer counting them is
+ # entitled to say so when the totals disagree.
+ rest="$rest${rest:+$'\n'}was: $result"
+ result="not ok - $(basename "$t" .sh): undeclared skip"
+ result="$result (use gcc_only/clang_only or require_input_*)"
+ fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;;
+ "not ok"*"# TODO"*) xfail=$((xfail + 1)) ;;
+ "ok"*"# TODO"*) xpass=$((xpass + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;;
+ "not ok"*) fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;;
+ "ok"*) pass=$((pass + 1)) ;;
+ *)
+ # No result line at all: the test died before reporting.
+ rest="$rest${rest:+$'\n'}exited $rc without a result line"
+ result="not ok - $(basename "$t" .sh): no TAP result"
+ fail=$((fail + 1)); failed_dirs+=( "$rundir/${t%.sh}" ) ;;
+ esac
+
+ echo "$result"
+ [ -n "$rest" ] && printf '%s\n' "$rest" | sed 's/^[^#]/# &/'
+
+done
+
+echo "# pass:$pass fail:$fail static-skip:$static_skip" \
+ "probe-skip:$probe_skip xfail:$xfail xpass:$xpass"
+
+case "$KLP_TEST_KEEP" in
+all)
+ echo "# keep=all: workdirs kept in $rundir"
+ echo "# inspect: diff.log, readelf -S out.o under each test-* subdirectory"
+ echo "# cleanup: rm -rf $rundir"
+ ;;
+failed)
+ if [ "${#failed_dirs[@]}" -gt 0 ]; then
+ echo "# keep=failed: ${#failed_dirs[@]} failing test(s) kept under $rundir:"
+ for d in "${failed_dirs[@]}"; do
+ echo "# ${d#"$rundir"/}/"
+ done
+ echo "# inspect: diff.log readelf -S out.o"
+ echo "# one test: $PWD/run-tests.sh <name>"
+ echo "# cleanup: rm -rf $rundir"
+ elif reap_rundir; then
+ echo "# $rundir was not empty;" \
+ "a test did not clean up after itself"
+ fi
+ ;;
+none)
+ if reap_rundir; then
+ echo "# $rundir was not empty;" \
+ "a test did not clean up after itself"
+ elif [ "$fail" != 0 ] || [ "$xpass" != 0 ]; then
+ echo "# keep=none: artifacts were removed" \
+ "(re-run with KEEP=failed or KEEP=all)"
+ fi
+ ;;
+esac
+
+[ "$fail" = 0 ] && [ "$xpass" = 0 ]
diff --git a/tools/objtool/tests/x86/fixtures/alt_annotate.c b/tools/objtool/tests/x86/fixtures/alt_annotate.c
new file mode 100644
index 000000000000..af44d320dcb0
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/alt_annotate.c
@@ -0,0 +1,57 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An x86 alternative whose replacement instruction carries a text annotation.
+ *
+ * The kernel does this wherever an ALTERNATIVE contains something objtool has
+ * to be told about -- a retpoline-safe indirect branch, an intentionally
+ * missing ENDBR -- so the .discard.annotate_insn entry references an address
+ * inside .altinstr_replacement rather than inside a function.
+ *
+ * Two things make that awkward for klp diff, and both are why this fixture
+ * exists. Replacement code has no real symbol: objtool invents a NOTYPE fake
+ * symbol for it, so an annotation pointing there does not reference a FUNC.
+ * And .discard.annotate_insn has to be cloned after .altinstructions, or the
+ * replacement it names has no clone to point at yet.
+ *
+ * struct alt_instr is written out by hand as in empty_alternative.c: s32
+ * instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen, u8 replacementlen,
+ * with an entsize so klp diff can find the entry boundaries.
+ * .discard.annotate_insn entries are s32 offset, s32 type; type 2 is
+ * ANNOTYPE_RETPOLINE_SAFE.
+ *
+ * The replacement label is global so the relocations name it rather than
+ * .altinstr_replacement plus an addend, which klp diff cannot convert. It is
+ * still NOTYPE, which is the shape that matters here.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+int target(int x)
+{
+ asm volatile(
+ "661: nop\n\t"
+ ".pushsection .altinstr_replacement, \"ax\"\n\t"
+ ".globl target_repl\n\t"
+ "target_repl:\n\t"
+ " nop\n\t"
+ /* The annotation lands inside the replacement. */
+ ".pushsection .discard.annotate_insn, \"M\", @progbits, 8\n\t"
+ ".long target_repl - .\n\t"
+ ".long 2\n\t"
+ ".popsection\n\t"
+ "target_repl_end:\n\t"
+ ".popsection\n\t"
+ ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t"
+ ".long 661b - .\n\t"
+ ".long target_repl - .\n\t"
+ ".long 0\n\t"
+ ".byte 1\n\t"
+ ".byte target_repl_end - target_repl\n\t"
+ ".popsection\n\t");
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/x86/fixtures/checksum_alt.c b/tools/objtool/tests/x86/fixtures/checksum_alt.c
new file mode 100644
index 000000000000..ed342d94eb9e
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/checksum_alt.c
@@ -0,0 +1,66 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An x86 alternative whose replacement code is part of the patched function's
+ * checksum.
+ *
+ * checksum_update_insn() walks insn->alts after hashing the instruction
+ * itself, hashing the alternative's type and, when the replacement forms a
+ * group, its feature number and every instruction in it. So editing only the
+ * replacement -- code the CPU may or may not ever run -- has to move the
+ * function's checksum.
+ *
+ * It is reached through objtool's own alternative handling, so the object has
+ * to go through the check pass first: insn->alts is built there, not by the
+ * compiler.
+ *
+ * struct alt_instr is written out by hand as in empty_alternative.c: s32
+ * instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen, u8 replacementlen.
+ *
+ * Variants, applied to the patched build only:
+ *
+ * ALT_REPL the replacement instruction changes; the original does not
+ * ALT_FEATURE the feature number changes; no code changes at all
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/*
+ * Both spellings are two bytes, because a replacement may not be longer than
+ * the instruction it replaces: "xchg %ax, %ax" is 66 90 and two nops are
+ * 90 90. The original below is padded to match.
+ */
+#if defined(PATCHED) && defined(ALT_REPL)
+#define REPL_INSN " nop\n\t nop\n\t"
+#else
+#define REPL_INSN " xchg %ax, %ax\n\t"
+#endif
+
+#if defined(PATCHED) && defined(ALT_FEATURE)
+#define FEATURE "7"
+#else
+#define FEATURE "3"
+#endif
+
+int target(int x)
+{
+ asm volatile(
+ "661: nop\n\t"
+ " nop\n\t"
+ "662:\n\t"
+ ".pushsection .altinstr_replacement, \"ax\"\n\t"
+ ".globl target_repl\n\t"
+ "target_repl:\n\t"
+ REPL_INSN
+ "target_repl_end:\n\t"
+ ".popsection\n\t"
+ ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t"
+ ".long 661b - .\n\t"
+ ".long target_repl - .\n\t"
+ ".long " FEATURE "\n\t"
+ ".byte 662b - 661b\n\t"
+ ".byte target_repl_end - target_repl\n\t"
+ ".popsection\n\t");
+
+ return x + 1;
+}
diff --git a/tools/objtool/tests/x86/fixtures/empty_alternative.c b/tools/objtool/tests/x86/fixtures/empty_alternative.c
new file mode 100644
index 000000000000..9336d74bfa92
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/empty_alternative.c
@@ -0,0 +1,77 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An x86 alternative with an empty replacement, as the second entry of
+ * ALTERNATIVE_2("orig", "repl", ft1, "", ft2) produces. Its replacement
+ * offset still gets a relocation, but the label it points at is the end of the
+ * previous replacement, which is also where the *next* one begins -- here,
+ * neighbor()'s. The value is meaningless; it is only ever used with a length
+ * of zero.
+ *
+ * struct alt_instr is written out by hand so the fixture builds without kernel
+ * headers: s32 instr_offset, s32 repl_offset, u32 ft_flags, u8 instrlen,
+ * u8 replacementlen. The section carries an entsize because klp diff needs
+ * either that or an ANNOTATE_DATA_SPECIAL annotation to find entry boundaries.
+ *
+ * The replacement labels are global so the relocations name them rather than
+ * .altinstr_replacement plus an addend.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+extern int neighbor_only(int x);
+
+int target(int x)
+{
+ asm volatile(
+ "661: nop\n\t"
+ ".pushsection .altinstr_replacement, \"ax\"\n\t"
+ ".globl target_repl\n\t"
+ "target_repl:\n\t"
+ " nop\n\t"
+ "target_repl_end:\n\t"
+ ".popsection\n\t"
+ ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t"
+ /* a real replacement */
+ ".long 661b - .\n\t"
+ ".long target_repl - .\n\t"
+ ".long 0\n\t"
+ ".byte 1\n\t"
+ ".byte target_repl_end - target_repl\n\t"
+ /* an empty one, pointing at neighbor()'s replacement */
+ ".long 661b - .\n\t"
+ ".long neighbor_repl - .\n\t"
+ ".long 0\n\t"
+ ".byte 1\n\t"
+ ".byte 0\n\t"
+ ".popsection\n\t");
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
+
+/*
+ * Unrelated, unpatched, and referencing a symbol nothing else does, so that
+ * dragging its replacement in is visible.
+ */
+int neighbor(int x)
+{
+ asm volatile(
+ "771: nop\n\t"
+ ".pushsection .altinstr_replacement, \"ax\"\n\t"
+ ".globl neighbor_repl\n\t"
+ "neighbor_repl:\n\t"
+ " call neighbor_only\n\t"
+ "neighbor_repl_end:\n\t"
+ ".popsection\n\t"
+ ".pushsection .altinstructions, \"aM\", @progbits, 14\n\t"
+ ".long 771b - .\n\t"
+ ".long neighbor_repl - .\n\t"
+ ".long 0\n\t"
+ ".byte 1\n\t"
+ ".byte neighbor_repl_end - neighbor_repl\n\t"
+ ".popsection\n\t");
+ return x;
+}
diff --git a/tools/objtool/tests/x86/fixtures/kcfi.c b/tools/objtool/tests/x86/fixtures/kcfi.c
new file mode 100644
index 000000000000..b62fc60634cf
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/kcfi.c
@@ -0,0 +1,39 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An indirect call, which under kCFI is preceded by a type check and a trap.
+ *
+ * Clang emits a __cfi_<func> prefix symbol carrying the type hash ahead of
+ * every address-taken function, and records the trap site in .kcfi_traps.
+ * Both belong to the function and both have to come with it into a patch.
+ *
+ * Needs -fsanitize=kcfi, which only Clang has.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+static int impl_a(int x)
+{
+ return x + 1;
+}
+
+static int impl_b(int x)
+{
+ return x * 2;
+}
+
+__attribute__((noinline)) int (*pick(int x))(int)
+{
+ return (x & 1) ? impl_a : impl_b;
+}
+
+int target(int x)
+{
+ int (*fn)(int arg) = pick(x);
+
+#ifdef PATCHED
+ return fn(x) + 2;
+#else
+ return fn(x) + 1;
+#endif
+}
diff --git a/tools/objtool/tests/x86/fixtures/special_sections.c b/tools/objtool/tests/x86/fixtures/special_sections.c
new file mode 100644
index 000000000000..d42798848fda
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/special_sections.c
@@ -0,0 +1,77 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A special section entry belonging to a patched function. SPECIAL_SEC picks
+ * which section, since klp diff treats eight of them alike and each needs
+ * extracting for the patched function and no other.
+ *
+ * The entry is written out by hand so the fixture builds without kernel
+ * headers. Only the leading relocation matters to klp diff; the rest is
+ * padded to the section's real entry size, because the entries have to be the
+ * right length for the boundaries between them to fall in the right places.
+ *
+ * SPECIAL_RELOCS covers __ex_table, whose entries relocate both the faulting
+ * instruction and its fixup; objtool rejects one with only the first.
+ */
+
+#ifndef SPECIAL_SEC
+#define SPECIAL_SEC "__bug_table"
+#endif
+#ifndef SPECIAL_ENTSIZE
+#define SPECIAL_ENTSIZE 12
+#endif
+#ifndef SPECIAL_RELOCS
+#define SPECIAL_RELOCS 1
+#endif
+
+#define STR_(x) #x
+#define STR(x) STR_(x)
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) = "\0name=vmlinux";
+
+/*
+ * other() gets an entry of its own, so that "the untouched function's entry
+ * was not dragged in" is a question the section can actually answer. With
+ * only target() contributing, there is nothing for klp diff to leave behind
+ * and the negative assertion holds however the extraction behaves.
+ */
+int other(int x)
+{
+ asm volatile(
+ "3: nop\n\t"
+ "4:\n\t"
+ ".pushsection " SPECIAL_SEC ", \"aM\", @progbits, "
+ STR(SPECIAL_ENTSIZE) "\n\t"
+ ".long 3b - .\n\t"
+#if SPECIAL_RELOCS > 1
+ ".long 4b - .\n\t"
+ ".fill " STR(SPECIAL_ENTSIZE) " - 8, 1, 0\n\t"
+#else
+ ".fill " STR(SPECIAL_ENTSIZE) " - 4, 1, 0\n\t"
+#endif
+ ".popsection\n\t");
+
+ return x + 9;
+}
+
+int target(int x)
+{
+ asm volatile(
+ "1: nop\n\t"
+ "2:\n\t"
+ ".pushsection " SPECIAL_SEC ", \"aM\", @progbits, "
+ STR(SPECIAL_ENTSIZE) "\n\t"
+ ".long 1b - .\n\t"
+#if SPECIAL_RELOCS > 1
+ ".long 2b - .\n\t"
+ ".fill " STR(SPECIAL_ENTSIZE) " - 8, 1, 0\n\t"
+#else
+ ".fill " STR(SPECIAL_ENTSIZE) " - 4, 1, 0\n\t"
+#endif
+ ".popsection\n\t");
+#ifdef PATCHED
+ return x + 2;
+#else
+ return x + 1;
+#endif
+}
diff --git a/tools/objtool/tests/x86/fixtures/static_call_no_key.c b/tools/objtool/tests/x86/fixtures/static_call_no_key.c
new file mode 100644
index 000000000000..748ba7c0f86d
--- /dev/null
+++ b/tools/objtool/tests/x86/fixtures/static_call_no_key.c
@@ -0,0 +1,32 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A static call to a trampoline whose key symbol this object cannot see.
+ *
+ * That is the normal situation for a module: __SCK__* keys are not exported,
+ * and read-only access is granted at load time instead. objtool's static call
+ * handling has to accept it for any module, including a livepatch module built
+ * by hand rather than by klp-build.
+ *
+ * LIVEPATCH adds the .modinfo tag which makes objtool treat this as a
+ * livepatch module.
+ */
+
+static const char __modinfo[]
+ __attribute__((section(".modinfo"), used, aligned(1))) =
+#ifdef LIVEPATCH
+ "\0livepatch=Y"
+#endif
+ "\0name=klp_testmod";
+
+/*
+ * The trampoline is undefined here, exactly as it is for a module calling a
+ * static call defined in vmlinux. No __SCK__klp_test_call accompanies it.
+ */
+extern void __SCT__klp_test_call(void);
+
+int target(int x)
+{
+ __asm__ volatile("call __SCT__klp_test_call\n\t" ::: "memory");
+
+ return x + 1;
+}
diff --git a/tools/objtool/tests/x86/test-alt-annotation.sh b/tools/objtool/tests/x86/test-alt-annotation.sh
new file mode 100755
index 000000000000..96760e6df9e5
--- /dev/null
+++ b/tools/objtool/tests/x86/test-alt-annotation.sh
@@ -0,0 +1,38 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# A text annotation on an instruction inside an alternative's replacement must
+# be carried into the patch.
+#
+# The kernel annotates replacement code wherever objtool has to be told
+# something about it -- a retpoline-safe indirect branch, a deliberately absent
+# ENDBR. Two things made klp diff drop those annotations:
+#
+# - replacement code has no real symbol, so objtool invents a NOTYPE fake
+# one, and the extraction only kept references to FUNC symbols;
+# - .discard.annotate_insn was processed before .altinstructions, so the
+# replacement it referenced had no clone to point at yet.
+#
+# Nothing fails at build time when the annotation goes missing. It surfaces
+# later as objtool warning about, or rejecting, the patched code it was there
+# to explain.
+#
+# Fixed by 62a7a01fde87 ("objtool/klp: Fix extraction of text annotations for
+# alternatives").
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair alt_annotate.c
+
+assert_input_section .altinstructions
+assert_input_section .discard.annotate_insn
+
+run_diff
+
+# The annotation has to survive, and to still name the replacement. Checking
+# only the section would pass on an entry whose relocation was dropped.
+assert_section .discard.annotate_insn
+assert_reloc_sym .discard.annotate_insn target_repl
+
+pass "text annotation on an alternative replacement carried into the patch"
diff --git a/tools/objtool/tests/x86/test-checksum-alt.sh b/tools/objtool/tests/x86/test-checksum-alt.sh
new file mode 100755
index 000000000000..74ebfca2b3ff
--- /dev/null
+++ b/tools/objtool/tests/x86/test-checksum-alt.sh
@@ -0,0 +1,45 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# An alternative's replacement code counts towards the checksum of the function
+# it belongs to.
+#
+# checksum_update_insn() walks insn->alts after hashing the instruction itself:
+# the alternative's type, and where the replacement forms a group, its feature
+# number and every instruction in it. So a patch which edits only the
+# replacement -- code that runs on some CPUs and not others -- still has to
+# move the function's checksum.
+#
+# If it does not, klp diff decides the function is unchanged and leaves it out.
+# The patch then ships the old replacement, and the bug is fixed only on
+# machines whose CPU takes the other arm. Which machines those are depends on
+# the feature bit, so the failure looks like a machine-specific bug rather than
+# a missing patch.
+#
+# insn->alts exists only after objtool's check pass, so the pair goes through
+# that first -- the compiler emits none of this structure itself.
+#
+# Covers the same ground as corpus/x86_64/checksum-alt-group,
+# checksum-alt-no-group and checksum-alt-recursion-guard in Joe Lawrence's
+# klp-build unit test corpus.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+check()
+{
+ build_pair checksum_alt.c "-D$1"
+ assert_input_section .altinstructions
+ run_objtool_check --mcount
+ run_checksum
+
+ assert_checksum_differs target
+}
+
+# The replacement instruction itself.
+check ALT_REPL
+# The feature number, with no instruction anywhere changed.
+check ALT_FEATURE
+
+pass "alternative replacement code counts towards the checksum"
diff --git a/tools/objtool/tests/x86/test-empty-alternative.sh b/tools/objtool/tests/x86/test-empty-alternative.sh
new file mode 100755
index 000000000000..9d40c3a405af
--- /dev/null
+++ b/tools/objtool/tests/x86/test-empty-alternative.sh
@@ -0,0 +1,31 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# An x86 alternative with an empty replacement still gets a relocation for its
+# replacement offset, but the label it points at is the end of the previous
+# replacement -- which is also the start of the next one. The value is
+# meaningless, and get_alt_entry() already ignores it.
+#
+# Cloning it drags in an unrelated neighboring replacement and everything that
+# replacement references. In the reported case an empty alternative in
+# meminfo_proc_show() pulled in one from proc_kcore_init(), emitting a klp
+# relocation against init text which is long freed by the time the patch is
+# applied.
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+build_pair empty_alternative.c
+
+assert_input_section .altinstructions
+assert_input_section .altinstr_replacement
+
+run_diff
+
+# target's own replacement comes along ...
+assert_symbol target_repl
+# ... neighbor's does not, nor what it references.
+assert_no_symbol neighbor_repl
+assert_no_symbol neighbor_only
+
+pass "empty alternative's replacement offset ignored when cloning"
diff --git a/tools/objtool/tests/x86/test-kcfi.sh b/tools/objtool/tests/x86/test-kcfi.sh
new file mode 100755
index 000000000000..b583ef608747
--- /dev/null
+++ b/tools/objtool/tests/x86/test-kcfi.sh
@@ -0,0 +1,39 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# Under kCFI an indirect call checks a type hash before jumping, and traps on a
+# mismatch. Two things belong to the calling function and must come with it
+# into a patch:
+#
+# - the __cfi_<func> prefix symbol holding the hash. Lose it and the patched
+# function has no type identity, so indirect calls to it trap.
+# - its .kcfi_traps entry. Lose that and the trap is not recognised as a
+# CFI failure, so what should be a clean report becomes an oops.
+#
+# Neither shows up at build time.
+
+. "$(dirname "$0")/../lib.sh"
+
+clang_only "kCFI is a Clang feature"
+
+setup
+
+# Declared above that this is Clang's; a given Clang may still be too old.
+cc_supports -fsanitize=kcfi ||
+ probe_skip "this clang does not support -fsanitize=kcfi"
+
+build_pair kcfi.c -fsanitize=kcfi
+
+assert_input_section .kcfi_traps
+assert_input_symbol __cfi_target
+
+run_diff
+
+assert_patched target
+
+# The prefix symbol comes with its function ...
+assert_symbol __cfi_target
+# ... and so does the trap entry.
+assert_section .kcfi_traps
+
+pass "kCFI prefix symbol and trap entry carried with the patched function"
diff --git a/tools/objtool/tests/x86/test-manual-klp-static-call.sh b/tools/objtool/tests/x86/test-manual-klp-static-call.sh
new file mode 100755
index 000000000000..6c4d275e3549
--- /dev/null
+++ b/tools/objtool/tests/x86/test-manual-klp-static-call.sh
@@ -0,0 +1,40 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# objtool's static call handling must accept a livepatch module which cannot
+# see a static call's key symbol.
+#
+# __SCK__* keys are not exported; modules get read-only access at load time
+# instead. Livepatch modules built by klp-build do have full access to their
+# keys, and a check was added on the strength of that -- but a livepatch module
+# can also be written by hand, and samples/livepatch is full of them. One of
+# those needs a key it cannot see as soon as it does anything that expands to a
+# static call, which with CONFIG_MEM_ALLOC_PROFILING_DEBUG includes allocating
+# memory:
+#
+# samples/livepatch/livepatch-shadow-fix1.o: error: objtool: static_call:
+# can't find static_call_key symbol: __SCK__WARN_trap
+#
+# The module built without the livepatch tag is the control: it takes the same
+# path and has always been accepted, so a test which only built the livepatch
+# one could not tell this fix from the check being removed altogether.
+#
+# Fixed by f495054bd12e ("objtool/klp: Fix unexported static call key access
+# for manually built livepatch modules").
+
+. "$(dirname "$0")/../lib.sh"
+
+setup
+
+# Not a klp subcommand: this is objtool's ordinary check pass, which is what
+# runs over a hand-built livepatch module during a normal kernel build.
+for tag in "" -DLIVEPATCH; do
+ build_one static_call_no_key.c mod.o $tag
+
+ "$OBJTOOL" --module --static-call "$workdir/mod.o" \
+ > "$workdir/objtool.log" 2>&1 ||
+ fail "objtool rejected a ${tag:+livepatch }module which cannot" \
+ "see its static call key: $(tail -1 "$workdir/objtool.log")"
+done
+
+pass "livepatch module accepted without access to its static call key"
diff --git a/tools/objtool/tests/x86/test-special-sections.sh b/tools/objtool/tests/x86/test-special-sections.sh
new file mode 100755
index 000000000000..8dfaa4fc9a36
--- /dev/null
+++ b/tools/objtool/tests/x86/test-special-sections.sh
@@ -0,0 +1,42 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+#
+# klp diff extracts entries from eight special sections. Between them the
+# existing tests reach .kcfi_traps, __jump_table, .static_call_sites and
+# .altinstructions; __bug_table, __ex_table and __mcount_loc are covered by
+# nothing, though the same extraction code serves all of them.
+#
+# Losing an entry is quiet in every case and wrong in a different way for each:
+# a WARN() in patched code that no longer reports where it came from, an
+# exception fixup that is simply not there when the faulting instruction traps,
+# a function ftrace can no longer see.
+
+. "$(dirname "$0")/../lib.sh"
+
+
+# section, entry size, relocations per entry
+for spec in "__bug_table 12 1" "__ex_table 12 2" "__mcount_loc 8 1"; do
+ set -- $spec
+ sec=$1
+
+ # A fresh workdir per section: run_diff caches its checksums.
+ setup
+ build_pair special_sections.c \
+ -DSPECIAL_SEC="\"$1\"" -DSPECIAL_ENTSIZE="$2" -DSPECIAL_RELOCS="$3"
+
+ assert_input_section "$sec"
+ run_diff
+
+ # Extracted, and pointing at the function that was patched.
+ assert_section "$sec"
+ assert_reloc_sym "$sec" target
+ assert_patched target
+
+ # Nothing belonging to the function that was not.
+ assert_not_patched other
+ assert_no_reloc_sym "$sec" other
+
+ cleanup
+done
+
+pass "__bug_table, __ex_table and __mcount_loc entries extracted"
diff --git a/tools/perf/trace/beauty/include/uapi/linux/sched.h b/tools/perf/trace/beauty/include/uapi/linux/sched.h
index 33a4624285cd..19ffeba89428 100644
--- a/tools/perf/trace/beauty/include/uapi/linux/sched.h
+++ b/tools/perf/trace/beauty/include/uapi/linux/sched.h
@@ -53,7 +53,7 @@
*/
#define UNSHARE_EMPTY_MNTNS 0x00100000 /* Unshare an empty mount namespace. */
-#ifndef __ASSEMBLY__
+#ifndef __ASSEMBLER__
/**
* struct clone_args - arguments for the clone3 syscall
* @flags: Flags for the new process as listed above.
diff --git a/tools/testing/selftests/timers/clocksource-switch.c b/tools/testing/selftests/timers/clocksource-switch.c
index db62a764c29e..2e86f56d953e 100644
--- a/tools/testing/selftests/timers/clocksource-switch.c
+++ b/tools/testing/selftests/timers/clocksource-switch.c
@@ -40,16 +40,23 @@
int get_clocksources(char list[][30])
{
int fd, i;
- size_t size;
+ ssize_t size;
char buf[512];
char *head, *tmp;
fd = open("/sys/devices/system/clocksource/clocksource0/available_clocksource", O_RDONLY);
+ if (fd < 0)
+ return 0;
- size = read(fd, buf, 512);
+ size = read(fd, buf, sizeof(buf) - 1);
close(fd);
+ if (size <= 0)
+ return 0;
+
+ buf[size] = '\0';
+
for (i = 0; i < 10; i++)
list[i][0] = '\0';
@@ -74,11 +81,21 @@ int get_clocksources(char list[][30])
int get_cur_clocksource(char *buf, size_t size)
{
+ ssize_t len;
int fd;
fd = open("/sys/devices/system/clocksource/clocksource0/current_clocksource", O_RDONLY);
+ if (fd < 0)
+ return -1;
+
+ len = read(fd, buf, size - 1);
+
+ close(fd);
+
+ if (len <= 0)
+ return -1;
- size = read(fd, buf, size);
+ buf[len] = '\0';
return 0;
}
diff --git a/tools/testing/selftests/timers/posix_timers.c b/tools/testing/selftests/timers/posix_timers.c
index a92d4b957747..52b7289e1450 100644
--- a/tools/testing/selftests/timers/posix_timers.c
+++ b/tools/testing/selftests/timers/posix_timers.c
@@ -113,9 +113,10 @@ static void check_itimer(int which, const char *name)
done = 0;
- if (which == ITIMER_VIRTUAL)
+ if (which == ITIMER_VIRTUAL) {
+ clock_id = CLOCK_THREAD_CPUTIME_ID;
signal(SIGVTALRM, sig_handler);
- else if (which == ITIMER_PROF) {
+ } else if (which == ITIMER_PROF) {
clock_id = CLOCK_THREAD_CPUTIME_ID;
signal(SIGPROF, sig_handler);
}
@@ -148,7 +149,7 @@ static void check_timer_create(int which)
struct itimerspec val = {
.it_value.tv_sec = DELAY,
};
- int clock_id = CLOCK_REALTIME;
+ int clock_id = which;
timer_t id;
done = 0;
diff --git a/tools/testing/selftests/timers/raw_skew.c b/tools/testing/selftests/timers/raw_skew.c
index 0c87a8fb0d7f..dfca3e2d50d2 100644
--- a/tools/testing/selftests/timers/raw_skew.c
+++ b/tools/testing/selftests/timers/raw_skew.c
@@ -91,6 +91,7 @@ int main(int argc, char **argv)
{
struct timespec mon, raw, start, end;
long long delta1, delta2, interval, eppm, ppm;
+ long tick_nom;
struct timex tx1, tx2;
setbuf(stdout, NULL);
@@ -129,6 +130,17 @@ int main(int argc, char **argv)
/* Avg the two actual freq samples adjtimex gave us */
ppm = (long long)(tx1.freq + tx2.freq) * 1000 / 2;
ppm = shift_right(ppm, 16);
+
+ /*
+ * The tick value holds the coarse part of the adjustment, a
+ * microsecond of the tick length per unit, and time sync daemons
+ * do put part of their correction there. What it is worth has to
+ * be counted in as well, or a clock disciplined through it looks
+ * off by a hundred ppm a unit against what the two clocks show.
+ */
+ tick_nom = USEC_PER_SEC / sysconf(_SC_CLK_TCK);
+ ppm += (long long)(tx1.tick + tx2.tick - 2 * tick_nom) *
+ (USEC_PER_SEC / tick_nom) * 1000 / 2;
printf(" %lld.%i(act)", ppm/1000, abs((int)(ppm%1000)));
if (llabs(eppm - ppm) > 1000) {
diff --git a/tools/testing/selftests/x86/Makefile b/tools/testing/selftests/x86/Makefile
index d478b13cc8d5..7565d2cf6a3c 100644
--- a/tools/testing/selftests/x86/Makefile
+++ b/tools/testing/selftests/x86/Makefile
@@ -19,7 +19,8 @@ TARGETS_C_32BIT_ONLY := entry_from_vm86 test_syscall_vdso unwind_vdso \
test_FCMOV test_FCOMI test_FISTTP \
vdso_restorer
TARGETS_C_64BIT_ONLY := fsgsbase sysret_rip syscall_numbering \
- corrupt_xstate_header amx lam test_shadow_stack avx apx
+ corrupt_xstate_header amx lam test_shadow_stack avx apx \
+ sigframe_fpu_portability
# Some selftests require 32bit support enabled also on 64bit systems
TARGETS_C_32BIT_NEEDED := ldt_gdt ptrace_syscall
@@ -138,3 +139,5 @@ $(OUTPUT)/avx_64: CFLAGS += -mno-avx -mno-avx512f
$(OUTPUT)/amx_64: EXTRA_FILES += xstate.c
$(OUTPUT)/avx_64: EXTRA_FILES += xstate.c
$(OUTPUT)/apx_64: EXTRA_FILES += xstate.c
+
+$(OUTPUT)/sigframe_fpu_portability_64: CFLAGS += -mno-avx -mno-avx512f
diff --git a/tools/testing/selftests/x86/sigframe_fpu_portability.c b/tools/testing/selftests/x86/sigframe_fpu_portability.c
new file mode 100644
index 000000000000..8377de052032
--- /dev/null
+++ b/tools/testing/selftests/x86/sigframe_fpu_portability.c
@@ -0,0 +1,235 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <signal.h>
+#include <string.h>
+#include <sys/ucontext.h>
+#include <stdlib.h>
+#include <stdint.h>
+#include <stdbool.h>
+#include <cpuid.h>
+#include <unistd.h>
+#include <sys/syscall.h>
+#include <stddef.h>
+#include <setjmp.h>
+
+#include "helpers.h"
+#include "xstate.h"
+
+#ifndef FP_XSTATE_MAGIC2_SIZE
+#define FP_XSTATE_MAGIC2_SIZE sizeof(FP_XSTATE_MAGIC2)
+#endif
+
+/*
+ * This test verifies the FPU portability and consistency of the signal frame.
+ *
+ * - test_valid_shrunk_xstate_size:
+ * Verifies that the kernel restores state from a frame with xstate_size
+ * shrunk to only include active features.
+ *
+ * - test_invalid_shrunk_xstate_size:
+ * Verifies that the kernel rejects a frame if xstate_size is too small for
+ * the features enabled in xfeatures.
+ */
+
+#define SIGFRAME_XSTATE_HDR_OFFSET 512
+#define XSTATE_SSE_ONLY_SIZE (SIGFRAME_XSTATE_HDR_OFFSET + XSAVE_HDR_SIZE)
+#define XFEATURE_MASK_FPSSE ((1 << XFEATURE_FP) | (1 << XFEATURE_SSE))
+
+static uint32_t ymm_offset;
+static uint32_t xstate_size_ymm;
+static pid_t self_pid;
+
+/*
+ * Load %ymm0 from @v, invoke SYS_kill to deliver @sig, and store the
+ * restored %ymm0 state back into @v within a single inline assembly
+ * block so the compiler cannot clobber %xmm0/%ymm0 between steps.
+ */
+__attribute__((target("avx")))
+static void raise_with_ymm0(int sig, uint64_t *v)
+{
+ register long rax asm("rax") = SYS_kill;
+ register long rdi asm("rdi") = self_pid;
+ register long rsi asm("rsi") = sig;
+
+ asm volatile ("vmovdqu %0, %%ymm0\n\t"
+ "syscall\n\t"
+ "vmovdqu %%ymm0, %0"
+ : "+m" (*(char (*)[32])v), "+r" (rax)
+ : "r" (rdi), "r" (rsi)
+ : "rcx", "r11", "ymm0", "memory");
+}
+
+/*
+ * Avoid using printf() in signal handlers as it is not
+ * async-signal-safe.
+ */
+#define SIGNAL_BUF_LEN 1024
+static char sig_err_buf[SIGNAL_BUF_LEN];
+
+static void sig_print(const char *msg)
+{
+ int left = SIGNAL_BUF_LEN - strlen(sig_err_buf) - 1;
+
+ strncat(sig_err_buf, msg, left);
+}
+
+static void check_avx_support(void)
+{
+ uint32_t eax, ebx, ecx, edx;
+ struct xstate_info xstate;
+
+ /* Check CPUID.01H:ECX.OSXSAVE[bit 27] before calling xgetbv to avoid #UD */
+ __cpuid(1, eax, ebx, ecx, edx);
+ if (!(ecx & (1 << 27)))
+ ksft_exit_skip("OSXSAVE not enabled by OS\n");
+
+ /* Check XCR0[2] (YMM) is enabled by OS */
+ if (!(xgetbv(0) & (1 << XFEATURE_YMM)))
+ ksft_exit_skip("AVX (YMM) not enabled in XCR0\n");
+
+ xstate = get_xstate_info(XFEATURE_YMM);
+ if (!xstate.size)
+ ksft_exit_skip("AVX not supported by hardware\n");
+
+ ymm_offset = xstate.xbuf_offset;
+ xstate_size_ymm = xstate.xbuf_offset + xstate.size;
+}
+
+#define TEST_YMMH_VAL (0x5656565656565656UL)
+
+static void __handle_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp, bool valid_size)
+{
+ uint64_t xfeatures, *ymmh_p;
+ struct xsave_buffer *xbuf;
+ struct _fpx_sw_bytes *sw;
+ ucontext_t *uc = ucp;
+ void *fp;
+
+ fp = uc->uc_mcontext.fpregs;
+ if (!fp) {
+ sig_print("fpregs is NULL\n");
+ return;
+ }
+
+ sw = get_fpx_sw_bytes(fp);
+ if (sw->magic1 != FP_XSTATE_MAGIC1) {
+ sig_print("magic1 is not valid\n");
+ return;
+ }
+
+ xbuf = (struct xsave_buffer *)fp;
+
+ /*
+ * Both test cases shrink the frame to contain only AVX (FP + SSE + YMM).
+ * If valid_size is true, set xstate_size to match the enabled features.
+ * If valid_size is false, set xstate_size too small (SSE only), which
+ * the kernel must reject.
+ */
+ if (valid_size)
+ sw->xstate_size = xstate_size_ymm;
+ else
+ sw->xstate_size = XSTATE_SSE_ONLY_SIZE;
+
+ xfeatures = get_xstatebv(xbuf);
+ xfeatures &= XFEATURE_MASK_FPSSE | (1 << XFEATURE_YMM);
+ set_xstatebv(xbuf, xfeatures);
+ set_fpx_sw_bytes_features(fp, xfeatures);
+
+ *(uint32_t *)(fp + sw->xstate_size) = FP_XSTATE_MAGIC2;
+
+ if (valid_size) {
+ ymmh_p = (uint64_t *)(fp + ymm_offset);
+ ymmh_p[0] = TEST_YMMH_VAL;
+ ymmh_p[1] = TEST_YMMH_VAL + 1;
+ }
+
+ /* clear everything after MAGIC2. */
+ if (sw->xstate_size + FP_XSTATE_MAGIC2_SIZE < sw->extended_size)
+ memset(fp + sw->xstate_size + FP_XSTATE_MAGIC2_SIZE, 0,
+ sw->extended_size - sw->xstate_size - FP_XSTATE_MAGIC2_SIZE);
+}
+
+static void handle_valid_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp)
+{
+ __handle_shrunk_xstate_size(sig, si, ucp, true);
+}
+
+static void handle_invalid_shrunk_xstate_size(int sig, siginfo_t *si, void *ucp)
+{
+ __handle_shrunk_xstate_size(sig, si, ucp, false);
+}
+
+static void test_valid_shrunk_xstate_size(void)
+{
+ uint64_t v[4];
+
+ sig_err_buf[0] = 0;
+ sethandler(SIGUSR1, handle_valid_shrunk_xstate_size, 0);
+
+ v[0] = 0x1111111111111111ULL;
+ v[1] = 0x2222222222222222ULL;
+ v[2] = 0x3333333333333333ULL;
+ v[3] = 0x4444444444444444ULL;
+ raise_with_ymm0(SIGUSR1, v);
+
+ if (sig_err_buf[0])
+ ksft_test_result_fail("%s\n", sig_err_buf);
+ else if (v[2] == TEST_YMMH_VAL && v[3] == (TEST_YMMH_VAL + 1))
+ ksft_test_result_pass("YMM state restored correctly from shrunk frame\n");
+ else
+ ksft_test_result_fail(
+ "Got upper bits: 0x%lx 0x%lx (expected %lx %lx)\n",
+ v[2], v[3], TEST_YMMH_VAL, TEST_YMMH_VAL + 1);
+
+ clearhandler(SIGUSR1);
+}
+
+static sigjmp_buf segv_jmpbuf;
+
+static void handle_segv(int sig, siginfo_t *si, void *ucp)
+{
+ siglongjmp(segv_jmpbuf, 1);
+}
+
+static void test_invalid_shrunk_xstate_size(void)
+{
+ uint64_t v[4];
+
+ sig_err_buf[0] = 0;
+ sethandler(SIGUSR1, handle_invalid_shrunk_xstate_size, 0);
+ sethandler(SIGSEGV, handle_segv, 0);
+
+ if (sigsetjmp(segv_jmpbuf, 1) == 0) {
+ v[0] = 0x1111111111111111ULL;
+ v[1] = 0x2222222222222222ULL;
+ v[2] = 0x3333333333333333ULL;
+ v[3] = 0x4444444444444444ULL;
+ raise_with_ymm0(SIGUSR1, v);
+ sig_print("Inconsistent size was NOT rejected\n");
+ }
+
+ clearhandler(SIGUSR1);
+ clearhandler(SIGSEGV);
+
+ if (sig_err_buf[0])
+ ksft_test_result_fail("%s\n", sig_err_buf);
+ else
+ ksft_test_result_pass("Inconsistent size correctly rejected\n");
+}
+
+int main(void)
+{
+ ksft_print_header();
+ ksft_set_plan(2);
+
+ self_pid = getpid();
+
+ check_avx_support();
+
+ test_valid_shrunk_xstate_size();
+ test_invalid_shrunk_xstate_size();
+
+ ksft_finished();
+ return 0;
+}
diff --git a/tools/testing/selftests/x86/xstate.c b/tools/testing/selftests/x86/xstate.c
index 97fe4bd8bc77..0ab577157cd7 100644
--- a/tools/testing/selftests/x86/xstate.c
+++ b/tools/testing/selftests/x86/xstate.c
@@ -34,18 +34,6 @@
(1 << XFEATURE_XTILEDATA) | \
(1 << XFEATURE_APX))
-static inline uint64_t xgetbv(uint32_t index)
-{
- uint32_t eax, edx;
-
- asm volatile("xgetbv" : "=a" (eax), "=d" (edx) : "c" (index));
- return eax + ((uint64_t)edx << 32);
-}
-
-static inline uint64_t get_xstatebv(struct xsave_buffer *xbuf)
-{
- return *(uint64_t *)(&xbuf->header);
-}
static struct xstate_info xstate;
diff --git a/tools/testing/selftests/x86/xstate.h b/tools/testing/selftests/x86/xstate.h
index 6ee816e7625a..eedf0cab7ccb 100644
--- a/tools/testing/selftests/x86/xstate.h
+++ b/tools/testing/selftests/x86/xstate.h
@@ -3,6 +3,8 @@
#define __SELFTESTS_X86_XSTATE_H
#include <stdint.h>
+#include <stdlib.h>
+#include <string.h>
#include "kselftest.h"
@@ -94,6 +96,14 @@ static inline void xrstor(struct xsave_buffer *xbuf, uint64_t rfbm)
: : "D" (xbuf), "a" (rfbm_lo), "d" (rfbm_hi));
}
+static inline uint64_t xgetbv(uint32_t index)
+{
+ uint32_t eax, edx;
+
+ asm volatile("xgetbv" : "=a" (eax), "=d" (edx) : "c" (index));
+ return eax + ((uint64_t)edx << 32);
+}
+
#define CPUID_LEAF_XSTATE 0xd
#define CPUID_SUBLEAF_XSTATE_USER 0x0
@@ -160,6 +170,11 @@ static inline void set_xstatebv(struct xsave_buffer *xbuf, uint64_t bv)
*(uint64_t *)(&xbuf->header) = bv;
}
+static inline uint64_t get_xstatebv(struct xsave_buffer *xbuf)
+{
+ return *(uint64_t *)(&xbuf->header);
+}
+
/* See 'struct _fpx_sw_bytes' at sigcontext.h */
#define SW_BYTES_OFFSET 464
/* N.B. The struct's field name varies so read from the offset. */
@@ -175,6 +190,11 @@ static inline uint64_t get_fpx_sw_bytes_features(void *buffer)
return *(uint64_t *)(buffer + SW_BYTES_BV_OFFSET);
}
+static inline void set_fpx_sw_bytes_features(void *buffer, uint64_t features)
+{
+ *(uint64_t *)(buffer + SW_BYTES_BV_OFFSET) = features;
+}
+
static inline void set_rand_data(struct xstate_info *xstate, struct xsave_buffer *xbuf)
{
int *ptr = (int *)&xbuf->bytes[xstate->xbuf_offset];